diff --git a/alembic/versions/0001_initial_unified_schema.py b/alembic/versions/0001_initial_unified_schema.py new file mode 100644 index 0000000..0580b45 --- /dev/null +++ b/alembic/versions/0001_initial_unified_schema.py @@ -0,0 +1,277 @@ +"""initial unified schema + +Revision ID: 0001 +Revises: +Create Date: 2026-05-13 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from pgvector.sqlalchemy import Vector + +revision: str = "0001" +down_revision: Union[str, None] = None +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.execute("CREATE EXTENSION IF NOT EXISTS vector") + + op.create_table( + "artist", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("name", sa.String(length=255), nullable=False), + sa.Column("slug", sa.String(length=255), nullable=False), + sa.Column("notes", sa.Text(), nullable=True), + sa.Column("is_subscription", sa.Boolean(), nullable=False, server_default=sa.false()), + sa.Column("auto_check", sa.Boolean(), nullable=False, server_default=sa.true()), + sa.Column("check_interval_seconds", sa.Integer(), nullable=True), + sa.Column( + "created_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.PrimaryKeyConstraint("id", name="pk_artist"), + sa.UniqueConstraint("name", name="uq_artist_name"), + sa.UniqueConstraint("slug", name="uq_artist_slug"), + ) + + op.create_table( + "source", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("artist_id", sa.Integer(), nullable=False), + sa.Column("platform", sa.String(length=64), nullable=False), + sa.Column("url", sa.Text(), nullable=False), + sa.Column("enabled", sa.Boolean(), nullable=False, server_default=sa.true()), + sa.Column("config_overrides", sa.JSON(), nullable=True), + sa.Column("last_checked_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("last_error", sa.Text(), nullable=True), + sa.Column("check_interval_override", sa.Integer(), nullable=True), + sa.ForeignKeyConstraint( + ["artist_id"], ["artist.id"], name="fk_source_artist_id_artist", ondelete="CASCADE" + ), + sa.PrimaryKeyConstraint("id", name="pk_source"), + ) + op.create_index("ix_source_artist_id", "source", ["artist_id"]) + + op.create_table( + "credential", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("platform", sa.String(length=64), nullable=False), + sa.Column("kind", sa.String(length=32), nullable=False), + sa.Column("encrypted_blob", sa.LargeBinary(), nullable=False), + sa.Column("status", sa.String(length=32), nullable=False, server_default="active"), + sa.Column( + "captured_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.Column("expires_at", sa.DateTime(timezone=True), nullable=True), + sa.PrimaryKeyConstraint("id", name="pk_credential"), + sa.UniqueConstraint("platform", name="uq_credential_platform"), + ) + + op.create_table( + "post", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("source_id", sa.Integer(), nullable=False), + sa.Column("external_post_id", sa.String(length=128), nullable=False), + sa.Column("post_url", sa.Text(), nullable=True), + sa.Column("post_title", sa.Text(), nullable=True), + sa.Column("post_date", sa.DateTime(timezone=True), nullable=True), + sa.Column("raw_metadata", sa.JSON(), nullable=True), + sa.Column( + "downloaded_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["source_id"], ["source.id"], name="fk_post_source_id_source", ondelete="CASCADE" + ), + sa.PrimaryKeyConstraint("id", name="pk_post"), + sa.UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"), + ) + op.create_index("ix_post_source_id", "post", ["source_id"]) + + op.create_table( + "image_record", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("path", sa.Text(), nullable=False), + sa.Column("sha256", sa.String(length=64), nullable=False), + sa.Column("phash", sa.String(length=32), nullable=True), + sa.Column("size_bytes", sa.BigInteger(), nullable=False), + sa.Column("mime", sa.String(length=64), nullable=False), + sa.Column("width", sa.Integer(), nullable=True), + sa.Column("height", sa.Integer(), nullable=True), + sa.Column("thumbnail_path", sa.Text(), nullable=True), + sa.Column( + "origin", + sa.Enum( + "downloaded", + "imported_filesystem", + "uploaded", + name="origin_enum", + ), + nullable=False, + ), + sa.Column("primary_post_id", sa.Integer(), nullable=True), + sa.Column("wd14_predictions", sa.JSON(), nullable=True), + sa.Column("wd14_model_version", sa.String(length=128), nullable=True), + sa.Column("siglip_embedding", Vector(1152), nullable=True), + sa.Column("siglip_model_version", sa.String(length=128), nullable=True), + sa.Column("centroid_scores", sa.JSON(), nullable=True), + sa.Column( + "created_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.Column( + "updated_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["primary_post_id"], + ["post.id"], + name="fk_image_record_primary_post_id_post", + ondelete="SET NULL", + ), + sa.PrimaryKeyConstraint("id", name="pk_image_record"), + sa.UniqueConstraint("path", name="uq_image_record_path"), + sa.UniqueConstraint("sha256", name="uq_image_record_sha256"), + ) + op.create_index("ix_image_record_sha256", "image_record", ["sha256"]) + op.create_index("ix_image_record_phash", "image_record", ["phash"]) + op.create_index("ix_image_record_primary_post_id", "image_record", ["primary_post_id"]) + + op.create_table( + "image_provenance", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("image_record_id", sa.Integer(), nullable=False), + sa.Column("post_id", sa.Integer(), nullable=False), + sa.Column("source_id", sa.Integer(), nullable=False), + sa.Column("captured_metadata", sa.JSON(), nullable=True), + sa.Column( + "captured_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["image_record_id"], + ["image_record.id"], + name="fk_image_provenance_image_record_id_image_record", + ondelete="CASCADE", + ), + sa.ForeignKeyConstraint( + ["post_id"], + ["post.id"], + name="fk_image_provenance_post_id_post", + ondelete="CASCADE", + ), + sa.ForeignKeyConstraint( + ["source_id"], + ["source.id"], + name="fk_image_provenance_source_id_source", + ondelete="CASCADE", + ), + sa.PrimaryKeyConstraint("id", name="pk_image_provenance"), + ) + op.create_index("ix_image_provenance_image_record_id", "image_provenance", ["image_record_id"]) + op.create_index("ix_image_provenance_post_id", "image_provenance", ["post_id"]) + op.create_index("ix_image_provenance_source_id", "image_provenance", ["source_id"]) + + op.create_table( + "tag", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("name", sa.String(length=255), nullable=False), + sa.Column("namespace", sa.String(length=64), nullable=True), + sa.Column( + "created_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.PrimaryKeyConstraint("id", name="pk_tag"), + sa.UniqueConstraint("name", name="uq_tag_name"), + ) + op.create_index("ix_tag_name", "tag", ["name"]) + op.create_index("ix_tag_namespace", "tag", ["namespace"]) + + op.create_table( + "image_tag", + sa.Column("image_record_id", sa.Integer(), nullable=False), + sa.Column("tag_id", sa.Integer(), nullable=False), + sa.Column("source", sa.String(length=32), nullable=False, server_default="manual"), + sa.Column( + "created_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["image_record_id"], + ["image_record.id"], + name="fk_image_tag_image_record_id_image_record", + ondelete="CASCADE", + ), + sa.ForeignKeyConstraint( + ["tag_id"], ["tag.id"], name="fk_image_tag_tag_id_tag", ondelete="CASCADE" + ), + sa.PrimaryKeyConstraint("image_record_id", "tag_id", name="pk_image_tag"), + ) + + op.create_table( + "download_event", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("source_id", sa.Integer(), nullable=False), + sa.Column("post_id", sa.Integer(), nullable=True), + sa.Column("status", sa.String(length=32), nullable=False), + sa.Column( + "started_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("bytes_downloaded", sa.BigInteger(), nullable=False, server_default="0"), + sa.Column("files_count", sa.Integer(), nullable=False, server_default="0"), + sa.Column("error", sa.Text(), nullable=True), + sa.ForeignKeyConstraint( + ["source_id"], + ["source.id"], + name="fk_download_event_source_id_source", + ondelete="CASCADE", + ), + sa.ForeignKeyConstraint( + ["post_id"], + ["post.id"], + name="fk_download_event_post_id_post", + ondelete="SET NULL", + ), + sa.PrimaryKeyConstraint("id", name="pk_download_event"), + ) + op.create_index("ix_download_event_source_id", "download_event", ["source_id"]) + op.create_index("ix_download_event_post_id", "download_event", ["post_id"]) + + +def downgrade() -> None: + op.drop_table("download_event") + op.drop_table("image_tag") + op.drop_table("tag") + op.drop_table("image_provenance") + op.drop_table("image_record") + op.execute("DROP TYPE IF EXISTS origin_enum") + op.drop_table("post") + op.drop_table("credential") + op.drop_table("source") + op.drop_table("artist") + op.execute("DROP EXTENSION IF EXISTS vector") diff --git a/alembic/versions/0002_fc2a_tag_kinds_and_import_tasks.py b/alembic/versions/0002_fc2a_tag_kinds_and_import_tasks.py new file mode 100644 index 0000000..b9dac2c --- /dev/null +++ b/alembic/versions/0002_fc2a_tag_kinds_and_import_tasks.py @@ -0,0 +1,208 @@ +"""fc2a: tag kinds, import_task, import_batch, integrity_status + +Revision ID: 0002 +Revises: 0001 +Create Date: 2026-05-14 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0002" +down_revision: Union[str, None] = "0001" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +TAG_KINDS = ( + "artist", + "character", + "fandom", + "general", + "series", + "archive", + "post", + "meta", + "rating", +) + + +def upgrade() -> None: + # --- Tag kind enum + fandom_id --- + tag_kind = sa.Enum(*TAG_KINDS, name="tag_kind") + tag_kind.create(op.get_bind(), checkfirst=True) + + op.add_column( + "tag", + sa.Column("kind", tag_kind, nullable=False, server_default="general"), + ) + op.add_column( + "tag", + sa.Column("fandom_id", sa.Integer(), nullable=True), + ) + op.create_foreign_key( + "fk_tag_fandom_id_tag", + "tag", + "tag", + ["fandom_id"], + ["id"], + ondelete="SET NULL", + ) + + # Drop the old global uniqueness on name; add kind+fandom-aware uniqueness. + op.drop_constraint("uq_tag_name", "tag", type_="unique") + op.drop_index("ix_tag_name", table_name="tag") + op.execute( + """ + CREATE UNIQUE INDEX uq_tag_name_kind_fandom + ON tag (name, kind, COALESCE(fandom_id, 0)) + """ + ) + + # CHECK: fandom_id is only allowed for character kind. + op.create_check_constraint( + "ck_tag_fandom_requires_character", + "tag", + "(fandom_id IS NULL) OR (kind = 'character')", + ) + + # Drop the old namespace column — superseded by kind. + op.drop_index("ix_tag_namespace", table_name="tag") + op.drop_column("tag", "namespace") + + # --- ImportBatch --- + op.create_table( + "import_batch", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("triggered_by", sa.String(length=32), nullable=False), + sa.Column("source_path", sa.Text(), nullable=False), + sa.Column("scan_mode", sa.String(length=16), nullable=False), + sa.Column( + "started_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("total_files", sa.Integer(), nullable=False, server_default="0"), + sa.Column("imported", sa.Integer(), nullable=False, server_default="0"), + sa.Column("skipped", sa.Integer(), nullable=False, server_default="0"), + sa.Column("failed", sa.Integer(), nullable=False, server_default="0"), + sa.Column("status", sa.String(length=16), nullable=False, server_default="running"), + sa.PrimaryKeyConstraint("id", name="pk_import_batch"), + ) + op.create_index("ix_import_batch_status", "import_batch", ["status"]) + + # --- ImportTask --- + op.create_table( + "import_task", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("batch_id", sa.Integer(), nullable=False), + sa.Column("source_path", sa.Text(), nullable=False), + sa.Column("task_type", sa.String(length=16), nullable=False), + sa.Column("status", sa.String(length=16), nullable=False, server_default="pending"), + sa.Column("result_image_id", sa.Integer(), nullable=True), + sa.Column("error", sa.Text(), nullable=True), + sa.Column("size_bytes", sa.BigInteger(), nullable=True), + sa.Column( + "created_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.func.now(), + ), + sa.Column("started_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.ForeignKeyConstraint( + ["batch_id"], + ["import_batch.id"], + name="fk_import_task_batch_id_import_batch", + ondelete="CASCADE", + ), + sa.ForeignKeyConstraint( + ["result_image_id"], + ["image_record.id"], + name="fk_import_task_result_image_id_image_record", + ondelete="SET NULL", + ), + sa.PrimaryKeyConstraint("id", name="pk_import_task"), + ) + op.create_index("ix_import_task_batch_id", "import_task", ["batch_id"]) + op.create_index("ix_import_task_status", "import_task", ["status"]) + op.create_index( + "ix_import_task_created_at_desc", + "import_task", + [sa.text("created_at DESC")], + ) + + # --- ImportSettings (single-row table) --- + op.create_table( + "import_settings", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("import_scan_path", sa.Text(), nullable=False, server_default="/import"), + sa.Column("min_width", sa.Integer(), nullable=False, server_default="0"), + sa.Column("min_height", sa.Integer(), nullable=False, server_default="0"), + sa.Column( + "skip_transparent", sa.Boolean(), nullable=False, server_default=sa.false() + ), + sa.Column( + "transparency_threshold", + sa.Float(), + nullable=False, + server_default="0.9", + ), + sa.Column( + "skip_single_color", sa.Boolean(), nullable=False, server_default=sa.false() + ), + sa.Column( + "single_color_threshold", + sa.Float(), + nullable=False, + server_default="0.95", + ), + sa.Column("single_color_tolerance", sa.Integer(), nullable=False, server_default="30"), + sa.PrimaryKeyConstraint("id", name="pk_import_settings"), + sa.CheckConstraint("id = 1", name="ck_import_settings_singleton"), + ) + # Seed the single row immediately so callers can always SELECT id=1. + op.execute("INSERT INTO import_settings (id) VALUES (1)") + + # --- ImageRecord additions --- + op.add_column( + "image_record", + sa.Column( + "integrity_status", + sa.String(length=24), + nullable=False, + server_default="unknown", + ), + ) + op.create_index( + "ix_image_record_integrity_status", + "image_record", + ["integrity_status"], + ) + + +def downgrade() -> None: + op.drop_index("ix_image_record_integrity_status", table_name="image_record") + op.drop_column("image_record", "integrity_status") + + op.drop_table("import_settings") + op.drop_index("ix_import_task_created_at_desc", table_name="import_task") + op.drop_index("ix_import_task_status", table_name="import_task") + op.drop_index("ix_import_task_batch_id", table_name="import_task") + op.drop_table("import_task") + op.drop_index("ix_import_batch_status", table_name="import_batch") + op.drop_table("import_batch") + + op.drop_constraint("ck_tag_fandom_requires_character", "tag", type_="check") + op.execute("DROP INDEX uq_tag_name_kind_fandom") + op.add_column("tag", sa.Column("namespace", sa.String(length=64), nullable=True)) + op.create_index("ix_tag_namespace", "tag", ["namespace"]) + op.create_index("ix_tag_name", "tag", ["name"], unique=False) + op.create_unique_constraint("uq_tag_name", "tag", ["name"]) + op.drop_constraint("fk_tag_fandom_id_tag", "tag", type_="foreignkey") + op.drop_column("tag", "fandom_id") + op.drop_column("tag", "kind") + sa.Enum(name="tag_kind").drop(op.get_bind(), checkfirst=True) diff --git a/alembic/versions/0003_fc2b_ml_pipeline.py b/alembic/versions/0003_fc2b_ml_pipeline.py new file mode 100644 index 0000000..584bffe --- /dev/null +++ b/alembic/versions/0003_fc2b_ml_pipeline.py @@ -0,0 +1,172 @@ +"""fc2b: ML pipeline — allowlist, aliases, centroids, ml_settings + +Revision ID: 0003 +Revises: 0002 +Create Date: 2026-05-15 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from pgvector.sqlalchemy import Vector + +revision: str = "0003" +down_revision: Union[str, None] = "0002" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # 3.1 rename wd14_* -> tagger_* + op.alter_column("image_record", "wd14_predictions", new_column_name="tagger_predictions") + op.alter_column( + "image_record", "wd14_model_version", new_column_name="tagger_model_version" + ) + + # 3.2 tag_allowlist + op.create_table( + "tag_allowlist", + sa.Column("tag_id", sa.Integer(), nullable=False), + sa.Column( + "min_confidence", sa.Float(), nullable=False, server_default="0.95" + ), + sa.Column( + "added_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["tag_id"], ["tag.id"], name="fk_tag_allowlist_tag_id_tag", + ondelete="CASCADE", + ), + sa.PrimaryKeyConstraint("tag_id", name="pk_tag_allowlist"), + sa.CheckConstraint( + "min_confidence > 0 AND min_confidence <= 1", + name="ck_tag_allowlist_confidence_range", + ), + ) + + # 3.3 tag_suggestion_rejection + op.create_table( + "tag_suggestion_rejection", + sa.Column("image_record_id", sa.Integer(), nullable=False), + sa.Column("tag_id", sa.Integer(), nullable=False), + sa.Column( + "rejected_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["image_record_id"], ["image_record.id"], + name="fk_tsr_image_record_id_image_record", ondelete="CASCADE", + ), + sa.ForeignKeyConstraint( + ["tag_id"], ["tag.id"], name="fk_tsr_tag_id_tag", ondelete="CASCADE", + ), + sa.PrimaryKeyConstraint( + "image_record_id", "tag_id", name="pk_tag_suggestion_rejection" + ), + ) + op.create_index( + "ix_tag_suggestion_rejection_tag", "tag_suggestion_rejection", ["tag_id"] + ) + + # 3.4 tag_alias + op.create_table( + "tag_alias", + sa.Column("alias_string", sa.String(length=255), nullable=False), + sa.Column("alias_category", sa.String(length=32), nullable=False), + sa.Column("canonical_tag_id", sa.Integer(), nullable=False), + sa.Column( + "created_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["canonical_tag_id"], ["tag.id"], + name="fk_tag_alias_canonical_tag_id_tag", ondelete="CASCADE", + ), + sa.PrimaryKeyConstraint( + "alias_string", "alias_category", name="pk_tag_alias" + ), + ) + op.create_index("ix_tag_alias_canonical", "tag_alias", ["canonical_tag_id"]) + + # 3.5 tag_reference_embedding (centroids) + op.create_table( + "tag_reference_embedding", + sa.Column("tag_id", sa.Integer(), nullable=False), + sa.Column("embedding", Vector(1152), nullable=False), + sa.Column("reference_count", sa.Integer(), nullable=False), + sa.Column("model_version", sa.String(length=128), nullable=False), + sa.Column( + "updated_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["tag_id"], ["tag.id"], + name="fk_tag_reference_embedding_tag_id_tag", ondelete="CASCADE", + ), + sa.PrimaryKeyConstraint("tag_id", name="pk_tag_reference_embedding"), + ) + + # 3.6 ml_settings singleton + op.create_table( + "ml_settings", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column( + "suggestion_threshold_artist", sa.Float(), nullable=False, + server_default="0.30", + ), + sa.Column( + "suggestion_threshold_character", sa.Float(), nullable=False, + server_default="0.50", + ), + sa.Column( + "suggestion_threshold_copyright", sa.Float(), nullable=False, + server_default="0.50", + ), + sa.Column( + "suggestion_threshold_general", sa.Float(), nullable=False, + server_default="0.95", + ), + sa.Column( + "centroid_similarity_threshold", sa.Float(), nullable=False, + server_default="0.55", + ), + sa.Column( + "min_reference_images", sa.Integer(), nullable=False, + server_default="5", + ), + sa.Column( + "tagger_model_version", sa.String(length=128), nullable=False, + server_default="camie-tagger-v2", + ), + sa.Column( + "embedder_model_version", sa.String(length=128), nullable=False, + server_default="siglip-so400m-patch14-384", + ), + sa.Column( + "updated_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.PrimaryKeyConstraint("id", name="pk_ml_settings"), + sa.CheckConstraint("id = 1", name="ck_ml_settings_singleton"), + ) + op.execute("INSERT INTO ml_settings (id) VALUES (1)") + + +def downgrade() -> None: + op.drop_table("ml_settings") + op.drop_table("tag_reference_embedding") + op.drop_index("ix_tag_alias_canonical", table_name="tag_alias") + op.drop_table("tag_alias") + op.drop_index( + "ix_tag_suggestion_rejection_tag", table_name="tag_suggestion_rejection" + ) + op.drop_table("tag_suggestion_rejection") + op.drop_table("tag_allowlist") + op.alter_column( + "image_record", "tagger_model_version", new_column_name="wd14_model_version" + ) + op.alter_column( + "image_record", "tagger_predictions", new_column_name="wd14_predictions" + ) diff --git a/alembic/versions/0004_fc2c_i_tsm_system_rows.py b/alembic/versions/0004_fc2c_i_tsm_system_rows.py new file mode 100644 index 0000000..e8bd920 --- /dev/null +++ b/alembic/versions/0004_fc2c_i_tsm_system_rows.py @@ -0,0 +1,23 @@ +"""fc2c-i: enable tsm_system_rows for scalable random sampling + +Revision ID: 0004 +Revises: 0003 +Create Date: 2026-05-15 + +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0004" +down_revision: Union[str, None] = "0003" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows") + + +def downgrade() -> None: + op.execute("DROP EXTENSION IF EXISTS tsm_system_rows") diff --git a/alembic/versions/0005_fc2c_iii_a_series_page.py b/alembic/versions/0005_fc2c_iii_a_series_page.py new file mode 100644 index 0000000..ffe397e --- /dev/null +++ b/alembic/versions/0005_fc2c_iii_a_series_page.py @@ -0,0 +1,50 @@ +"""fc2c-iii-a: series_page ordered membership + +Revision ID: 0005 +Revises: 0004 +Create Date: 2026-05-16 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0005" +down_revision: Union[str, None] = "0004" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "series_page", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("series_tag_id", sa.Integer(), nullable=False), + sa.Column("image_id", sa.Integer(), nullable=False), + sa.Column("page_number", sa.Integer(), nullable=False), + sa.Column( + "created_at", sa.DateTime(timezone=True), + nullable=False, server_default=sa.func.now(), + ), + sa.Column( + "updated_at", sa.DateTime(timezone=True), + nullable=False, server_default=sa.func.now(), + ), + sa.ForeignKeyConstraint( + ["series_tag_id"], ["tag.id"], ondelete="CASCADE" + ), + sa.ForeignKeyConstraint( + ["image_id"], ["image_record.id"], ondelete="CASCADE" + ), + sa.PrimaryKeyConstraint("id"), + sa.UniqueConstraint("image_id", name="uq_series_page_image"), + ) + op.create_index( + "ix_series_page_series_tag_id", "series_page", ["series_tag_id"] + ) + + +def downgrade() -> None: + op.drop_index("ix_series_page_series_tag_id", table_name="series_page") + op.drop_table("series_page") diff --git a/alembic/versions/0006_fc2d_phash_threshold.py b/alembic/versions/0006_fc2d_phash_threshold.py new file mode 100644 index 0000000..895ed46 --- /dev/null +++ b/alembic/versions/0006_fc2d_phash_threshold.py @@ -0,0 +1,30 @@ +"""fc2d: import_settings.phash_threshold + +Revision ID: 0006 +Revises: 0005 +Create Date: 2026-05-17 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0006" +down_revision: Union[str, None] = "0005" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "import_settings", + sa.Column( + "phash_threshold", sa.Integer(), + nullable=False, server_default="10", + ), + ) + + +def downgrade() -> None: + op.drop_column("import_settings", "phash_threshold") diff --git a/alembic/versions/0007_fc2d_post_metadata_fields.py b/alembic/versions/0007_fc2d_post_metadata_fields.py new file mode 100644 index 0000000..24e8ba9 --- /dev/null +++ b/alembic/versions/0007_fc2d_post_metadata_fields.py @@ -0,0 +1,31 @@ +"""fc2d-iv: post.description + post.attachment_count + +Revision ID: 0007 +Revises: 0006 +Create Date: 2026-05-18 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0007" +down_revision: Union[str, None] = "0006" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "post", sa.Column("description", sa.Text(), nullable=True) + ) + op.add_column( + "post", + sa.Column("attachment_count", sa.Integer(), nullable=True), + ) + + +def downgrade() -> None: + op.drop_column("post", "attachment_count") + op.drop_column("post", "description") diff --git a/alembic/versions/0008_fc2d_vii_c_artist_deconfliction.py b/alembic/versions/0008_fc2d_vii_c_artist_deconfliction.py new file mode 100644 index 0000000..4019e2d --- /dev/null +++ b/alembic/versions/0008_fc2d_vii_c_artist_deconfliction.py @@ -0,0 +1,52 @@ +"""fc2d-vii-c: image_record.artist_id + backfill + drop artist tags + +Revision ID: 0008 +Revises: 0007 +Create Date: 2026-05-18 + +Internal forward-correctness migration (the big legacy-import migration +stays deferred). downgrade() does NOT recreate deleted artist tags; +downgrade is dev-only and the data is reconstructable by re-import. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +from backend.app.utils.artist_backfill import ( + BACKFILL_PRIMARY_SQL, + BACKFILL_PROVENANCE_SQL, + BACKFILL_TAG_SQL, + DELETE_ARTIST_TAGS_SQL, +) + +revision: str = "0008" +down_revision: Union[str, None] = "0007" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "image_record", + sa.Column("artist_id", sa.Integer(), nullable=True), + ) + op.create_foreign_key( + "fk_image_record_artist_id", "image_record", "artist", + ["artist_id"], ["id"], ondelete="SET NULL", + ) + op.create_index( + "ix_image_record_artist_id", "image_record", ["artist_id"], + ) + op.execute(BACKFILL_PRIMARY_SQL) + op.execute(BACKFILL_PROVENANCE_SQL) + op.execute(BACKFILL_TAG_SQL) + op.execute(DELETE_ARTIST_TAGS_SQL) + + +def downgrade() -> None: + op.drop_index("ix_image_record_artist_id", table_name="image_record") + op.drop_constraint( + "fk_image_record_artist_id", "image_record", type_="foreignkey" + ) + op.drop_column("image_record", "artist_id") diff --git a/alembic/versions/0009_fc2d_iii_post_attachment.py b/alembic/versions/0009_fc2d_iii_post_attachment.py new file mode 100644 index 0000000..4820fa4 --- /dev/null +++ b/alembic/versions/0009_fc2d_iii_post_attachment.py @@ -0,0 +1,68 @@ +"""fc2d-iii: post_attachment + import_batch.attachments + +Revision ID: 0009 +Revises: 0008 +Create Date: 2026-05-19 + +Internal forward-correctness migration (big legacy-import migration +stays deferred). No backfill — no attachments exist yet. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0009" +down_revision: Union[str, None] = "0008" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "post_attachment", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "post_id", sa.Integer(), + sa.ForeignKey("post.id", ondelete="SET NULL"), nullable=True, + ), + sa.Column( + "artist_id", sa.Integer(), + sa.ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, + ), + sa.Column("sha256", sa.String(64), nullable=False), + sa.Column("path", sa.Text(), nullable=False), + sa.Column("original_filename", sa.Text(), nullable=False), + sa.Column("ext", sa.String(32), nullable=False), + sa.Column("mime", sa.String(128), nullable=True), + sa.Column("size_bytes", sa.BigInteger(), nullable=False), + sa.Column( + "captured_at", sa.DateTime(timezone=True), + server_default=sa.func.now(), nullable=False, + ), + ) + op.create_index( + "ix_post_attachment_sha256", "post_attachment", ["sha256"], + unique=True, + ) + op.create_index( + "ix_post_attachment_post_id", "post_attachment", ["post_id"], + ) + op.create_index( + "ix_post_attachment_artist_id", "post_attachment", ["artist_id"], + ) + op.add_column( + "import_batch", + sa.Column( + "attachments", sa.Integer(), nullable=False, + server_default="0", + ), + ) + + +def downgrade() -> None: + op.drop_column("import_batch", "attachments") + op.drop_index("ix_post_attachment_artist_id", table_name="post_attachment") + op.drop_index("ix_post_attachment_post_id", table_name="post_attachment") + op.drop_index("ix_post_attachment_sha256", table_name="post_attachment") + op.drop_table("post_attachment") diff --git a/alembic/versions/0010_fc3a_source_unique_artist_platform_url.py b/alembic/versions/0010_fc3a_source_unique_artist_platform_url.py new file mode 100644 index 0000000..10f502b --- /dev/null +++ b/alembic/versions/0010_fc3a_source_unique_artist_platform_url.py @@ -0,0 +1,32 @@ +"""fc3a: unique(source.artist_id, source.platform, source.url) + +Revision ID: 0010 +Revises: 0009 +Create Date: 2026-05-20 + +Enforces FC-3a's dedup invariant at the DB level. No backfill — no +existing rows are expected to collide; if they do the migration will +fail loudly (intended). +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0010" +down_revision: Union[str, None] = "0009" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_unique_constraint( + "uq_source_artist_platform_url", + "source", + ["artist_id", "platform", "url"], + ) + + +def downgrade() -> None: + op.drop_constraint( + "uq_source_artist_platform_url", "source", type_="unique" + ) diff --git a/alembic/versions/0011_fc3b_credential_schema_alignment.py b/alembic/versions/0011_fc3b_credential_schema_alignment.py new file mode 100644 index 0000000..00a31a5 --- /dev/null +++ b/alembic/versions/0011_fc3b_credential_schema_alignment.py @@ -0,0 +1,41 @@ +"""fc3b: rename credential.kind -> credential_type, drop status, add last_verified + +Revision ID: 0011 +Revises: 0010 +Create Date: 2026-05-20 + +Aligns the credential table with the GallerySubscriber wire-field names +so the existing browser extension can POST to FC unmodified. Greenfield — +no rows exist in production yet, so no data preservation logic is +needed; the rename uses ALTER COLUMN rather than copy-then-drop. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0011" +down_revision: Union[str, None] = "0010" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.alter_column("credential", "kind", new_column_name="credential_type") + op.drop_column("credential", "status") + op.add_column( + "credential", + sa.Column("last_verified", sa.DateTime(timezone=True), nullable=True), + ) + + +def downgrade() -> None: + op.drop_column("credential", "last_verified") + op.add_column( + "credential", + sa.Column( + "status", sa.String(length=32), nullable=False, + server_default="active", + ), + ) + op.alter_column("credential", "credential_type", new_column_name="kind") diff --git a/alembic/versions/0012_fc3b_app_setting.py b/alembic/versions/0012_fc3b_app_setting.py new file mode 100644 index 0000000..42c06eb --- /dev/null +++ b/alembic/versions/0012_fc3b_app_setting.py @@ -0,0 +1,36 @@ +"""fc3b: app_setting key/value table + +Revision ID: 0012 +Revises: 0011 +Create Date: 2026-05-20 + +A simple key/value table for small app settings that don't fit +ImportSettings. Initially seeds only `extension_api_key` (done in +create_app on first boot — not in the migration, to keep it +deterministic and independent of randomness). +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0012" +down_revision: Union[str, None] = "0011" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "app_setting", + sa.Column("key", sa.String(length=64), primary_key=True), + sa.Column("value", sa.Text(), nullable=False), + sa.Column( + "updated_at", sa.DateTime(timezone=True), + nullable=False, server_default=sa.func.now(), + ), + ) + + +def downgrade() -> None: + op.drop_table("app_setting") diff --git a/alembic/versions/0013_fc3c_download_event_metadata.py b/alembic/versions/0013_fc3c_download_event_metadata.py new file mode 100644 index 0000000..c88b3f9 --- /dev/null +++ b/alembic/versions/0013_fc3c_download_event_metadata.py @@ -0,0 +1,52 @@ +"""fc3c: download_event.metadata + import_settings downloader fields + +Revision ID: 0013 +Revises: 0012 +Create Date: 2026-05-20 + +Additive only. download_event.metadata is the rich JSONB blob FC-3c +populates per run (run_stats, stdout/stderr, quarantined paths, import +summary). import_settings gains two operator-tunable downloader knobs: +download_rate_limit_seconds (gallery-dl extractor.sleep) and +download_validate_files (toggle the magic-byte validator). +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from sqlalchemy.dialects import postgresql + +revision: str = "0013" +down_revision: Union[str, None] = "0012" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "download_event", + sa.Column( + "metadata", postgresql.JSONB, + nullable=False, server_default=sa.text("'{}'::jsonb"), + ), + ) + op.add_column( + "import_settings", + sa.Column( + "download_rate_limit_seconds", sa.Float(), + nullable=False, server_default="3.0", + ), + ) + op.add_column( + "import_settings", + sa.Column( + "download_validate_files", sa.Boolean(), + nullable=False, server_default=sa.true(), + ), + ) + + +def downgrade() -> None: + op.drop_column("import_settings", "download_validate_files") + op.drop_column("import_settings", "download_rate_limit_seconds") + op.drop_column("download_event", "metadata") diff --git a/alembic/versions/0014_fc3d_scheduling.py b/alembic/versions/0014_fc3d_scheduling.py new file mode 100644 index 0000000..955e956 --- /dev/null +++ b/alembic/versions/0014_fc3d_scheduling.py @@ -0,0 +1,58 @@ +"""fc3d: scheduling + source health columns + +Revision ID: 0014 +Revises: 0013 +Create Date: 2026-05-21 + +Additive only. source.consecutive_failures (default 0, DownloadService +finalize hook owns the writes). import_settings gains the three +scheduling knobs (global default interval, event retention, failure +warning threshold). +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0014" +down_revision: Union[str, None] = "0013" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "source", + sa.Column( + "consecutive_failures", sa.Integer(), + nullable=False, server_default="0", + ), + ) + op.add_column( + "import_settings", + sa.Column( + "download_schedule_default_seconds", sa.Integer(), + nullable=False, server_default="28800", + ), + ) + op.add_column( + "import_settings", + sa.Column( + "download_event_retention_days", sa.Integer(), + nullable=False, server_default="90", + ), + ) + op.add_column( + "import_settings", + sa.Column( + "download_failure_warning_threshold", sa.Integer(), + nullable=False, server_default="5", + ), + ) + + +def downgrade() -> None: + op.drop_column("import_settings", "download_failure_warning_threshold") + op.drop_column("import_settings", "download_event_retention_days") + op.drop_column("import_settings", "download_schedule_default_seconds") + op.drop_column("source", "consecutive_failures") diff --git a/alembic/versions/0015_fc5_migration_run.py b/alembic/versions/0015_fc5_migration_run.py new file mode 100644 index 0000000..89d7d89 --- /dev/null +++ b/alembic/versions/0015_fc5_migration_run.py @@ -0,0 +1,51 @@ +"""fc5: migration_run table + +Revision ID: 0015 +Revises: 0014 +Create Date: 2026-05-22 + +Additive only. New table tracks each invocation of the FC-5 migration +tooling (backup, gs, ir, ml_queue, verify, rollback). kind/status are +plain String(32) — values validated at the API layer per the spec, not +a Postgres ENUM (so adding kinds later doesn't need a schema migration). +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from sqlalchemy.dialects import postgresql + +revision: str = "0015" +down_revision: Union[str, None] = "0014" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "migration_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("kind", sa.String(32), nullable=False, index=True), + sa.Column("status", sa.String(32), nullable=False, index=True), + sa.Column( + "dry_run", sa.Boolean(), nullable=False, server_default=sa.false(), + ), + sa.Column( + "started_at", sa.DateTime(timezone=True), + nullable=False, server_default=sa.func.now(), + ), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column( + "counts", postgresql.JSONB, + nullable=False, server_default=sa.text("'{}'::jsonb"), + ), + sa.Column("error", sa.Text(), nullable=True), + sa.Column( + "metadata", postgresql.JSONB, + nullable=False, server_default=sa.text("'{}'::jsonb"), + ), + ) + + +def downgrade() -> None: + op.drop_table("migration_run") diff --git a/alembic/versions/0016_fc3i_task_run.py b/alembic/versions/0016_fc3i_task_run.py new file mode 100644 index 0000000..678b1ee --- /dev/null +++ b/alembic/versions/0016_fc3i_task_run.py @@ -0,0 +1,86 @@ +"""fc3i: task_run table + +Revision ID: 0016 +Revises: 0015 +Create Date: 2026-05-24 + +Additive only. New table records every Celery task attempt via signal +handlers (backend.app.celery_signals). Status is plain String(16) not +Postgres ENUM (per feedback_check_existing_enums: ENUM columns hard- +fail at INSERT, String columns extend cleanly). + +Composite indexes anticipate the three dashboard panes: +- (queue, started_at desc) — per-lane recent activity +- (status, started_at desc) — recent failures pane +- (task_name, started_at desc) — drill-down by task + +Indexed columns get individual indexes via `index=True` on the model; +the composites below cover the multi-column lookups. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0016" +down_revision: Union[str, None] = "0015" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "task_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("celery_task_id", sa.String(length=64), nullable=False), + sa.Column("queue", sa.String(length=32), nullable=False), + sa.Column("task_name", sa.String(length=128), nullable=False), + sa.Column("target_id", sa.Integer(), nullable=True), + sa.Column("started_at", sa.DateTime(timezone=True), nullable=False), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("duration_ms", sa.Integer(), nullable=True), + sa.Column( + "status", sa.String(length=16), nullable=False, + server_default="running", + ), + sa.Column("error_type", sa.String(length=128), nullable=True), + sa.Column("error_message", sa.Text(), nullable=True), + sa.Column("retry_count", sa.Integer(), nullable=True), + sa.Column("worker_hostname", sa.String(length=128), nullable=True), + sa.Column("args_summary", sa.String(length=255), nullable=True), + ) + + # Single-column indexes (matches Mapped[...].index=True on model). + op.create_index("ix_task_run_celery_task_id", "task_run", ["celery_task_id"]) + op.create_index("ix_task_run_queue", "task_run", ["queue"]) + op.create_index("ix_task_run_task_name", "task_run", ["task_name"]) + op.create_index("ix_task_run_started_at", "task_run", ["started_at"]) + op.create_index("ix_task_run_finished_at", "task_run", ["finished_at"]) + op.create_index("ix_task_run_status", "task_run", ["status"]) + + # Composite indexes for dashboard query patterns. + op.create_index( + "ix_task_run_queue_started", + "task_run", ["queue", sa.text("started_at DESC")], + ) + op.create_index( + "ix_task_run_status_started", + "task_run", ["status", sa.text("started_at DESC")], + ) + op.create_index( + "ix_task_run_name_started", + "task_run", ["task_name", sa.text("started_at DESC")], + ) + + +def downgrade() -> None: + op.drop_index("ix_task_run_name_started", table_name="task_run") + op.drop_index("ix_task_run_status_started", table_name="task_run") + op.drop_index("ix_task_run_queue_started", table_name="task_run") + op.drop_index("ix_task_run_status", table_name="task_run") + op.drop_index("ix_task_run_finished_at", table_name="task_run") + op.drop_index("ix_task_run_started_at", table_name="task_run") + op.drop_index("ix_task_run_task_name", table_name="task_run") + op.drop_index("ix_task_run_queue", table_name="task_run") + op.drop_index("ix_task_run_celery_task_id", table_name="task_run") + op.drop_table("task_run") diff --git a/alembic/versions/0017_fc3h_backup_run.py b/alembic/versions/0017_fc3h_backup_run.py new file mode 100644 index 0000000..5b6a839 --- /dev/null +++ b/alembic/versions/0017_fc3h_backup_run.py @@ -0,0 +1,82 @@ +"""fc3h: backup_run table + +Revision ID: 0017 +Revises: 0016 +Create Date: 2026-05-24 + +Additive. New table records every backup/restore attempt with artifact +metadata. Lifecycle tracking lives in task_run from FC-3i; this is +artifact-only (paths, sizes, tag, restore lineage). +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0017" +down_revision: Union[str, None] = "0016" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "backup_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("kind", sa.String(length=16), nullable=False), + sa.Column( + "status", sa.String(length=16), nullable=False, + server_default="pending", + ), + sa.Column("tag", sa.String(length=64), nullable=True), + sa.Column("triggered_by", sa.String(length=32), nullable=False), + sa.Column("started_at", sa.DateTime(timezone=True), nullable=False), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("sql_path", sa.Text(), nullable=True), + sa.Column("tar_path", sa.Text(), nullable=True), + sa.Column("size_bytes", sa.BigInteger(), nullable=True), + sa.Column("error", sa.Text(), nullable=True), + sa.Column( + "manifest", sa.JSON(), nullable=False, server_default="{}", + ), + sa.Column( + "restored_from_id", sa.Integer(), + sa.ForeignKey("backup_run.id", ondelete="SET NULL"), + nullable=True, + ), + ) + + # Single-column indexes (matches Mapped[...].index=True). + op.create_index("ix_backup_run_kind", "backup_run", ["kind"]) + op.create_index("ix_backup_run_status", "backup_run", ["status"]) + op.create_index("ix_backup_run_tag", "backup_run", ["tag"]) + op.create_index("ix_backup_run_started_at", "backup_run", ["started_at"]) + op.create_index("ix_backup_run_finished_at", "backup_run", ["finished_at"]) + + # Composite indexes for dashboard query patterns. + op.create_index( + "ix_backup_run_kind_started", + "backup_run", ["kind", sa.text("started_at DESC")], + ) + op.create_index( + "ix_backup_run_status_finished", + "backup_run", ["status", sa.text("finished_at DESC")], + ) + # Partial index: only tagged rows participate in retention-exempt query. + op.create_index( + "ix_backup_run_tag_partial", + "backup_run", ["tag"], + postgresql_where=sa.text("tag IS NOT NULL"), + ) + + +def downgrade() -> None: + op.drop_index("ix_backup_run_tag_partial", table_name="backup_run") + op.drop_index("ix_backup_run_status_finished", table_name="backup_run") + op.drop_index("ix_backup_run_kind_started", table_name="backup_run") + op.drop_index("ix_backup_run_finished_at", table_name="backup_run") + op.drop_index("ix_backup_run_started_at", table_name="backup_run") + op.drop_index("ix_backup_run_tag", table_name="backup_run") + op.drop_index("ix_backup_run_status", table_name="backup_run") + op.drop_index("ix_backup_run_kind", table_name="backup_run") + op.drop_table("backup_run") diff --git a/alembic/versions/0018_fc3h_backup_settings.py b/alembic/versions/0018_fc3h_backup_settings.py new file mode 100644 index 0000000..517c8f1 --- /dev/null +++ b/alembic/versions/0018_fc3h_backup_settings.py @@ -0,0 +1,62 @@ +"""fc3h: backup_* knobs on import_settings + +Revision ID: 0018 +Revises: 0017 +Create Date: 2026-05-24 + +Adds four columns to the singleton import_settings row: + - backup_db_nightly_enabled (default False — opt-in) + - backup_db_nightly_hour_utc (default 3) + - backup_db_keep_last_n (default 14) + - backup_images_keep_last_n (default 3) + +server_default ensures the singleton row is backfilled in place +without an UPDATE statement. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0018" +down_revision: Union[str, None] = "0017" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "import_settings", + sa.Column( + "backup_db_nightly_enabled", sa.Boolean(), + nullable=False, server_default=sa.false(), + ), + ) + op.add_column( + "import_settings", + sa.Column( + "backup_db_nightly_hour_utc", sa.Integer(), + nullable=False, server_default="3", + ), + ) + op.add_column( + "import_settings", + sa.Column( + "backup_db_keep_last_n", sa.Integer(), + nullable=False, server_default="14", + ), + ) + op.add_column( + "import_settings", + sa.Column( + "backup_images_keep_last_n", sa.Integer(), + nullable=False, server_default="3", + ), + ) + + +def downgrade() -> None: + op.drop_column("import_settings", "backup_images_keep_last_n") + op.drop_column("import_settings", "backup_db_keep_last_n") + op.drop_column("import_settings", "backup_db_nightly_hour_utc") + op.drop_column("import_settings", "backup_db_nightly_enabled") diff --git a/alembic/versions/0019_import_batch_refreshed.py b/alembic/versions/0019_import_batch_refreshed.py new file mode 100644 index 0000000..1770daa --- /dev/null +++ b/alembic/versions/0019_import_batch_refreshed.py @@ -0,0 +1,38 @@ +"""import_batch.refreshed counter for deep-scan sidecar re-application + +Revision ID: 0019 +Revises: 0018 +Create Date: 2026-05-25 + +Adds a `refreshed` counter to `import_batch`, mirroring the existing +`imported`/`skipped`/`failed`/`attachments` columns. Deep scan now +re-applies sidecar metadata to already-imported files (the IR feature +that didn't make the FC port the first time); a "refreshed" outcome +increments this counter so the UI can surface "X new, Y refreshed" +instead of the misleading "Scan complete — no new files" message. + +server_default=0 backfills existing rows in place — no UPDATE needed. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0019" +down_revision: Union[str, None] = "0018" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "import_batch", + sa.Column( + "refreshed", sa.Integer(), + nullable=False, server_default=sa.text("0"), + ), + ) + + +def downgrade() -> None: + op.drop_column("import_batch", "refreshed") diff --git a/alembic/versions/0020_library_audit_run.py b/alembic/versions/0020_library_audit_run.py new file mode 100644 index 0000000..07a8b87 --- /dev/null +++ b/alembic/versions/0020_library_audit_run.py @@ -0,0 +1,65 @@ +"""fc-cleanup: library_audit_run table for async transparency/single_color audits + +Revision ID: 0020 +Revises: 0019 +Create Date: 2026-05-26 + +The table backs the async audit lifecycle: rule + params snapshot, status +state machine ('running' → 'ready' → 'applied'/'cancelled'/'error'), and +the matched_ids JSONB array that the apply step deletes. Capped at 50k IDs +per row by the scan task (oversize = rule too aggressive, operator narrows +before re-running). +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from sqlalchemy.dialects import postgresql + +revision: str = "0020" +down_revision: Union[str, None] = "0019" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "library_audit_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("rule", sa.String(32), nullable=False), + sa.Column("params", postgresql.JSONB(astext_type=sa.Text()), nullable=False), + sa.Column( + "status", sa.String(16), + nullable=False, server_default="running", + ), + sa.Column( + "started_at", sa.DateTime(timezone=True), + nullable=False, server_default=sa.func.now(), + ), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column( + "scanned_count", sa.Integer(), + nullable=False, server_default="0", + ), + sa.Column( + "matched_count", sa.Integer(), + nullable=False, server_default="0", + ), + sa.Column( + "matched_ids", postgresql.JSONB(astext_type=sa.Text()), + nullable=False, server_default=sa.text("'[]'::jsonb"), + ), + sa.Column("error", sa.Text(), nullable=True), + ) + op.create_index( + "ix_library_audit_run_rule", "library_audit_run", ["rule"], + ) + op.create_index( + "ix_library_audit_run_status", "library_audit_run", ["status"], + ) + + +def downgrade() -> None: + op.drop_index("ix_library_audit_run_status", table_name="library_audit_run") + op.drop_index("ix_library_audit_run_rule", table_name="library_audit_run") + op.drop_table("library_audit_run") diff --git a/alembic/versions/0021_image_provenance_unique.py b/alembic/versions/0021_image_provenance_unique.py new file mode 100644 index 0000000..b9be941 --- /dev/null +++ b/alembic/versions/0021_image_provenance_unique.py @@ -0,0 +1,54 @@ +"""provenance-race: dedupe + UNIQUE(image_record_id, post_id) on image_provenance + +Revision ID: 0021 +Revises: 0020 +Create Date: 2026-05-26 + +Closes the race in Importer._apply_sidecar's existence-check + INSERT pattern. +Two workers writing for the same (image, post) pair both saw no existing row +and both inserted, leaving duplicates that then broke .scalar_one_or_none() +on every subsequent deep-scan rederive against those images +(MultipleResultsFound). Most plausibly seeded when the 5-min recovery sweep +re-enqueued a still-running long-import task and the second worker collided +with the first inside _apply_sidecar. + +Migration steps: + 1. DELETE all but min(id) per (image_record_id, post_id) pair. Operator's + DB had 2 affected pairs at write-time; harmless no-op if zero. + 2. Add UNIQUE constraint so the importer's new savepoint+IntegrityError + recovery path can trip on collision and re-select, mirroring + uq_source_artist_platform_url and uq_post_source_external_id. +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0021" +down_revision: Union[str, None] = "0020" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.execute( + """ + DELETE FROM image_provenance ip1 + USING image_provenance ip2 + WHERE ip1.image_record_id = ip2.image_record_id + AND ip1.post_id = ip2.post_id + AND ip1.id > ip2.id + """ + ) + op.create_unique_constraint( + "uq_image_provenance_image_post", + "image_provenance", + ["image_record_id", "post_id"], + ) + + +def downgrade() -> None: + op.drop_constraint( + "uq_image_provenance_image_post", + "image_provenance", + type_="unique", + ) diff --git a/alembic/versions/0022_source_per_artist_platform.py b/alembic/versions/0022_source_per_artist_platform.py new file mode 100644 index 0000000..d3e75dd --- /dev/null +++ b/alembic/versions/0022_source_per_artist_platform.py @@ -0,0 +1,223 @@ +"""source-collapse: one Source per (artist, platform) — consolidate junk per-post Sources + +Revision ID: 0022 +Revises: 0021 +Create Date: 2026-05-26 + +Closes the operator-flagged 2026-05-26 issue where the filesystem importer +called _find_or_create_source(url=sd.post_url), creating one Source row per +imported post URL. Operator's Atole artist had 406 Source rows where there +should have been 1 (the /cw/Atole subscription Source). + +Source represents a subscription feed (one per artist+platform — the +gallery-dl URL polled by the FC-3 downloader). Posts hang off it. The +filesystem importer was misusing Source as a per-post key. + +Migration steps per (artist_id, platform) group with >1 Source: + 1. Pick canonical — prefer a URL NOT matching '/posts/$' (real + campaign URL like /cw/Atole); else min(id). + 2. PRE-merge any Posts under non-canonical sources whose + external_post_id ALREADY exists under the canonical source. (Same + gallery-dl post imported via two different sidecar paths can plant + two Post rows with identical external_post_id under different + Sources for the same artist.) Repoint ImageProvenance + + ImageRecord.primary_post_id to the canonical-side Post, dedupe + ImageProvenance against alembic 0021's uq, then delete the + non-canonical-side Post. This MUST happen before step 3 — Postgres + fires uq_post_source_external_id row-by-row during the bulk UPDATE + and the merge-after-reparent ordering 500s on first collision + (operator-hit during v26.05.26.1 deploy, 2026-05-26). + 3. Reparent remaining Posts onto canonical (no collisions possible now). + 4. Reparent ImageProvenance.source_id off the non-canonical sources. + 5. Delete the orphan Source rows. + 6. If the canonical Source's URL still looks like a per-post URL (no + campaign URL existed among candidates), rewrite it to + 'sidecar::' so the artist detail page shows + something readable. +""" +from typing import Sequence, Union + +from alembic import op +from sqlalchemy import text + +revision: str = "0022" +down_revision: Union[str, None] = "0021" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +_POST_URL_RE = r"/posts/[^/]+$" + + +def upgrade() -> None: + conn = op.get_bind() + + # Find (artist_id, platform) groups with > 1 Source row. + groups = conn.execute(text(""" + SELECT artist_id, platform + FROM source + GROUP BY artist_id, platform + HAVING COUNT(*) > 1 + """)).fetchall() + + for artist_id, platform in groups: + rows = conn.execute( + text(""" + SELECT id, url FROM source + WHERE artist_id = :a AND platform = :p + ORDER BY id ASC + """), + {"a": artist_id, "p": platform}, + ).fetchall() + + # Canonical: first row whose URL doesn't look like a per-post URL; + # else min(id). + canonical_id = None + for sid, url in rows: + if not _matches_post_url(url): + canonical_id = sid + break + if canonical_id is None: + canonical_id = rows[0][0] + + other_ids = [sid for sid, _ in rows if sid != canonical_id] + if not other_ids: + continue + + # STEP 2: PRE-merge ALL Posts with duplicate external_post_id + # across the entire (canonical + others) group, BEFORE the bulk + # reparent. Two cases must both be handled: + # (A) canonical has Post X with epid=N; an "other" source has + # Post Y with epid=N → after bulk UPDATE, (canonical, N) + # collides with itself. + # (B) two different "other" sources each have a Post with + # epid=N; canonical has none → after bulk UPDATE, both + # are repointed to (canonical, N) and the second collides. + # The earlier version of this migration only handled (A); the + # operator's deploy 2026-05-26 tripped (B) at line 139. + # Fix: group ALL Posts in the (artist, platform) by epid; for + # any group with count>1, pick the keep (prefer one already + # under canonical; else lowest id) and merge the rest into it. + all_posts = conn.execute( + text(""" + SELECT external_post_id, id, source_id + FROM post + WHERE source_id = :canonical OR source_id = ANY(:others) + ORDER BY external_post_id, id + """), + {"canonical": canonical_id, "others": other_ids}, + ).fetchall() + by_epid: dict = {} + for epid, post_id, src_id in all_posts: + by_epid.setdefault(epid, []).append((post_id, src_id)) + for _epid, posts in by_epid.items(): + if len(posts) <= 1: + continue + # Prefer a Post already under canonical as the keep. + canonical_posts = [p for p in posts if p[1] == canonical_id] + if canonical_posts: + keep_id = canonical_posts[0][0] + else: + keep_id = posts[0][0] # already sorted by id ASC + drop_ids = [p[0] for p in posts if p[0] != keep_id] + for drop_id in drop_ids: + # Pre-delete image_provenance rows under drop_ whose + # image_record_id ALREADY has a provenance under keep — + # the UPDATE below would otherwise repoint them and + # trip uq_image_provenance_image_post (alembic 0021) + # row-by-row before any after-the-fact dedupe could + # run. Operator's v26.05.26.3 deploy 2026-05-26 tripped + # this at line 123. + conn.execute( + text(""" + DELETE FROM image_provenance + WHERE post_id = :drop_ + AND image_record_id IN ( + SELECT image_record_id FROM image_provenance + WHERE post_id = :keep + ) + """), + {"keep": keep_id, "drop_": drop_id}, + ) + # Now safe to repoint the survivors. + conn.execute( + text(""" + UPDATE image_provenance SET post_id = :keep + WHERE post_id = :drop_ + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text(""" + UPDATE image_record SET primary_post_id = :keep + WHERE primary_post_id = :drop_ + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text("DELETE FROM post WHERE id = :drop_"), + {"drop_": drop_id}, + ) + + # STEP 3: Bulk reparent the remaining Posts off the other + # Sources. After step 2, no collisions on + # (canonical, external_post_id) are possible. + conn.execute( + text(""" + UPDATE post SET source_id = :canonical + WHERE source_id = ANY(:others) + """), + {"canonical": canonical_id, "others": other_ids}, + ) + + # STEP 4: Reparent ImageProvenance.source_id (denormalized FK). + # No UNIQUE on source_id; safe bulk update. + conn.execute( + text(""" + UPDATE image_provenance SET source_id = :canonical + WHERE source_id = ANY(:others) + """), + {"canonical": canonical_id, "others": other_ids}, + ) + + # STEP 5: Drop the orphan Sources. + conn.execute( + text("DELETE FROM source WHERE id = ANY(:others)"), + {"others": other_ids}, + ) + + # If the canonical's URL still looks per-post (no campaign URL + # existed among the candidates), rewrite to a synthetic anchor so + # the artist detail page renders something readable. + canonical_url = conn.execute( + text("SELECT url FROM source WHERE id = :id"), + {"id": canonical_id}, + ).scalar_one() + if _matches_post_url(canonical_url): + slug = conn.execute( + text("SELECT slug FROM artist WHERE id = :id"), + {"id": artist_id}, + ).scalar_one() + conn.execute( + text(""" + UPDATE source + SET url = :new_url, enabled = false + WHERE id = :id + """), + { + "id": canonical_id, + "new_url": f"sidecar:{platform}:{slug}", + }, + ) + + +def downgrade() -> None: + # Lossy migration — orphan Sources deleted, Posts reparented, Posts + # merged. No safe downgrade. If you need to roll back the schema + # invariant, fork from 0021 and re-run filesystem imports. + pass + + +def _matches_post_url(url: str) -> bool: + """True if url ends with /posts/ (gallery-dl-style per-post URL).""" + import re + return bool(re.search(_POST_URL_RE, url or "")) diff --git a/alembic/versions/0023_drop_meta_rating_tag_kinds.py b/alembic/versions/0023_drop_meta_rating_tag_kinds.py new file mode 100644 index 0000000..fc65ea1 --- /dev/null +++ b/alembic/versions/0023_drop_meta_rating_tag_kinds.py @@ -0,0 +1,99 @@ +"""drop meta + rating tag kinds — operator-retired 2026-05-26 + +Revision ID: 0023 +Revises: 0022 +Create Date: 2026-05-26 + +Operator decided meta + rating aren't valid tag kinds for FC. Per-row +behavior: DELETE existing rows (operator chose "clean break" over +"convert to general"). All cascading FKs (image_tag, tag_alias, +tag_allowlist, tag_reference_embedding, tag_suggestion_rejection, +series_page) use ondelete="CASCADE" so a single DELETE on tag cleans +the related rows in one go. + +After the data cleanup, recreate the tag_kind ENUM without 'meta' / +'rating' (Postgres has no `ALTER TYPE ... DROP VALUE`; standard +rename-create-cast-drop dance). The server default 'general' is +dropped before the type swap and restored after. +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0023" +down_revision: Union[str, None] = "0022" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # 1. Delete tags of the retired kinds. CASCADE handles related tables. + op.execute("DELETE FROM tag WHERE kind IN ('meta', 'rating')") + + # 2. Drop the CHECK constraint that references the enum's literal + # values. Postgres can't resolve `kind = 'character'` across the + # type swap below — the literal would bind to the new tag_kind + # but the column is on tag_kind_old, producing + # "operator does not exist: tag_kind = tag_kind_old". + # (Operator-hit during the v26.05.26.5 deploy attempt; ck was + # originally added by alembic 0002.) Recreated post-swap. + op.drop_constraint( + "ck_tag_fandom_requires_character", "tag", type_="check" + ) + + # 3. Drop the server default — ALTER COLUMN TYPE can't carry it + # across the type swap below. + op.execute("ALTER TABLE tag ALTER COLUMN kind DROP DEFAULT") + + # 4. Recreate the tag_kind enum without meta/rating. + op.execute("ALTER TYPE tag_kind RENAME TO tag_kind_old") + op.execute( + "CREATE TYPE tag_kind AS ENUM (" + "'artist', 'character', 'fandom', 'general', " + "'series', 'archive', 'post'" + ")" + ) + op.execute( + "ALTER TABLE tag " + "ALTER COLUMN kind TYPE tag_kind " + "USING kind::text::tag_kind" + ) + op.execute("DROP TYPE tag_kind_old") + + # 5. Restore the server default. + op.execute("ALTER TABLE tag ALTER COLUMN kind SET DEFAULT 'general'") + + # 6. Restore the CHECK constraint (now bound to the new tag_kind). + op.create_check_constraint( + "ck_tag_fandom_requires_character", + "tag", + "(fandom_id IS NULL) OR (kind = 'character')", + ) + + +def downgrade() -> None: + # Add the values back to the enum so old code can boot. The deleted + # tag rows are gone permanently — no safe restore. + op.drop_constraint( + "ck_tag_fandom_requires_character", "tag", type_="check" + ) + op.execute("ALTER TABLE tag ALTER COLUMN kind DROP DEFAULT") + op.execute("ALTER TYPE tag_kind RENAME TO tag_kind_old") + op.execute( + "CREATE TYPE tag_kind AS ENUM (" + "'artist', 'character', 'fandom', 'general', " + "'series', 'archive', 'post', 'meta', 'rating'" + ")" + ) + op.execute( + "ALTER TABLE tag " + "ALTER COLUMN kind TYPE tag_kind " + "USING kind::text::tag_kind" + ) + op.execute("DROP TYPE tag_kind_old") + op.execute("ALTER TABLE tag ALTER COLUMN kind SET DEFAULT 'general'") + op.create_check_constraint( + "ck_tag_fandom_requires_character", + "tag", + "(fandom_id IS NULL) OR (kind = 'character')", + ) diff --git a/alembic/versions/0024_backfill_post_title_from_description.py b/alembic/versions/0024_backfill_post_title_from_description.py new file mode 100644 index 0000000..2b1385c --- /dev/null +++ b/alembic/versions/0024_backfill_post_title_from_description.py @@ -0,0 +1,80 @@ +"""backfill post.post_title from description first-line — 2026-05-27 + +Revision ID: 0024 +Revises: 0023 +Create Date: 2026-05-27 + +SubscribeStar gallery-dl always writes `title: ""` and embeds the leading +sentence inside `content` HTML. FC's sidecar parser was leaving +post_title NULL for every SubscribeStar post since FC-3 shipped. The +parser fix (sidecar._first_line_text fallback) now synthesizes a title +at parse time; this migration applies the same logic retroactively to +existing rows. + +Operator-flagged 2026-05-27 after inspecting +/mnt/Data/Patreon/Cheunart/subscribestar/ sidecars. + +Idempotent: only touches rows where post_title IS NULL or empty AND +description IS NOT NULL. Re-running the migration is a no-op. +""" +from __future__ import annotations + +import re +from typing import Sequence, Union + +from alembic import op +from sqlalchemy import text + +revision: str = "0024" +down_revision: Union[str, None] = "0023" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +_TAG_RE = re.compile(r"<[^>]+>") +_WS_RE = re.compile(r"\s+") + + +def _first_line_text(body: str, limit: int = 120) -> str | None: + """Mirror of sidecar._first_line_text. Kept inline so the migration + doesn't carry a runtime import dependency from app code that may + have moved by the time the migration is replayed years from now.""" + if not body: + return None + text_ = _TAG_RE.sub(" ", body) + text_ = text_.replace("\xa0", " ") + for line in text_.splitlines(): + line = _WS_RE.sub(" ", line).strip() + if line: + if len(line) > limit: + return line[: limit - 1].rstrip() + "…" + return line + return None + + +def upgrade() -> None: + bind = op.get_bind() + rows = bind.execute( + text( + "SELECT id, description FROM post " + "WHERE (post_title IS NULL OR post_title = '') " + "AND description IS NOT NULL AND description <> ''" + ) + ).fetchall() + updated = 0 + for row in rows: + derived = _first_line_text(row.description) + if not derived: + continue + bind.execute( + text("UPDATE post SET post_title = :t WHERE id = :id"), + {"t": derived, "id": row.id}, + ) + updated += 1 + print(f"0024: backfilled post_title on {updated} row(s)") + + +def downgrade() -> None: + # No safe restore — we can't tell which post_titles were derived vs + # genuinely present. Leave the column alone on rollback. + pass diff --git a/alembic/versions/0025_fix_subscribestar_post_ids.py b/alembic/versions/0025_fix_subscribestar_post_ids.py new file mode 100644 index 0000000..b44f430 --- /dev/null +++ b/alembic/versions/0025_fix_subscribestar_post_ids.py @@ -0,0 +1,288 @@ +"""sidecar-audit followup: correct external_post_id + post_url across all platforms + +Revision ID: 0025 +Revises: 0024 +Create Date: 2026-05-27 + +Closes the operator-flagged 2026-05-27 sidecar audit findings. Three +data-correctness bugs across non-Patreon platforms had been silently +corrupting Posts since FC-3 shipped; the parser fix (sidecar.py, same +commit) addresses new imports. This migration cleans up existing rows. + +Per-platform actions: + + subscribestar — gallery-dl wrote the per-attachment id in `id` and + the actual post id in `post_id`. FC's parser picked `id`, so every + multi-image SubscribeStar post was fragmented into N Post rows. + 1. For each SubscribeStar Post, read its sidecar (via the related + ImageRecord's on-disk path), pull `post_id`, overwrite + external_post_id and post_url. + 2. Merge groups of Posts under one source that now share an + external_post_id (fragments of the same actual post). Same + ImageProvenance pre-delete + repoint dance as alembic 0022. + + hentaifoundry — sidecars have NO `url` field; `src` is the image + URL. FC's parser stored post_url=NULL. Read each HF Post's sidecar + for `user` + `index`, derive the canonical /pictures/user// + permalink. external_post_id (= `index`) was already correct. + + discord — gallery-dl wrote the CDN attachment URL in `url`. FC's + parser stored that as post_url. Read each Discord Post's sidecar + for the server/channel/message triple, derive the proper + discord.com/channels/.../ permalink. external_post_id (= + `message_id`) was already correct. + + pixiv — pure-SQL backfill: replace any `i.pximg.net`-style URL on + Post.post_url with the derived `/artworks/` permalink. Pixiv + external_post_id (= `id`) was already correct; no sidecar IO + needed. + +Idempotent: re-running on already-corrected data is a no-op (skips +rows whose derived value matches what's already stored). + +Posts whose related ImageRecord paths don't resolve on disk (orphaned +filesystem state) are skipped with a count in the migration output — +those will be picked up by a future deep-scan. +""" +from __future__ import annotations + +import json +import re +from pathlib import Path +from typing import Sequence, Union + +from alembic import op +from sqlalchemy import text + +revision: str = "0025" +down_revision: Union[str, None] = "0024" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +# Mirror of sidecar._NUMBERING_PREFIX. Kept inline so the migration is +# self-contained (the operator's banked rule: +# reference_postgres_enum_swap_drop_checks.md says migrations shouldn't +# import from runtime app code). +_NUMBERING_PREFIX = re.compile(r"^\d+_(.+)$") + + +def _find_sidecar(media_path: Path) -> Path | None: + """gallery-dl writes the sidecar under the unprefixed stem + (`HOLLOW-ICHIGO.json`) while the media file gets a NN_ ordering + prefix (`01_HOLLOW-ICHIGO.png`). Try in order: + 1. .json next to the media + 2. .json next to the media (full-name variant) + 3. strip the NN_ prefix from the stem, then .json + """ + if not media_path: + return None + cand = media_path.with_suffix(".json") + if cand.is_file(): + return cand + cand = media_path.parent / f"{media_path.name}.json" + if cand.is_file(): + return cand + m = _NUMBERING_PREFIX.match(media_path.stem) + if m: + cand = media_path.parent / f"{m.group(1)}.json" + if cand.is_file(): + return cand + return None + + +def _str_id(v) -> str | None: + """str() a JSON scalar id; reject bool (JSON booleans are ints in + Python's eyes but they aren't valid sidecar ids).""" + if isinstance(v, bool): + return None + if isinstance(v, (str, int)) and str(v).strip(): + return str(v).strip() + return None + + +def _str_field(v) -> str | None: + if isinstance(v, str) and v.strip(): + return v.strip() + return None + + +def upgrade() -> None: + conn = op.get_bind() + + # ── PART 1: Per-platform corrections requiring filesystem IO ───── + # SubscribeStar, HentaiFoundry, Discord all need fields from the + # sidecar to construct the right post_url. We walk each Post's + # related ImageRecord.path to find the sidecar, read it, derive, + # and update. + targets = conn.execute(text(""" + SELECT p.id, p.external_post_id, p.post_url, s.platform + FROM post p + JOIN source s ON s.id = p.source_id + WHERE s.platform IN ('subscribestar', 'hentaifoundry', 'discord') + """)).fetchall() + + stats: dict[str, dict[str, int]] = { + plat: {"read": 0, "updated": 0, "no_sidecar": 0} + for plat in ("subscribestar", "hentaifoundry", "discord") + } + for post_row in targets: + plat = post_row.platform + path = _first_attachment_path(conn, post_row.id) + if not path: + stats[plat]["no_sidecar"] += 1 + continue + sidecar = _find_sidecar(Path(path)) + if sidecar is None: + stats[plat]["no_sidecar"] += 1 + continue + try: + data = json.loads(sidecar.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + stats[plat]["no_sidecar"] += 1 + continue + stats[plat]["read"] += 1 + + new_epid = post_row.external_post_id + new_url = None + if plat == "subscribestar": + pid = _str_id(data.get("post_id")) + if pid: + new_epid = pid + new_url = f"https://www.subscribestar.com/posts/{pid}" + elif plat == "hentaifoundry": + user = _str_field(data.get("user")) or _str_field(data.get("artist")) + idx = _str_id(data.get("index")) + if user and idx: + new_url = f"https://www.hentai-foundry.com/pictures/user/{user}/{idx}" + elif plat == "discord": + sid = _str_id(data.get("server_id")) + cid = _str_id(data.get("channel_id")) + mid = _str_id(data.get("message_id")) + if sid and cid and mid: + new_url = f"https://discord.com/channels/{sid}/{cid}/{mid}" + + # Idempotent: skip if nothing changed. + if new_epid == post_row.external_post_id and new_url == post_row.post_url: + continue + conn.execute( + text(""" + UPDATE post + SET external_post_id = :epid, post_url = :url + WHERE id = :id + """), + {"epid": new_epid, "url": new_url, "id": post_row.id}, + ) + stats[plat]["updated"] += 1 + + for plat, s in stats.items(): + print( + f"0025: {plat} — read {s['read']} sidecars, " + f"updated {s['updated']} Posts, " + f"{s['no_sidecar']} Posts had no resolvable sidecar" + ) + + # ── PART 2: Merge SubscribeStar fragments now sharing epid ─────── + # After Part 1, each group of Posts under one source with the SAME + # new external_post_id is a fragment-set of the same actual post. + # Merge to one canonical row. Pre-handle the same ImageProvenance + # collision pattern as alembic 0022 (uq_image_provenance_image_post). + fragment_groups = conn.execute(text(""" + SELECT p.source_id, p.external_post_id, + ARRAY_AGG(p.id ORDER BY p.id ASC) AS post_ids + FROM post p + JOIN source s ON s.id = p.source_id + WHERE s.platform = 'subscribestar' + AND p.external_post_id IS NOT NULL + GROUP BY p.source_id, p.external_post_id + HAVING COUNT(*) > 1 + """)).fetchall() + + merged = 0 + for grp in fragment_groups: + post_ids = list(grp.post_ids) + keep_id, *drop_ids = post_ids + for drop_id in drop_ids: + # Pre-DELETE colliding ImageProvenance under drop_ that + # already exist under keep (alembic 0022 banked the pattern). + conn.execute( + text(""" + DELETE FROM image_provenance + WHERE post_id = :drop_ + AND image_record_id IN ( + SELECT image_record_id FROM image_provenance + WHERE post_id = :keep + ) + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text(""" + UPDATE image_provenance SET post_id = :keep + WHERE post_id = :drop_ + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text(""" + UPDATE image_record SET primary_post_id = :keep + WHERE primary_post_id = :drop_ + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text(""" + UPDATE post_attachment SET post_id = :keep + WHERE post_id = :drop_ + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text("DELETE FROM post WHERE id = :drop_"), + {"drop_": drop_id}, + ) + merged += 1 + print(f"0025: subscribestar — merged {merged} duplicate Post fragments") + + # ── PART 3: Pixiv post_url backfill (pure SQL) ─────────────────── + # Pixiv's external_post_id is already correct (gallery-dl's `id` is + # the post id). Only post_url needs derivation: replace anything + # under i.pximg.net (the file URL) with the /artworks/ permalink. + pixiv_updated = conn.execute(text(""" + UPDATE post p + SET post_url = 'https://www.pixiv.net/artworks/' || p.external_post_id + FROM source s + WHERE p.source_id = s.id + AND s.platform = 'pixiv' + AND p.external_post_id IS NOT NULL + AND (p.post_url IS NULL + OR p.post_url LIKE 'https://i.pximg.net/%' + OR p.post_url LIKE 'http://i.pximg.net/%') + """)).rowcount + print(f"0025: pixiv — backfilled post_url on {pixiv_updated} Posts") + + +def _first_attachment_path(conn, post_id: int) -> str | None: + """Return any ImageRecord.path attached to this post (via + ImageProvenance). Lowest-id row keeps the migration deterministic + so re-running on the same DB picks the same sidecar.""" + row = conn.execute( + text(""" + SELECT ir.path + FROM image_provenance ip + JOIN image_record ir ON ir.id = ip.image_record_id + WHERE ip.post_id = :pid + ORDER BY ip.id ASC + LIMIT 1 + """), + {"pid": post_id}, + ).first() + return row[0] if row else None + + +def downgrade() -> None: + # Lossy: external_post_id values were overwritten with the correct + # post_id; original per-attachment ids weren't preserved. Post-merge + # also deleted drop rows. No safe restore. To roll back the schema + # invariant, fork from 0024 and re-run sidecar imports. + pass diff --git a/alembic/versions/0026_import_task_recovery_count_refetched.py b/alembic/versions/0026_import_task_recovery_count_refetched.py new file mode 100644 index 0000000..ccbc3da --- /dev/null +++ b/alembic/versions/0026_import_task_recovery_count_refetched.py @@ -0,0 +1,53 @@ +"""import_task.recovery_count + refetched — poison-pill circuit breaker + +Revision ID: 0026 +Revises: 0025 +Create Date: 2026-05-28 + +Backs the import-task resilience work (operator-flagged 2026-05-28): + +- recovery_count: how many times recover_interrupted_tasks has + re-queued this row from a stuck 'processing' state. A row that + hard-crashes the worker (OOM / segfault on a corrupt or oversized + input) leaves no terminal flip, so the sweep re-queues it — and + without a cap it would loop forever, re-crashing the worker each + time. After MAX_RECOVERY_ATTEMPTS the sweep marks it 'failed' with a + diagnostic instead. + +- refetched: whether a one-shot re-download has already been attempted + for this task's file. Bounds the Layer-2 re-fetch remediation to a + single attempt so source-side corruption doesn't loop. + +Both default to 0 / false; additive, no backfill needed. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0026" +down_revision: Union[str, None] = "0025" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "import_task", + sa.Column( + "recovery_count", sa.Integer(), nullable=False, + server_default="0", + ), + ) + op.add_column( + "import_task", + sa.Column( + "refetched", sa.Boolean(), nullable=False, + server_default=sa.false(), + ), + ) + + +def downgrade() -> None: + op.drop_column("import_task", "refetched") + op.drop_column("import_task", "recovery_count") diff --git a/alembic/versions/0027_drop_migration_run.py b/alembic/versions/0027_drop_migration_run.py new file mode 100644 index 0000000..454481b --- /dev/null +++ b/alembic/versions/0027_drop_migration_run.py @@ -0,0 +1,50 @@ +"""drop migration_run — one-and-done GS/IR migration tooling removed + +Revision ID: 0027 +Revises: 0026 +Create Date: 2026-05-29 + +The GS/IR migration tooling (services/migrators, /api/migrate, the +run_migration task, LegacyMigrationCard, and the MigrationRun model) was +removed after the migration cutover completed. This drops its now-orphaned +run-log table. Downgrade recreates the table (mirrors the old model) so the +migration is reversible. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from sqlalchemy.dialects.postgresql import JSONB + +revision: str = "0027" +down_revision: Union[str, None] = "0026" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.drop_table("migration_run") + + +def downgrade() -> None: + op.create_table( + "migration_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("kind", sa.String(length=32), nullable=False), + sa.Column("status", sa.String(length=32), nullable=False), + sa.Column("dry_run", sa.Boolean(), nullable=False, server_default=sa.false()), + sa.Column( + "started_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column( + "counts", JSONB(), nullable=False, server_default=sa.text("'{}'::jsonb"), + ), + sa.Column("error", sa.Text(), nullable=True), + sa.Column( + "metadata", JSONB(), nullable=False, server_default=sa.text("'{}'::jsonb"), + ), + ) + op.create_index("ix_migration_run_kind", "migration_run", ["kind"]) + op.create_index("ix_migration_run_status", "migration_run", ["status"]) diff --git a/alembic/versions/0028_collapse_sidecar_synthetics_into_real_sources.py b/alembic/versions/0028_collapse_sidecar_synthetics_into_real_sources.py new file mode 100644 index 0000000..5ec9267 --- /dev/null +++ b/alembic/versions/0028_collapse_sidecar_synthetics_into_real_sources.py @@ -0,0 +1,190 @@ +"""collapse-sidecar-synthetic: repoint Posts/ImageProvenance/DownloadEvents +from `sidecar::` synthetic Source anchors onto the real +Source for the same (artist, platform) when one exists, then delete the +synthetic. + +Revision ID: 0028 +Revises: 0027 +Create Date: 2026-05-31 + +Background: alembic 0022 (2026-05-26) consolidated the old per-post-URL +Source rows into one canonical Source per (artist, platform). When NO +real campaign URL was salvageable among the candidates, it rewrote the +canonical row to url='sidecar::' enabled=false as a +disabled anchor for any Posts already attached. + +That was fine while it was the only Source for that artist+platform. +But: the unique constraint on Source is (artist_id, platform, url), not +(artist_id, platform). When the operator later added the real +subscription via the UI / extension / etc., a SECOND row landed — +the real one — with id > the synthetic. Both coexisted. + +Two follow-on problems surfaced 2026-05-31: + + 1. The Subscriptions UI listed both rows. The synthetic was disabled + so the scheduler never polled it, but it looked like a phantom + subscription. (Fixed in same commit by SourceService.list filter.) + 2. importer._source_for_sidecar picked Source by `ORDER BY id ASC + LIMIT 1`, so EVERY gallery-dl download since the real Source was + added attached its Post to the SYNTHETIC anchor, not the real + Source. (Fixed in same commit by preferring non-sidecar URLs.) + +This migration is the data half of the cleanup: for every (artist, +platform) with both a synthetic AND a real Source, repoint the +synthetic's children (Posts, ImageProvenance, DownloadEvents) onto the +real Source and delete the synthetic. Reuses the same epid/provenance +collision dance from alembic 0022 because the same uniqueness +constraints fire row-by-row during bulk UPDATEs. + +Lone synthetic anchors — those where no real Source for the same +(artist, platform) exists (e.g., filesystem-imported artist with no +subscription added) — are LEFT INTACT. They anchor real imported +content; deleting them would CASCADE-delete the Posts the operator +imported. The SourceService.list filter hides them from the UI; the +operator can delete them by hand if they want the underlying imports +gone. +""" +from typing import Sequence, Union + +from alembic import op +from sqlalchemy import text + +revision: str = "0028" +down_revision: Union[str, None] = "0027" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + conn = op.get_bind() + + # Find (artist_id, platform) groups where BOTH a sidecar synthetic + # and at least one real Source exist. + groups = conn.execute(text(""" + SELECT artist_id, platform + FROM source + GROUP BY artist_id, platform + HAVING bool_or(url LIKE 'sidecar:%') + AND bool_or(url NOT LIKE 'sidecar:%') + """)).fetchall() + + for artist_id, platform in groups: + rows = conn.execute( + text(""" + SELECT id, url FROM source + WHERE artist_id = :a AND platform = :p + ORDER BY id ASC + """), + {"a": artist_id, "p": platform}, + ).fetchall() + + synthetic_ids = [sid for sid, url in rows if url.startswith("sidecar:")] + real_rows = [(sid, url) for sid, url in rows if not url.startswith("sidecar:")] + if not synthetic_ids or not real_rows: + continue # belt+suspenders; the GROUP BY already filtered + + # Canonical real: lowest-id non-sidecar Source. + canonical_id = real_rows[0][0] + + # STEP A: PRE-merge Post collisions on (canonical, external_post_id). + # Mirror alembic 0022's pre-merge logic — when synth has Post X + # epid=N and real has Post Y epid=N, the bulk UPDATE below would + # trip uq_post_source_external_id row-by-row. Group all Posts + # under (canonical + synthetics) by epid; for any group >1, + # pick a keep (prefer one already under canonical, else lowest + # id) and merge the rest into it. + all_posts = conn.execute( + text(""" + SELECT external_post_id, id, source_id + FROM post + WHERE source_id = :canonical OR source_id = ANY(:synths) + ORDER BY external_post_id, id + """), + {"canonical": canonical_id, "synths": synthetic_ids}, + ).fetchall() + by_epid: dict = {} + for epid, post_id, src_id in all_posts: + by_epid.setdefault(epid, []).append((post_id, src_id)) + for _epid, posts in by_epid.items(): + if len(posts) <= 1: + continue + canonical_side = [p for p in posts if p[1] == canonical_id] + keep_id = canonical_side[0][0] if canonical_side else posts[0][0] + drop_ids = [p[0] for p in posts if p[0] != keep_id] + for drop_id in drop_ids: + # Pre-delete image_provenance rows under drop_ whose + # image_record_id already has provenance under keep — + # avoids tripping uq_image_provenance_image_post (0021) + # row-by-row during the repoint UPDATE. + conn.execute( + text(""" + DELETE FROM image_provenance + WHERE post_id = :drop_ + AND image_record_id IN ( + SELECT image_record_id FROM image_provenance + WHERE post_id = :keep + ) + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text(""" + UPDATE image_provenance SET post_id = :keep + WHERE post_id = :drop_ + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text(""" + UPDATE image_record SET primary_post_id = :keep + WHERE primary_post_id = :drop_ + """), + {"keep": keep_id, "drop_": drop_id}, + ) + conn.execute( + text("DELETE FROM post WHERE id = :drop_"), + {"drop_": drop_id}, + ) + + # STEP B: Bulk reparent the remaining Posts off the synthetics. + conn.execute( + text(""" + UPDATE post SET source_id = :canonical + WHERE source_id = ANY(:synths) + """), + {"canonical": canonical_id, "synths": synthetic_ids}, + ) + + # STEP C: Reparent ImageProvenance.source_id (denormalized FK; + # no UNIQUE on source_id, safe bulk). + conn.execute( + text(""" + UPDATE image_provenance SET source_id = :canonical + WHERE source_id = ANY(:synths) + """), + {"canonical": canonical_id, "synths": synthetic_ids}, + ) + + # STEP D: Reparent any DownloadEvent.source_id. Synthetics are + # enabled=false so the scheduler never created events for them; + # this is belt+suspenders for any rows planted by manual force + # or older code paths. + conn.execute( + text(""" + UPDATE download_event SET source_id = :canonical + WHERE source_id = ANY(:synths) + """), + {"canonical": canonical_id, "synths": synthetic_ids}, + ) + + # STEP E: Drop the now-empty synthetics. + conn.execute( + text("DELETE FROM source WHERE id = ANY(:synths)"), + {"synths": synthetic_ids}, + ) + + +def downgrade() -> None: + # Lossy migration — synthetic Sources deleted, Posts repointed and + # potentially merged. No safe downgrade. + pass diff --git a/alembic/versions/0029_drop_artist_copyright_ml_thresholds.py b/alembic/versions/0029_drop_artist_copyright_ml_thresholds.py new file mode 100644 index 0000000..e6c044b --- /dev/null +++ b/alembic/versions/0029_drop_artist_copyright_ml_thresholds.py @@ -0,0 +1,71 @@ +"""drop artist + copyright ml thresholds; lower general default to 0.50 + +Revision ID: 0029 +Revises: 0028 +Create Date: 2026-06-01 + +Operator-flagged 2026-06-01: the view modal's Suggestions panel hides +most general-category predictions because the default threshold is +0.95. Lowering the default to 0.50 (matches character) so general +suggestions surface more aggressively; the value remains tunable in +Settings → ML. + +Same change retires two ML suggestion categories whose Tag.kind +surfaces are unused: + +- `artist`: retired in FC-2d-vii-c — artist identity is acquisition- + derived (image_record.artist_id), never ML-inferred. The threshold + column was a leftover from before that retirement. +- `copyright`: retired 2026-06-01 — the app uses `fandom` for the + franchise/copyright concept (per TagsView.vue's doc comment); no + Tag rows of kind=copyright exist, and the threshold column never + fed anything user-visible. + +Both columns are dropped from ml_settings; the existing row's +suggestion_threshold_general value is bumped from 0.95 to 0.50 iff +it's still at the old default, so deployed installs pick up the new +UX without overriding any operator tuning. +""" +from typing import Sequence, Union + +from alembic import op +from sqlalchemy import text + +revision: str = "0029" +down_revision: Union[str, None] = "0028" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # Bump the general threshold for installs still at the old default. + op.execute(text( + "UPDATE ml_settings " + "SET suggestion_threshold_general = 0.50 " + "WHERE id = 1 AND suggestion_threshold_general = 0.95" + )) + op.drop_column("ml_settings", "suggestion_threshold_artist") + op.drop_column("ml_settings", "suggestion_threshold_copyright") + + +def downgrade() -> None: + # Restore the columns with their prior defaults. The bump from + # 0.95 → 0.50 isn't reversible without remembering whether the + # operator had explicitly set 0.95 (unlikely — that was just the + # default) so we leave the current general value as-is. + from sqlalchemy import Column, Float + + op.add_column( + "ml_settings", + Column( + "suggestion_threshold_artist", + Float, nullable=False, server_default="0.30", + ), + ) + op.add_column( + "ml_settings", + Column( + "suggestion_threshold_copyright", + Float, nullable=False, server_default="0.50", + ), + ) diff --git a/alembic/versions/0030_nullable_post_source_id_denorm_artist_id.py b/alembic/versions/0030_nullable_post_source_id_denorm_artist_id.py new file mode 100644 index 0000000..c37e499 --- /dev/null +++ b/alembic/versions/0030_nullable_post_source_id_denorm_artist_id.py @@ -0,0 +1,145 @@ +"""nullable post.source_id + denormalized post.artist_id; retire sidecar synthetics + +Revision ID: 0030 +Revises: 0029 +Create Date: 2026-06-01 + +Operator-asked 2026-06-01 after the Dymkens orphan investigation: the +sidecar synthetic Source pattern (`sidecar::` rows +with enabled=false) was technically correct but misled the operator +into thinking they had phantom subscriptions. The synthetics existed +solely to satisfy `Post.source_id NOT NULL` for filesystem-imported +content with no real subscription. + +This migration makes the data model honest: + +1. **Post gets a denormalized `artist_id` column** so artist filters + work without traversing `Post → Source.artist_id`. Backfilled from + the existing Source linkage, then NOT NULL'd. +2. **`Post.source_id` becomes nullable**, FK ondelete `CASCADE` → `SET + NULL`. Deleting a Source detaches its Posts instead of destroying + imported content (semantically: subscription ends, archive stays). +3. **`ImageProvenance.source_id` becomes nullable** with the same FK + semantic change. +4. **Sidecar synthetic Sources are deleted** — first NULL out the + FKs from Post + ImageProvenance pointing at them (so the implicit + CASCADE doesn't fire), then delete. DownloadEvent FK is unchanged + (still CASCADE'd, NOT NULL'd) — synthetics have `enabled=false` + so no events exist for them. + +Uniqueness handling: the existing `uq_post_source_external_id` +(source_id, external_post_id) keeps working for source-bound Posts +(Postgres treats NULL != NULL so NULL-source rows aren't deduped by +it). A second partial unique index covers the NULL-source case on +(artist_id, external_post_id) so filesystem-imported posts still +dedupe within an artist. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from sqlalchemy import text + +revision: str = "0030" +down_revision: Union[str, None] = "0029" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + conn = op.get_bind() + + # Step 1: add Post.artist_id, initially nullable for backfill. + # FK naming follows the Base.metadata naming_convention + # (fk___) — alembic 0001 set this up. + op.add_column( + "post", + sa.Column("artist_id", sa.Integer, nullable=True), + ) + op.create_foreign_key( + "fk_post_artist_id_artist", "post", "artist", + ["artist_id"], ["id"], ondelete="CASCADE", + ) + + # Step 2: backfill from Source.artist_id (every existing Post has a + # Source today, so every row gets populated). + conn.execute(text(""" + UPDATE post p + SET artist_id = s.artist_id + FROM source s + WHERE p.source_id = s.id AND p.artist_id IS NULL + """)) + + # Sanity: count any remaining NULLs. Should be zero pre-this-migration. + remaining = conn.execute(text( + "SELECT COUNT(*) FROM post WHERE artist_id IS NULL" + )).scalar_one() + if remaining: + raise RuntimeError( + f"alembic 0030: {remaining} post rows have no resolvable " + f"artist_id after backfill. Investigate before continuing." + ) + + # Step 3: enforce NOT NULL + add index for artist-filter queries. + op.alter_column("post", "artist_id", nullable=False) + op.create_index("ix_post_artist_id", "post", ["artist_id"]) + + # Step 4: relax post.source_id + flip FK to SET NULL. The original FK + # name from alembic 0001 is `fk_post_source_id_source` per the + # NAMING_CONVENTION in models/base.py. + op.alter_column("post", "source_id", nullable=True) + op.drop_constraint("fk_post_source_id_source", "post", type_="foreignkey") + op.create_foreign_key( + "fk_post_source_id_source", "post", "source", + ["source_id"], ["id"], ondelete="SET NULL", + ) + + # Step 5: relax image_provenance.source_id + flip FK to SET NULL. + op.alter_column("image_provenance", "source_id", nullable=True) + op.drop_constraint( + "fk_image_provenance_source_id_source", "image_provenance", + type_="foreignkey", + ) + op.create_foreign_key( + "fk_image_provenance_source_id_source", "image_provenance", "source", + ["source_id"], ["id"], ondelete="SET NULL", + ) + + # Step 6: partial unique index on (artist_id, external_post_id) for + # NULL-source Posts. The existing uq_post_source_external_id keeps + # guarding source-bound rows; NULL-source rows now dedupe within + # an artist. + op.execute( + "CREATE UNIQUE INDEX uq_post_artist_external_id_null_source " + "ON post (artist_id, external_post_id) " + "WHERE source_id IS NULL" + ) + + # Step 7: retire sidecar synthetic Sources. NULL out the references + # FIRST (the new FK is SET NULL so CASCADE wouldn't fire anyway, but + # being explicit makes the intent clear). Then delete the synthetic + # source rows. Any DownloadEvent rows under synthetics CASCADE-die + # with the source — synthetics have enabled=false so there shouldn't + # be any in practice. + conn.execute(text(""" + UPDATE post + SET source_id = NULL + WHERE source_id IN (SELECT id FROM source WHERE url LIKE 'sidecar:%') + """)) + conn.execute(text(""" + UPDATE image_provenance + SET source_id = NULL + WHERE source_id IN (SELECT id FROM source WHERE url LIKE 'sidecar:%') + """)) + deleted = conn.execute(text( + "DELETE FROM source WHERE url LIKE 'sidecar:%' RETURNING id" + )).rowcount + print(f"alembic 0030: deleted {deleted} sidecar synthetic source rows") + + +def downgrade() -> None: + # Lossy migration — the deleted sidecar synthetics can't be + # restored from the orphan post.source_id / image_provenance.source_id + # values, and the partial unique index encodes a constraint that + # NULL-source Posts may now exist. No safe downgrade. + pass diff --git a/alembic/versions/0031_source_backfill_runs_remaining.py b/alembic/versions/0031_source_backfill_runs_remaining.py new file mode 100644 index 0000000..ed140fc --- /dev/null +++ b/alembic/versions/0031_source_backfill_runs_remaining.py @@ -0,0 +1,45 @@ +"""source.backfill_runs_remaining: sticky deep-scan mode + +Revision ID: 0031 +Revises: 0030 +Create Date: 2026-06-01 + +Tick vs backfill mode for subscription downloads. When +`backfill_runs_remaining > 0`, the next N download runs use +`skip: True` + 30-min timeout (walk full history). When 0, runs use +`skip: "exit:20"` + 14.5-min timeout (catch-up mode, exits early once +20 contiguous archived items are seen). + +Operator-flagged 2026-06-01 (Knuxy run #38887): a creator with ~550 +archived posts saturates the 870s catch-up timeout even when there is +no new content, because gallery-dl's default `skip: True` keeps walking. +Tick mode short-circuits that; backfill mode is the explicit opt-in for +deep history scans. + +Default 0 (all existing subscriptions start in tick mode). +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0031" +down_revision: Union[str, None] = "0030" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "source", + sa.Column( + "backfill_runs_remaining", + sa.Integer, + nullable=False, + server_default="0", + ), + ) + + +def downgrade() -> None: + op.drop_column("source", "backfill_runs_remaining") diff --git a/alembic/versions/0032_source_error_type.py b/alembic/versions/0032_source_error_type.py new file mode 100644 index 0000000..e264b35 --- /dev/null +++ b/alembic/versions/0032_source_error_type.py @@ -0,0 +1,41 @@ +"""source.error_type: surface ErrorType taxonomy in FailingSourcesCard + +Revision ID: 0032 +Revises: 0031 +Create Date: 2026-06-02 + +Audit 2026-06-02: the backend computes 13 ErrorType categories (auth_error, +rate_limited, not_found, access_denied, validation_failed, etc.) and +stamps each one on DownloadEvent.metadata, but the Source row only carried +the free-text last_error. Operators couldn't bulk-triage failing sources +("all auth_error → rotate cookies, all rate_limited → just wait") without +opening Logs per row. + +This column receives the last error_type from _update_source_health +and gets cleared on a successful run. Nullable + indexed so the failing- +sources rollup can filter/group cheaply. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0032" +down_revision: Union[str, None] = "0031" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "source", + sa.Column("error_type", sa.String(length=32), nullable=True), + ) + op.create_index( + "ix_source_error_type", "source", ["error_type"], + ) + + +def downgrade() -> None: + op.drop_index("ix_source_error_type", table_name="source") + op.drop_column("source", "error_type") diff --git a/alembic/versions/0033_suggestion_threshold_default_070.py b/alembic/versions/0033_suggestion_threshold_default_070.py new file mode 100644 index 0000000..652cf44 --- /dev/null +++ b/alembic/versions/0033_suggestion_threshold_default_070.py @@ -0,0 +1,48 @@ +"""suggestion_threshold default 0.50 → 0.70 + +Revision ID: 0033 +Revises: 0032 +Create Date: 2026-06-02 + +Operator-flagged 2026-06-02 — the 0.50 default (set on 2026-06-01) is +too noisy in practice; raise to 0.70 for both suggestion categories. + +Only conditionally updates singletons whose current value is still the +2026-06-01 default (0.50). Operators who deliberately tuned their row +to some other value (0.55, 0.65, 0.80, etc. via the Settings UI) keep +their pick — the migration only catches the unchanged-default case. +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0033" +down_revision: Union[str, None] = "0032" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.execute( + "UPDATE ml_settings " + "SET suggestion_threshold_character = 0.70 " + "WHERE id = 1 AND suggestion_threshold_character = 0.50" + ) + op.execute( + "UPDATE ml_settings " + "SET suggestion_threshold_general = 0.70 " + "WHERE id = 1 AND suggestion_threshold_general = 0.50" + ) + + +def downgrade() -> None: + op.execute( + "UPDATE ml_settings " + "SET suggestion_threshold_character = 0.50 " + "WHERE id = 1 AND suggestion_threshold_character = 0.70" + ) + op.execute( + "UPDATE ml_settings " + "SET suggestion_threshold_general = 0.50 " + "WHERE id = 1 AND suggestion_threshold_general = 0.70" + ) diff --git a/alembic/versions/0034_artist_visit.py b/alembic/versions/0034_artist_visit.py new file mode 100644 index 0000000..a2234a6 --- /dev/null +++ b/alembic/versions/0034_artist_visit.py @@ -0,0 +1,53 @@ +"""artist_visit: per-artist last-viewed timestamp for the "+N new" badge + +Revision ID: 0034 +Revises: 0033 +Create Date: 2026-06-03 + +Powers the artists-directory "+N new since last visit" badge + ArtistView +banner. Single row per artist (no user_id yet — rule #47 multi-user ACL +is aspirational; widens to (user_id, artist_id) PK when User lands). + +Seed every existing artist with `last_viewed_at = NOW()` so the badge +starts at 0 across the board — no noisy "you have 5000 unseen images" +on first deploy. New artists auto-get a row via +`ArtistService.find_or_create`. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0034" +down_revision: Union[str, None] = "0033" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "artist_visit", + sa.Column( + "artist_id", + sa.Integer, + sa.ForeignKey("artist.id", ondelete="CASCADE"), + primary_key=True, + ), + sa.Column( + "last_viewed_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + ) + # Seed: every existing artist starts "fully caught up". Without this, + # every operator with N artists would see N badges (worth of every + # image ever imported) on first deploy. + op.execute( + "INSERT INTO artist_visit (artist_id, last_viewed_at) " + "SELECT id, NOW() FROM artist" + ) + + +def downgrade() -> None: + op.drop_table("artist_visit") diff --git a/alembic/versions/0035_image_record_effective_date.py b/alembic/versions/0035_image_record_effective_date.py new file mode 100644 index 0000000..586cf51 --- /dev/null +++ b/alembic/versions/0035_image_record_effective_date.py @@ -0,0 +1,70 @@ +"""image_record.effective_date: materialized gallery sort key + index + +Revision ID: 0035 +Revises: 0034 +Create Date: 2026-06-04 + +The gallery ordered/cursored on COALESCE(post.post_date, +image_record.created_at) across the Post outer join. That expression spans +two tables, so no index can serve it — every /scroll sorted a large slice +of the library, and the frontend fired ten of them serially per initial +load. Materialize the value into image_record.effective_date and index +(effective_date DESC, id DESC) so the cursor scroll is an index range scan. + +Backfill = COALESCE(primary post's post_date, created_at) so existing rows +keep their exact ordering. New rows get the created_at-equivalent server +default; services/importer.py overrides it with the post's date when a +primary post with a date is linked. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0035" +down_revision: Union[str, None] = "0034" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # Add nullable first so the backfill can populate before NOT NULL. + op.add_column( + "image_record", + sa.Column("effective_date", sa.DateTime(timezone=True), nullable=True), + ) + # Pure set-based UPDATEs (no per-row params) — immune to the 65535 + # bind-parameter ceiling regardless of library size. + op.execute( + """ + UPDATE image_record AS ir + SET effective_date = COALESCE(p.post_date, ir.created_at) + FROM post AS p + WHERE ir.primary_post_id = p.id + """ + ) + op.execute( + """ + UPDATE image_record + SET effective_date = created_at + WHERE effective_date IS NULL + """ + ) + op.alter_column( + "image_record", + "effective_date", + nullable=False, + server_default=sa.text("now()"), + ) + # DESC/DESC matches the gallery's ORDER BY effective_date DESC, id DESC + # so the scroll is a forward index scan; raw SQL because alembic's + # column list doesn't express per-column DESC cleanly. + op.execute( + "CREATE INDEX ix_image_record_effective_date " + "ON image_record (effective_date DESC, id DESC)" + ) + + +def downgrade() -> None: + op.drop_index("ix_image_record_effective_date", table_name="image_record") + op.drop_column("image_record", "effective_date") diff --git a/alembic/versions/0036_siglip_embedding_hnsw_index.py b/alembic/versions/0036_siglip_embedding_hnsw_index.py new file mode 100644 index 0000000..a8c1251 --- /dev/null +++ b/alembic/versions/0036_siglip_embedding_hnsw_index.py @@ -0,0 +1,41 @@ +"""image_record.siglip_embedding: HNSW cosine index for "more like this" + +Revision ID: 0036 +Revises: 0035 +Create Date: 2026-06-04 + +Gallery Phase 3 (visual similarity search) ranks images by +`siglip_embedding.cosine_distance(source_embedding)`. Without an index that's +a sequential scan computing a 1152-dim distance for every row — fine at small +scale, but it grows linearly with the library. Add an HNSW index with +`vector_cosine_ops` so the top-N nearest search is sub-50ms ANN. + +1152 dims is under pgvector's 2000-dim HNSW limit, so HNSW (no training, +better recall than IVFFlat) is the right choice. ONE-TIME COST: building the +index over the existing embeddings (~57k vectors on the operator's library) +locks image_record for ~30-60s during this migration on deploy — acceptable +for a single-operator homelab. NULL embeddings (videos / not-yet-embedded +rows) are simply not indexed. +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0036" +down_revision: Union[str, None] = "0035" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # Raw SQL: alembic's create_index doesn't express the `USING hnsw (... + # vector_cosine_ops)` access-method + opclass cleanly. Must match the + # query's cosine_distance operator class to be usable by the planner. + op.execute( + "CREATE INDEX ix_image_record_siglip_hnsw " + "ON image_record USING hnsw (siglip_embedding vector_cosine_ops)" + ) + + +def downgrade() -> None: + op.drop_index("ix_image_record_siglip_hnsw", table_name="image_record") diff --git a/alembic/versions/0037_patreon_seen_media.py b/alembic/versions/0037_patreon_seen_media.py new file mode 100644 index 0000000..255484e --- /dev/null +++ b/alembic/versions/0037_patreon_seen_media.py @@ -0,0 +1,53 @@ +"""patreon_seen_media: per-source ledger of already-ingested Patreon media + +Revision ID: 0037 +Revises: 0036 +Create Date: 2026-06-05 + +Native Patreon ingester (build step 2a). Replaces gallery-dl's +archive.sqlite3 with our own queryable table. The downloader upserts one +row per (source, media) so routine walks skip media we've already +processed; a future "recovery" mode bypasses the ledger to re-walk. + +`filehash` is a 32-hex Patreon CDN MD5, OR a video sentinel of the form +``video::`` — hence String(128). The unique +constraint on (source_id, filehash) is the dedup upsert key. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0037" +down_revision: Union[str, None] = "0036" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "patreon_seen_media", + sa.Column("id", sa.Integer, primary_key=True), + sa.Column( + "source_id", + sa.Integer, + sa.ForeignKey("source.id", ondelete="CASCADE"), + nullable=False, + index=True, + ), + sa.Column("filehash", sa.String(128), nullable=False), + sa.Column("post_id", sa.String(64), nullable=True), + sa.Column( + "seen_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.UniqueConstraint( + "source_id", "filehash", name="uq_patreon_seen_media_source_id" + ), + ) + + +def downgrade() -> None: + op.drop_table("patreon_seen_media") diff --git a/alembic/versions/0038_patreon_failed_media.py b/alembic/versions/0038_patreon_failed_media.py new file mode 100644 index 0000000..e907ae1 --- /dev/null +++ b/alembic/versions/0038_patreon_failed_media.py @@ -0,0 +1,58 @@ +"""patreon_failed_media: per-source dead-letter ledger for failing Patreon media + +Revision ID: 0038 +Revises: 0037 +Create Date: 2026-06-06 + +Plan #705 (#7). Media that keeps failing to download/validate (404'd CDN, +deleted post, geo-blocked Mux, persistently-corrupt bytes) gets recorded here +with an attempt counter; once it crosses the dead-letter threshold the ingester +skips it on routine walks (recovery still re-attempts). A clean download clears +the row. UNIQUE (source_id, filehash) is the upsert key (same media key the +seen-ledger uses). +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0038" +down_revision: Union[str, None] = "0037" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "patreon_failed_media", + sa.Column("id", sa.Integer, primary_key=True), + sa.Column( + "source_id", + sa.Integer, + sa.ForeignKey("source.id", ondelete="CASCADE"), + nullable=False, + index=True, + ), + sa.Column("filehash", sa.String(128), nullable=False), + sa.Column("attempts", sa.Integer, nullable=False, server_default="1"), + sa.Column("last_error", sa.Text, nullable=True), + sa.Column( + "first_failed_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.Column( + "last_failed_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.UniqueConstraint( + "source_id", "filehash", name="uq_patreon_failed_media_source_id" + ), + ) + + +def downgrade() -> None: + op.drop_table("patreon_failed_media") diff --git a/alembic/versions/0039_library_audit_resume.py b/alembic/versions/0039_library_audit_resume.py new file mode 100644 index 0000000..6cfb8f9 --- /dev/null +++ b/alembic/versions/0039_library_audit_resume.py @@ -0,0 +1,40 @@ +"""library_audit_run: resume cursor + progress timestamp for chunked scans + +Revision ID: 0039 +Revises: 0038 +Create Date: 2026-06-07 + +scan_library_for_rule used to run one 2h pass that timed out on large libraries +and monopolized the concurrency-1 maintenance queue (operator-flagged). It now +runs short time-boxed chunks that re-enqueue: `resume_after_id` persists the +keyset cursor so the next chunk continues where it left off, and +`last_progress_at` lets the recovery sweep tell a progressing multi-chunk audit +from a genuinely stuck one. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0039" +down_revision: Union[str, None] = "0038" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "library_audit_run", + sa.Column( + "resume_after_id", sa.Integer, nullable=False, server_default="0" + ), + ) + op.add_column( + "library_audit_run", + sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True), + ) + + +def downgrade() -> None: + op.drop_column("library_audit_run", "last_progress_at") + op.drop_column("library_audit_run", "resume_after_id") diff --git a/alembic/versions/0040_series_chapters.py b/alembic/versions/0040_series_chapters.py new file mode 100644 index 0000000..a0808df --- /dev/null +++ b/alembic/versions/0040_series_chapters.py @@ -0,0 +1,108 @@ +"""series chapters: chapter layer over series_page (FC-6.1) + +Revision ID: 0040 +Revises: 0039 +Create Date: 2026-06-07 + +A series (Tag kind='series') gains an ordered chapter layer. Reading order +becomes (series_chapter.chapter_number, series_page.page_number). Every existing +series is backfilled into a single auto-chapter (chapter_number=1) holding its +current flat pages, so no data is lost and the old flat ordering is preserved. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0040" +down_revision: Union[str, None] = "0039" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "series_chapter", + sa.Column("id", sa.Integer, primary_key=True), + sa.Column( + "series_tag_id", + sa.Integer, + sa.ForeignKey("tag.id", ondelete="CASCADE"), + nullable=False, + ), + sa.Column("chapter_number", sa.Integer, nullable=False), + sa.Column("title", sa.Text, nullable=True), + sa.Column( + "is_placeholder", sa.Boolean, nullable=False, server_default="false" + ), + sa.Column("stated_page_start", sa.Integer, nullable=True), + sa.Column("stated_page_end", sa.Integer, nullable=True), + sa.Column( + "created_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("now()"), + ), + sa.Column( + "updated_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("now()"), + ), + ) + op.create_index( + "ix_series_chapter_series_tag_id", "series_chapter", ["series_tag_id"] + ) + + # New columns on series_page; chapter_id starts nullable so we can backfill. + op.add_column( + "series_page", sa.Column("chapter_id", sa.Integer, nullable=True) + ) + op.add_column( + "series_page", sa.Column("stated_page", sa.Integer, nullable=True) + ) + + conn = op.get_bind() + # One auto-chapter per existing series (any series_tag_id present in pages). + conn.execute( + sa.text( + "INSERT INTO series_chapter " + "(series_tag_id, chapter_number, is_placeholder, created_at, updated_at) " + "SELECT DISTINCT series_tag_id, 1, false, now(), now() " + "FROM series_page" + ) + ) + # Point every existing page at its series' auto-chapter. + conn.execute( + sa.text( + "UPDATE series_page sp " + "SET chapter_id = sc.id " + "FROM series_chapter sc " + "WHERE sc.series_tag_id = sp.series_tag_id" + ) + ) + + # Now lock chapter_id down: NOT NULL + FK (cascade) + index. + op.alter_column("series_page", "chapter_id", nullable=False) + op.create_foreign_key( + "fk_series_page_chapter_id", + "series_page", + "series_chapter", + ["chapter_id"], + ["id"], + ondelete="CASCADE", + ) + op.create_index( + "ix_series_page_chapter_id", "series_page", ["chapter_id"] + ) + + +def downgrade() -> None: + op.drop_index("ix_series_page_chapter_id", table_name="series_page") + op.drop_constraint( + "fk_series_page_chapter_id", "series_page", type_="foreignkey" + ) + op.drop_column("series_page", "stated_page") + op.drop_column("series_page", "chapter_id") + op.drop_index("ix_series_chapter_series_tag_id", table_name="series_chapter") + op.drop_table("series_chapter") diff --git a/alembic/versions/0041_series_suggestions.py b/alembic/versions/0041_series_suggestions.py new file mode 100644 index 0000000..51b690a --- /dev/null +++ b/alembic/versions/0041_series_suggestions.py @@ -0,0 +1,98 @@ +"""series suggestions: assisted-continuation matcher (FC-6.3) + +Revision ID: 0041 +Revises: 0040 +Create Date: 2026-06-07 + +A confirm-only queue of "this post may continue this series" hints, plus two +import_settings knobs (enable + score threshold) for the matcher. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0041" +down_revision: Union[str, None] = "0040" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "series_suggestion", + sa.Column("id", sa.Integer, primary_key=True), + sa.Column( + "post_id", + sa.Integer, + sa.ForeignKey("post.id", ondelete="CASCADE"), + nullable=False, + ), + sa.Column( + "series_tag_id", + sa.Integer, + sa.ForeignKey("tag.id", ondelete="CASCADE"), + nullable=False, + ), + sa.Column("score", sa.Float, nullable=False), + sa.Column("signals", sa.JSON, nullable=True), + sa.Column( + "status", sa.String(16), nullable=False, server_default="pending" + ), + sa.Column( + "created_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("now()"), + ), + sa.Column( + "updated_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("now()"), + ), + sa.UniqueConstraint( + "post_id", "series_tag_id", name="uq_series_suggestion_post_series" + ), + ) + op.create_index( + "ix_series_suggestion_post_id", "series_suggestion", ["post_id"] + ) + op.create_index( + "ix_series_suggestion_series_tag_id", + "series_suggestion", + ["series_tag_id"], + ) + op.create_index( + "ix_series_suggestion_status", "series_suggestion", ["status"] + ) + + op.add_column( + "import_settings", + sa.Column( + "series_suggest_enabled", + sa.Boolean, + nullable=False, + server_default=sa.true(), + ), + ) + op.add_column( + "import_settings", + sa.Column( + "series_suggest_threshold", + sa.Float, + nullable=False, + server_default="0.5", + ), + ) + + +def downgrade() -> None: + op.drop_column("import_settings", "series_suggest_threshold") + op.drop_column("import_settings", "series_suggest_enabled") + op.drop_index("ix_series_suggestion_status", table_name="series_suggestion") + op.drop_index( + "ix_series_suggestion_series_tag_id", table_name="series_suggestion" + ) + op.drop_index("ix_series_suggestion_post_id", table_name="series_suggestion") + op.drop_table("series_suggestion") diff --git a/alembic/versions/0042_series_chapter_stated_part.py b/alembic/versions/0042_series_chapter_stated_part.py new file mode 100644 index 0000000..f898e56 --- /dev/null +++ b/alembic/versions/0042_series_chapter_stated_part.py @@ -0,0 +1,32 @@ +"""series chapter stated_part: operator-facing Part N label (FC-6.4) + +Revision ID: 0042 +Revises: 0041 +Create Date: 2026-06-07 + +A chapter's positional chapter_number is auto-managed (rewritten 1..N on +reorder/delete), so it can't double as the installment number the operator wants +to type (e.g. a series authored from a post that is Part 2). Add a nullable +stated_part alongside it — the same split as series_page.page_number (order) vs +series_page.stated_page (printed number). Nullable; the UI falls back to +chapter_number when unset. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0042" +down_revision: Union[str, None] = "0041" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "series_chapter", sa.Column("stated_part", sa.Integer, nullable=True) + ) + + +def downgrade() -> None: + op.drop_column("series_chapter", "stated_part") diff --git a/alembic/versions/0043_post_attachment_per_post_unique.py b/alembic/versions/0043_post_attachment_per_post_unique.py new file mode 100644 index 0000000..e8e38ce --- /dev/null +++ b/alembic/versions/0043_post_attachment_per_post_unique.py @@ -0,0 +1,62 @@ +"""post_attachment: per-post sha uniqueness (empty-post flood fix) + +Revision ID: 0043 +Revises: 0042 +Create Date: 2026-06-08 + +PostAttachment.sha256 was GLOBALLY unique, so a non-art file the creator attaches +to many posts (a standard pdf/zip/link-card) only ever got ONE row — on the first +post — leaving every later post a bare shell (no image, no attachment). The native +Patreon backfill of Anduo surfaced 1589 such shells (operator-flagged 2026-06-08). + +Switch to PER-POST uniqueness: the on-disk blob stays sha-deduped, but each post +gets its own row. Replace the unique sha256 index with a plain lookup index plus +two partial uniques — (post_id, sha256) for real posts and (sha256) for the +NULL-post filesystem case (still one row per file there). + +Existing data has ≤1 row per sha (the old global unique), so the new partial +uniques can't be violated on upgrade — no data backfill needed here. The bare-post +shells themselves are removed by the separate prune-empty-posts cleanup tool. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0043" +down_revision: Union[str, None] = "0042" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # Drop the global unique index; recreate it as a plain (non-unique) lookup + # index so sha-based reads keep their index (matches the model's index=True). + op.drop_index("ix_post_attachment_sha256", table_name="post_attachment") + op.create_index( + "ix_post_attachment_sha256", "post_attachment", ["sha256"], + ) + op.create_index( + "uq_post_attachment_post_sha", "post_attachment", + ["post_id", "sha256"], unique=True, + postgresql_where=sa.text("post_id IS NOT NULL"), + ) + op.create_index( + "uq_post_attachment_null_post_sha", "post_attachment", + ["sha256"], unique=True, + postgresql_where=sa.text("post_id IS NULL"), + ) + + +def downgrade() -> None: + op.drop_index( + "uq_post_attachment_null_post_sha", table_name="post_attachment" + ) + op.drop_index( + "uq_post_attachment_post_sha", table_name="post_attachment" + ) + op.drop_index("ix_post_attachment_sha256", table_name="post_attachment") + op.create_index( + "ix_post_attachment_sha256", "post_attachment", ["sha256"], + unique=True, + ) diff --git a/alembic/versions/0044_ml_settings_tagger_store_floor.py b/alembic/versions/0044_ml_settings_tagger_store_floor.py new file mode 100644 index 0000000..e019e36 --- /dev/null +++ b/alembic/versions/0044_ml_settings_tagger_store_floor.py @@ -0,0 +1,37 @@ +"""ml_settings.tagger_store_floor + +The ingest confidence floor below which tagger predictions are not stored, +promoted from the TAGGER_STORE_FLOOR env var to a DB-backed, UI-tunable +setting. Default 0.70 (was an env default of 0.05): the suggestion path +already filters at 0.70 and the centroid/learned path covers low-confidence +preferred tags, so the sub-0.70 tail was redundant weight — it had grown +image_record's TOAST to ~100 GB. See plan-task #764. + +Revision ID: 0044 +Revises: 0043 +Create Date: 2026-06-10 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0044" +down_revision: Union[str, None] = "0043" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "tagger_store_floor", sa.Float(), + nullable=False, server_default="0.7", + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "tagger_store_floor") diff --git a/alembic/versions/0045_image_prediction_table.py b/alembic/versions/0045_image_prediction_table.py new file mode 100644 index 0000000..df11de2 --- /dev/null +++ b/alembic/versions/0045_image_prediction_table.py @@ -0,0 +1,69 @@ +"""image_prediction table (DDL only — backfill runs as a background task) + +Normalizes the per-image tagger predictions out of the JSON blob into a +queryable table (#768). This migration creates ONLY the table + indexes — it +is pure DDL and commits instantly, so web boots immediately. + +The data backfill from the existing image_record.tagger_predictions JSON is +deliberately NOT done here. Doing it inline made the whole migration one +transaction over the ~100 GB TOAST: nothing committed until the very end, it +was invisible/unmonitorable mid-run, and an early MATERIALIZED-CTE form spilled +the full 100 GB to temp. Instead the backfill is the +backend.app.tasks.admin.backfill_image_predictions_task — batched by id window, +committed per chunk (visible progress + resumable), idempotent +(ON CONFLICT DO NOTHING). Trigger it from Settings → Maintenance once web is up. + +The old image_record.tagger_predictions column is left in place (vestigial) and +dropped in a follow-up once the backfill + code cutover are verified — dropping +it needs an ACCESS EXCLUSIVE lock on the hot image_record table (the 0044 lock +class), so it's deferred to a quiesced-worker window. + +Revision ID: 0045 +Revises: 0044 +Create Date: 2026-06-10 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0045" +down_revision: Union[str, None] = "0044" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "image_prediction", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "image_record_id", sa.Integer(), + sa.ForeignKey("image_record.id", ondelete="CASCADE"), + nullable=False, + ), + sa.Column("raw_name", sa.String(length=255), nullable=False), + sa.Column("category", sa.String(length=64), nullable=False), + sa.Column("score", sa.Float(), nullable=False), + sa.UniqueConstraint( + "image_record_id", "raw_name", name="image_raw_name", + ), + ) + op.create_index( + "ix_image_prediction_image", "image_prediction", ["image_record_id"], + ) + op.create_index( + "ix_image_prediction_name_score", "image_prediction", + ["raw_name", "score"], + ) + # No data backfill here — see the module docstring. The one-time copy from + # image_record.tagger_predictions runs as backfill_image_predictions_task + # (batched, resumable, idempotent), kept out of this transaction so web boots + # without waiting on a ~100 GB pass. + + +def downgrade() -> None: + op.drop_index("ix_image_prediction_name_score", "image_prediction") + op.drop_index("ix_image_prediction_image", "image_prediction") + op.drop_table("image_prediction") diff --git a/alembic/versions/0046_drop_tagger_predictions.py b/alembic/versions/0046_drop_tagger_predictions.py new file mode 100644 index 0000000..84e543a --- /dev/null +++ b/alembic/versions/0046_drop_tagger_predictions.py @@ -0,0 +1,43 @@ +"""drop image_record.tagger_predictions (predictions normalized to image_prediction) + +Final step of #768. The per-tag predictions now live in the image_prediction +table (backfilled from the JSON, read by suggestions + allowlist, written by +tag_and_embed). The old JSON column is dead weight — and it's the ~100 GB of +sub-0.70 score tail that bloated image_record's TOAST and broke DB backups +(#739). Dropping it is a fast catalog change; it does NOT reclaim the disk on +its own — run `VACUUM FULL image_record` (or pg_repack) afterward, off-hours, +to return the space to the OS so backups go small. + +DROP COLUMN needs a brief ACCESS EXCLUSIVE lock on image_record; env.py's +lock_timeout guards it, so quiesce the ml-worker if a tagging run is in flight +(see the migration-lock reference). tagger_model_version is kept — it's the +"has this been tagged / is it current?" signal the backfill sweep reads. + +Revision ID: 0046 +Revises: 0045 +Create Date: 2026-06-11 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0046" +down_revision: Union[str, None] = "0045" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.drop_column("image_record", "tagger_predictions") + + +def downgrade() -> None: + # Re-add the column empty. The JSON data is not restored (it lived only in + # this column); a downgrade would re-tag or backfill from image_prediction + # separately if ever needed. + op.add_column( + "image_record", + sa.Column("tagger_predictions", sa.JSON(), nullable=True), + ) diff --git a/alembic/versions/0047_series_chapter_dividers.py b/alembic/versions/0047_series_chapter_dividers.py new file mode 100644 index 0000000..5afd074 --- /dev/null +++ b/alembic/versions/0047_series_chapter_dividers.py @@ -0,0 +1,175 @@ +"""series chapters become cosmetic dividers; pages become one series-global run + +FC-6.x reframe (#789). A series is now ONE flat, series-global ordered run of +pages; chapters stop owning pages and become labeled dividers anchored to the +page that begins them. + +Migration (order matters — series_page.chapter_id cascades, so it must be +dropped BEFORE any chapter row is deleted, or pages would cascade away): + a. Renumber series_page.page_number to a series-global 1..N (ordered by the + OLD (chapter_number, page_number)). + b. Add series_chapter.anchor_page_id and populate it with each chapter's first + page (lowest new page_number). + c. Drop series_page.chapter_id (severs the cascade link). + d. Prune chapters that shouldn't become dividers: empty/placeholder ones (no + anchor) and the redundant unlabeled chapter that would sit at page 1. + e. Reshape series_chapter into the divider: drop chapter_number, + is_placeholder, stated_page_start/end; make anchor_page_id NOT NULL + + UNIQUE + FK→series_page ON DELETE CASCADE. + +Revision ID: 0047 +Revises: 0046 +Create Date: 2026-06-11 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0047" +down_revision: Union[str, None] = "0046" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # a. series-global page numbering, preserving the old reading order. + op.execute( + """ + WITH ordered AS ( + SELECT sp.id, + ROW_NUMBER() OVER ( + PARTITION BY sp.series_tag_id + ORDER BY sc.chapter_number, sp.page_number, sp.id + ) AS rn + FROM series_page sp + JOIN series_chapter sc ON sc.id = sp.chapter_id + ) + UPDATE series_page sp + SET page_number = ordered.rn + FROM ordered + WHERE sp.id = ordered.id + """ + ) + + # b. anchor each existing chapter at its first page (lowest new page_number). + op.add_column( + "series_chapter", + sa.Column("anchor_page_id", sa.Integer(), nullable=True), + ) + op.execute( + """ + WITH firsts AS ( + SELECT DISTINCT ON (sp.chapter_id) + sp.chapter_id, sp.id AS page_id + FROM series_page sp + ORDER BY sp.chapter_id, sp.page_number, sp.id + ) + UPDATE series_chapter sc + SET anchor_page_id = firsts.page_id + FROM firsts + WHERE firsts.chapter_id = sc.id + """ + ) + + # c. sever the ownership link (drops the FK + index with the column) BEFORE + # pruning chapters, so deleting a chapter can't cascade-delete its pages. + op.drop_column("series_page", "chapter_id") + + # d. prune chapters that don't become dividers: placeholders / empty ones + # (no anchor), and the unlabeled chapter that would land redundantly at + # page 1 (the series just starts — no divider needed there). + op.execute( + """ + DELETE FROM series_chapter sc + USING ( + SELECT sc2.id + FROM series_chapter sc2 + LEFT JOIN series_page sp ON sp.id = sc2.anchor_page_id + WHERE sc2.anchor_page_id IS NULL + OR (sp.page_number = 1 + AND sc2.title IS NULL + AND sc2.stated_part IS NULL) + ) gone + WHERE sc.id = gone.id + """ + ) + + # e. reshape into the divider model. + op.drop_column("series_chapter", "chapter_number") + op.drop_column("series_chapter", "is_placeholder") + op.drop_column("series_chapter", "stated_page_start") + op.drop_column("series_chapter", "stated_page_end") + op.alter_column("series_chapter", "anchor_page_id", nullable=False) + op.create_unique_constraint( + "uq_series_chapter_anchor_page", "series_chapter", ["anchor_page_id"] + ) + op.create_foreign_key( + "fk_series_chapter_anchor_page", + "series_chapter", + "series_page", + ["anchor_page_id"], + ["id"], + ondelete="CASCADE", + ) + + +def downgrade() -> None: + # Lossy: dividers can't be reconstructed as owning chapters. Collapse back to + # exactly one chapter per series that owns all its pages in order. + op.add_column( + "series_page", sa.Column("chapter_id", sa.Integer(), nullable=True) + ) + op.drop_constraint( + "fk_series_chapter_anchor_page", "series_chapter", type_="foreignkey" + ) + op.drop_constraint( + "uq_series_chapter_anchor_page", "series_chapter", type_="unique" + ) + op.drop_column("series_chapter", "anchor_page_id") + op.add_column( + "series_chapter", + sa.Column( + "chapter_number", sa.Integer(), nullable=False, server_default="1" + ), + ) + op.add_column( + "series_chapter", + sa.Column( + "is_placeholder", sa.Boolean(), nullable=False, + server_default="false", + ), + ) + op.add_column( + "series_chapter", + sa.Column("stated_page_start", sa.Integer(), nullable=True), + ) + op.add_column( + "series_chapter", + sa.Column("stated_page_end", sa.Integer(), nullable=True), + ) + op.execute("DELETE FROM series_chapter") + op.execute( + """ + INSERT INTO series_chapter (series_tag_id, chapter_number) + SELECT DISTINCT series_tag_id, 1 FROM series_page + """ + ) + op.execute( + """ + UPDATE series_page sp + SET chapter_id = sc.id + FROM series_chapter sc + WHERE sc.series_tag_id = sp.series_tag_id + """ + ) + op.alter_column("series_page", "chapter_id", nullable=False) + op.create_foreign_key( + "fk_series_page_chapter", + "series_page", + "series_chapter", + ["chapter_id"], + ["id"], + ondelete="CASCADE", + ) diff --git a/alembic/versions/0048_series_page_pending_status.py b/alembic/versions/0048_series_page_pending_status.py new file mode 100644 index 0000000..25944a5 --- /dev/null +++ b/alembic/versions/0048_series_page_pending_status.py @@ -0,0 +1,45 @@ +"""series_page pending staging: status + nullable page_number (#789 Phase 2) + +Pages added from a post no longer append straight into the run — they land +'pending' with a NULL page_number, staged grouped by their source post so the +operator can drop junk (text-free alts, bumpers) and place the keepers into the +sequence. A page only gets a series-global page_number once it's 'placed'. + +Revision ID: 0048 +Revises: 0047 +Create Date: 2026-06-11 + +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0048" +down_revision: Union[str, None] = "0047" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "series_page", + sa.Column( + "status", sa.String(length=16), nullable=False, + server_default="placed", + ), + ) + op.alter_column( + "series_page", "page_number", + existing_type=sa.Integer(), nullable=True, + ) + + +def downgrade() -> None: + # Lossy: pending pages are unsorted staging rows with no order — drop them. + op.execute("DELETE FROM series_page WHERE status = 'pending'") + op.alter_column( + "series_page", "page_number", + existing_type=sa.Integer(), nullable=False, + ) + op.drop_column("series_page", "status") diff --git a/alembic/versions/0049_external_link_table.py b/alembic/versions/0049_external_link_table.py new file mode 100644 index 0000000..373c807 --- /dev/null +++ b/alembic/versions/0049_external_link_table.py @@ -0,0 +1,90 @@ +"""external_link table — off-platform file-host links found in post bodies + +Creators host the real files on mega.nz / Google Drive / MediaFire / Dropbox / +Pixeldrain and link them in the post text. This table records each such link +(so nothing is silently dropped), and doubles as the dedup + dead-letter ledger +the download worker (a later slice) walks. `url` keeps the FULL link including +the `#fragment` — mega.nz's decryption key lives there; truncating it makes the +file undownloadable. + +CHECK whitelists for host + status include the full enum up front (incl. the +download-worker statuses) so the worker slice needs no constraint migration. + +Revision ID: 0049 +Revises: 0048 +Create Date: 2026-06-14 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0049" +down_revision: Union[str, None] = "0048" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "external_link", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "post_id", sa.Integer(), + sa.ForeignKey("post.id", ondelete="CASCADE"), nullable=False, + ), + sa.Column( + "artist_id", sa.Integer(), + sa.ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, + ), + sa.Column("host", sa.String(length=16), nullable=False), + sa.Column("url", sa.Text(), nullable=False), + sa.Column("label", sa.Text(), nullable=True), + sa.Column( + "status", sa.String(length=16), nullable=False, + server_default="pending", + ), + sa.Column("attempts", sa.Integer(), nullable=False, server_default="0"), + sa.Column("last_error", sa.Text(), nullable=True), + sa.Column( + "attachment_id", sa.Integer(), + sa.ForeignKey("post_attachment.id", ondelete="SET NULL"), + nullable=True, + ), + sa.Column( + "created_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column("completed_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("duration_seconds", sa.Float(), nullable=True), + sa.CheckConstraint( + "host IN ('mega','gdrive','mediafire','dropbox','pixeldrain')", + name="ck_external_link_host", + ), + sa.CheckConstraint( + "status IN ('pending','downloading','downloaded','failed'," + "'skipped','dead')", + name="ck_external_link_status", + ), + ) + op.create_index( + "ix_external_link_post_id", "external_link", ["post_id"], + ) + op.create_index( + "ix_external_link_artist_id", "external_link", ["artist_id"], + ) + op.create_index( + "ix_external_link_status", "external_link", ["status"], + ) + op.create_index( + "uq_external_link_post_url", "external_link", ["post_id", "url"], + unique=True, + ) + + +def downgrade() -> None: + op.drop_index("uq_external_link_post_url", table_name="external_link") + op.drop_index("ix_external_link_status", table_name="external_link") + op.drop_index("ix_external_link_artist_id", table_name="external_link") + op.drop_index("ix_external_link_post_id", table_name="external_link") + op.drop_table("external_link") diff --git a/alembic/versions/0050_external_link_host_toggles.py b/alembic/versions/0050_external_link_host_toggles.py new file mode 100644 index 0000000..ac78e75 --- /dev/null +++ b/alembic/versions/0050_external_link_host_toggles.py @@ -0,0 +1,38 @@ +"""import_settings: per-host enable toggles for external file-host downloads + +Operator levers (#830): disable a single host (e.g. mega.nz when it's +rate-limiting/banning) without touching the others. The worker reads these via +getattr and defaults to enabled, so the toggles default TRUE (works out of the +box, rule #26). + +Revision ID: 0050 +Revises: 0049 +Create Date: 2026-06-14 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0050" +down_revision: Union[str, None] = "0049" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +_HOSTS = ("mega", "gdrive", "mediafire", "dropbox", "pixeldrain") + + +def upgrade() -> None: + for host in _HOSTS: + op.add_column( + "import_settings", + sa.Column( + f"extdl_{host}_enabled", sa.Boolean(), nullable=False, + server_default=sa.true(), + ), + ) + + +def downgrade() -> None: + for host in _HOSTS: + op.drop_column("import_settings", f"extdl_{host}_enabled") diff --git a/alembic/versions/0051_image_source_provenance.py b/alembic/versions/0051_image_source_provenance.py new file mode 100644 index 0000000..595077d --- /dev/null +++ b/alembic/versions/0051_image_source_provenance.py @@ -0,0 +1,38 @@ +"""image_record: source_url + source_filehash (inline-image localization) + +#830 Phase 2. To render a post body faithfully we serve LOCAL copies of inline +images instead of hotlinking the public CDN. The join key between a body +`` and the local file is the CDN's 32-hex filehash (the same +identity extract_media dedups by). Persist it (indexed) plus the full source +URL for provenance/debugging. Both NULL for filesystem-imported / pre-existing +rows — those fall back to hotlinking until re-downloaded. + +Revision ID: 0051 +Revises: 0050 +Create Date: 2026-06-14 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0051" +down_revision: Union[str, None] = "0050" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column("image_record", sa.Column("source_url", sa.Text(), nullable=True)) + op.add_column( + "image_record", sa.Column("source_filehash", sa.String(length=32), nullable=True) + ) + op.create_index( + "ix_image_record_source_filehash", "image_record", ["source_filehash"] + ) + + +def downgrade() -> None: + op.drop_index("ix_image_record_source_filehash", table_name="image_record") + op.drop_column("image_record", "source_filehash") + op.drop_column("image_record", "source_url") diff --git a/alembic/versions/0052_image_duration_seconds.py b/alembic/versions/0052_image_duration_seconds.py new file mode 100644 index 0000000..ec2a180 --- /dev/null +++ b/alembic/versions/0052_image_duration_seconds.py @@ -0,0 +1,32 @@ +"""image_record: duration_seconds (Tier-1 video near-dup key) + +#871. Videos previously deduped on sha256 only (pHash is images-only), so a +different encode/remux of the same video imported as a distinct record. Persist +the container duration so the importer can treat same-artist videos with matching +duration (+ aspect ratio) as the same content and dedup/supersede like images. +NULL for images and for video rows imported before this column existed (a +backfill re-probes those so they participate in dedup). + +Revision ID: 0052 +Revises: 0051 +Create Date: 2026-06-16 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0052" +down_revision: Union[str, None] = "0051" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "image_record", sa.Column("duration_seconds", sa.Float(), nullable=True) + ) + + +def downgrade() -> None: + op.drop_column("image_record", "duration_seconds") diff --git a/alembic/versions/0053_ml_settings_video_tagging.py b/alembic/versions/0053_ml_settings_video_tagging.py new file mode 100644 index 0000000..1f192a4 --- /dev/null +++ b/alembic/versions/0053_ml_settings_video_tagging.py @@ -0,0 +1,49 @@ +"""ml_settings: video tagging knobs (cadence sampling + noise floor) + +#747. Video tag quality/perf: sample frames at a fixed cadence (interval) so a +tag's frame-presence reflects real screen time, cap total frames so long videos +stay bounded, and keep a tag only if it appears in >= min_tag_frames sampled +frames. Operator-tunable via Settings → ML (replaces the VIDEO_ML_FRAMES env var). + +Revision ID: 0053 +Revises: 0052 +Create Date: 2026-06-16 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0053" +down_revision: Union[str, None] = "0052" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "video_frame_interval_seconds", sa.Float(), nullable=False, + server_default="4.0", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "video_max_frames", sa.Integer(), nullable=False, server_default="64", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "video_min_tag_frames", sa.Integer(), nullable=False, + server_default="3", + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "video_min_tag_frames") + op.drop_column("ml_settings", "video_max_frames") + op.drop_column("ml_settings", "video_frame_interval_seconds") diff --git a/alembic/versions/0054_subscribestar_ledgers.py b/alembic/versions/0054_subscribestar_ledgers.py new file mode 100644 index 0000000..59972ae --- /dev/null +++ b/alembic/versions/0054_subscribestar_ledgers.py @@ -0,0 +1,82 @@ +"""subscribestar_seen_media + subscribestar_failed_media: per-source ledgers + +Revision ID: 0054 +Revises: 0053 +Create Date: 2026-06-17 + +SubscribeStar native ingester (phase 1 of the gallery-dl → native-core +migration). Mirrors the Patreon ledger tables (0037/0038): a seen-ledger so +routine walks skip already-ingested media (recovery bypasses it) and a +dead-letter ledger so persistently-failing media stops re-burning backfill +chunks. `filehash` is a CDN content hash when present, else a synthesized +``:`` key — hence String(128). UNIQUE (source_id, filehash) +is the upsert key on each. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0054" +down_revision: Union[str, None] = "0053" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "subscribestar_seen_media", + sa.Column("id", sa.Integer, primary_key=True), + sa.Column( + "source_id", + sa.Integer, + sa.ForeignKey("source.id", ondelete="CASCADE"), + nullable=False, + index=True, + ), + sa.Column("filehash", sa.String(128), nullable=False), + sa.Column("post_id", sa.String(64), nullable=True), + sa.Column( + "seen_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.UniqueConstraint( + "source_id", "filehash", name="uq_subscribestar_seen_media_source_id" + ), + ) + op.create_table( + "subscribestar_failed_media", + sa.Column("id", sa.Integer, primary_key=True), + sa.Column( + "source_id", + sa.Integer, + sa.ForeignKey("source.id", ondelete="CASCADE"), + nullable=False, + index=True, + ), + sa.Column("filehash", sa.String(128), nullable=False), + sa.Column("attempts", sa.Integer, nullable=False, server_default="1"), + sa.Column("last_error", sa.Text, nullable=True), + sa.Column( + "first_failed_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.Column( + "last_failed_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.UniqueConstraint( + "source_id", "filehash", name="uq_subscribestar_failed_media_source_id" + ), + ) + + +def downgrade() -> None: + op.drop_table("subscribestar_failed_media") + op.drop_table("subscribestar_seen_media") diff --git a/alembic/versions/0055_image_provenance_from_attachment.py b/alembic/versions/0055_image_provenance_from_attachment.py new file mode 100644 index 0000000..8b2566b --- /dev/null +++ b/alembic/versions/0055_image_provenance_from_attachment.py @@ -0,0 +1,55 @@ +"""image_provenance: from_attachment_id (which archive an image was extracted from) + +Milestone #87. When an image is pulled out of a .zip/.rar, record WHICH archive +PostAttachment it came from, so the provenance UI can show the single archive a +file lives inside instead of every attachment on the post. Nullable FK with +ON DELETE SET NULL — a loose (non-archive) download leaves it NULL, and deleting +the archive attachment forgets the linkage without destroying the (image, post) +provenance edge. Existing rows are NULL until the reextract backfill stamps them. + +Revision ID: 0055 +Revises: 0054 +Create Date: 2026-06-22 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0055" +down_revision: Union[str, None] = "0054" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "image_provenance", + sa.Column("from_attachment_id", sa.Integer(), nullable=True), + ) + op.create_index( + "ix_image_provenance_from_attachment_id", + "image_provenance", + ["from_attachment_id"], + ) + op.create_foreign_key( + "fk_image_provenance_from_attachment", + "image_provenance", + "post_attachment", + ["from_attachment_id"], + ["id"], + ondelete="SET NULL", + ) + + +def downgrade() -> None: + op.drop_constraint( + "fk_image_provenance_from_attachment", + "image_provenance", + type_="foreignkey", + ) + op.drop_index( + "ix_image_provenance_from_attachment_id", + table_name="image_provenance", + ) + op.drop_column("image_provenance", "from_attachment_id") diff --git a/alembic/versions/0056_tag_eval_run.py b/alembic/versions/0056_tag_eval_run.py new file mode 100644 index 0000000..7d8e91f --- /dev/null +++ b/alembic/versions/0056_tag_eval_run.py @@ -0,0 +1,43 @@ +"""tag_eval_run: persisted head-vs-centroid tagging eval runs (#1130) + +Milestone #114 slice 1. A long ml-queue eval whose full report must SURVIVE +navigation, so the run + report live in a row the admin card rehydrates from +(mirrors library_audit_run). running -> ready / error. + +Revision ID: 0056 +Revises: 0055 +Create Date: 2026-06-28 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from sqlalchemy.dialects.postgresql import JSONB + +revision: str = "0056" +down_revision: Union[str, None] = "0055" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "tag_eval_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("params", JSONB(), nullable=False), + sa.Column("status", sa.String(length=16), nullable=False, server_default="running"), + sa.Column( + "started_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("report", JSONB(), nullable=True), + sa.Column("error", sa.Text(), nullable=True), + sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True), + ) + op.create_index("ix_tag_eval_run_status", "tag_eval_run", ["status"]) + + +def downgrade() -> None: + op.drop_index("ix_tag_eval_run_status", table_name="tag_eval_run") + op.drop_table("tag_eval_run") diff --git a/alembic/versions/0057_tag_positive_confirmation.py b/alembic/versions/0057_tag_positive_confirmation.py new file mode 100644 index 0000000..92335c2 --- /dev/null +++ b/alembic/versions/0057_tag_positive_confirmation.py @@ -0,0 +1,40 @@ +"""tag_positive_confirmation: operator-affirmed correct positives (#1130) + +Mirror of tag_suggestion_rejection. "Keep" on a doubted positive records here so +the eval's doubts list stops resurfacing confirmed-correct images every run. + +Revision ID: 0057 +Revises: 0056 +Create Date: 2026-06-28 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0057" +down_revision: Union[str, None] = "0056" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "tag_positive_confirmation", + sa.Column( + "image_record_id", sa.Integer(), + sa.ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True, + ), + sa.Column( + "tag_id", sa.Integer(), + sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, index=True, + ), + sa.Column( + "confirmed_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + ) + + +def downgrade() -> None: + op.drop_table("tag_positive_confirmation") diff --git a/alembic/versions/0058_tag_head.py b/alembic/versions/0058_tag_head.py new file mode 100644 index 0000000..7ff45f6 --- /dev/null +++ b/alembic/versions/0058_tag_head.py @@ -0,0 +1,95 @@ +"""tag_head + head_training_run: production heads that learn from tags (#114) + +The eval (#1130) proved the frozen-embedding + trained-head spine; this lands its +production form. tag_head stores one logistic-regression head per concept (the +new suggestion source, replacing Camie + centroid); head_training_run tracks the +batch that (re)trains them. Adds two head-training tunables to ml_settings. + +Revision ID: 0058 +Revises: 0057 +Create Date: 2026-06-28 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from pgvector.sqlalchemy import Vector +from sqlalchemy.dialects.postgresql import JSONB + +revision: str = "0058" +down_revision: Union[str, None] = "0057" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +_HEAD_DIM = 1152 + + +def upgrade() -> None: + op.create_table( + "tag_head", + sa.Column( + "tag_id", sa.Integer(), + sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, + ), + sa.Column("embedding_version", sa.String(length=128), nullable=False), + sa.Column("weights", Vector(_HEAD_DIM), nullable=False), + sa.Column("bias", sa.Float(), nullable=False), + sa.Column("suggest_threshold", sa.Float(), nullable=False), + sa.Column("auto_apply_threshold", sa.Float(), nullable=True), + sa.Column("n_pos", sa.Integer(), nullable=False), + sa.Column("n_neg", sa.Integer(), nullable=False), + sa.Column("ap", sa.Float(), nullable=False), + sa.Column("precision_cv", sa.Float(), nullable=False), + sa.Column("recall", sa.Float(), nullable=False), + sa.Column( + "trained_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column("metrics", JSONB(), nullable=True), + ) + + op.create_table( + "head_training_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("params", JSONB(), nullable=False), + sa.Column( + "status", sa.String(length=16), nullable=False, + server_default="running", + ), + sa.Column( + "started_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("n_trained", sa.Integer(), nullable=True), + sa.Column("n_skipped", sa.Integer(), nullable=True), + sa.Column("error", sa.Text(), nullable=True), + sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True), + ) + op.create_index( + "ix_head_training_run_status", "head_training_run", ["status"], + ) + + # Head-training tunables on the ml_settings singleton. + op.add_column( + "ml_settings", + sa.Column( + "head_min_positives", sa.Integer(), nullable=False, + server_default="8", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "head_auto_apply_precision", sa.Float(), nullable=False, + server_default="0.97", + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "head_auto_apply_precision") + op.drop_column("ml_settings", "head_min_positives") + op.drop_index("ix_head_training_run_status", table_name="head_training_run") + op.drop_table("head_training_run") + op.drop_table("tag_head") diff --git a/alembic/versions/0059_head_auto_apply.py b/alembic/versions/0059_head_auto_apply.py new file mode 100644 index 0000000..d0bb9b8 --- /dev/null +++ b/alembic/versions/0059_head_auto_apply.py @@ -0,0 +1,70 @@ +"""head_auto_apply_run + earned-auto-apply settings (#114) + +A graduated head can apply its tag without a human, gated by a master switch + +a support floor. head_auto_apply_run tracks each sweep / dry-run preview. + +Revision ID: 0059 +Revises: 0058 +Create Date: 2026-06-29 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from sqlalchemy.dialects.postgresql import JSONB + +revision: str = "0059" +down_revision: Union[str, None] = "0058" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "head_auto_apply_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "dry_run", sa.Boolean(), nullable=False, server_default=sa.false() + ), + sa.Column("params", JSONB(), nullable=False), + sa.Column( + "status", sa.String(length=16), nullable=False, + server_default="running", + ), + sa.Column( + "started_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("n_applied", sa.Integer(), nullable=True), + sa.Column("report", JSONB(), nullable=True), + sa.Column("error", sa.Text(), nullable=True), + sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True), + ) + op.create_index( + "ix_head_auto_apply_run_status", "head_auto_apply_run", ["status"], + ) + + op.add_column( + "ml_settings", + sa.Column( + "head_auto_apply_enabled", sa.Boolean(), nullable=False, + server_default=sa.true(), # opt-out: on by default (operator-asked) + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "head_auto_apply_min_positives", sa.Integer(), nullable=False, + server_default="30", + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "head_auto_apply_min_positives") + op.drop_column("ml_settings", "head_auto_apply_enabled") + op.drop_index( + "ix_head_auto_apply_run_status", table_name="head_auto_apply_run" + ) + op.drop_table("head_auto_apply_run") diff --git a/alembic/versions/0060_head_metrics.py b/alembic/versions/0060_head_metrics.py new file mode 100644 index 0000000..e94edb8 --- /dev/null +++ b/alembic/versions/0060_head_metrics.py @@ -0,0 +1,74 @@ +"""head_metric + head_metrics_snapshot: auto-apply observability (#114) + +Running misfire/under-fire counters per concept (captured at correction time, +since image_tag.source is lost on delete) + a daily per-concept time-series so +the operator can tune the precision target + support floor from real data. + +Revision ID: 0060 +Revises: 0059 +Create Date: 2026-06-29 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0060" +down_revision: Union[str, None] = "0059" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "head_metric", + sa.Column( + "tag_id", sa.Integer(), + sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, + ), + sa.Column("n_misfires", sa.Integer(), nullable=False, server_default="0"), + sa.Column("n_underfires", sa.Integer(), nullable=False, server_default="0"), + sa.Column( + "updated_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + ) + + op.create_table( + "head_metrics_snapshot", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "tag_id", sa.Integer(), + sa.ForeignKey("tag.id", ondelete="CASCADE"), + ), + sa.Column("name", sa.String(length=255), nullable=False), + sa.Column( + "snapshot_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column("n_auto_applied", sa.Integer(), nullable=False, server_default="0"), + sa.Column("n_misfires", sa.Integer(), nullable=False, server_default="0"), + sa.Column("n_underfires", sa.Integer(), nullable=False, server_default="0"), + sa.Column("ap", sa.Float(), nullable=True), + sa.Column("precision_cv", sa.Float(), nullable=True), + sa.Column("recall", sa.Float(), nullable=True), + sa.Column("n_pos", sa.Integer(), nullable=True), + ) + op.create_index( + "ix_head_metrics_snapshot_tag_id", "head_metrics_snapshot", ["tag_id"], + ) + op.create_index( + "ix_head_metrics_snapshot_snapshot_at", "head_metrics_snapshot", + ["snapshot_at"], + ) + + +def downgrade() -> None: + op.drop_index( + "ix_head_metrics_snapshot_snapshot_at", table_name="head_metrics_snapshot" + ) + op.drop_index( + "ix_head_metrics_snapshot_tag_id", table_name="head_metrics_snapshot" + ) + op.drop_table("head_metrics_snapshot") + op.drop_table("head_metric") diff --git a/alembic/versions/0061_image_region.py b/alembic/versions/0061_image_region.py new file mode 100644 index 0000000..b3af8a9 --- /dev/null +++ b/alembic/versions/0061_image_region.py @@ -0,0 +1,59 @@ +"""image_region: detected/proposed regions + their crop embeddings (#114) + +Storage backbone of the crop pipeline. A region = normalized bbox + the crop's +embedding (CCIP for face/figure → character id; SigLIP for concept regions → +head bag-of-embeddings). Also serves as grounded-tag bbox provenance. + +Revision ID: 0061 +Revises: 0060 +Create Date: 2026-06-29 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from pgvector.sqlalchemy import Vector + +revision: str = "0061" +down_revision: Union[str, None] = "0060" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +_CCIP_DIM = 768 +_SIGLIP_DIM = 1152 + + +def upgrade() -> None: + op.create_table( + "image_region", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "image_record_id", sa.Integer(), + sa.ForeignKey("image_record.id", ondelete="CASCADE"), nullable=False, + ), + sa.Column("kind", sa.String(length=16), nullable=False), + # Video/animated: source frame timestamp (seconds); NULL for stills. + sa.Column("frame_time", sa.Float(), nullable=True), + sa.Column("rx", sa.Float(), nullable=False), + sa.Column("ry", sa.Float(), nullable=False), + sa.Column("rw", sa.Float(), nullable=False), + sa.Column("rh", sa.Float(), nullable=False), + sa.Column("score", sa.Float(), nullable=True), + sa.Column("detector_version", sa.String(length=64), nullable=True), + sa.Column("crop_version", sa.String(length=64), nullable=True), + sa.Column("embedding_version", sa.String(length=128), nullable=True), + sa.Column("ccip_embedding", Vector(_CCIP_DIM), nullable=True), + sa.Column("siglip_embedding", Vector(_SIGLIP_DIM), nullable=True), + sa.Column( + "created_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + ) + op.create_index( + "ix_image_region_image_record_id", "image_region", ["image_record_id"], + ) + + +def downgrade() -> None: + op.drop_index("ix_image_region_image_record_id", table_name="image_region") + op.drop_table("image_region") diff --git a/alembic/versions/0062_gpu_job.py b/alembic/versions/0062_gpu_job.py new file mode 100644 index 0000000..a044995 --- /dev/null +++ b/alembic/versions/0062_gpu_job.py @@ -0,0 +1,55 @@ +"""gpu_job: the HTTP-leased GPU work queue for the desktop agent (#114) + +The agent stays HTTP-only — the server enqueues per-(image, task) jobs here and +the agent leases/submits over the web API; Redis/Postgres stay private. + +Revision ID: 0062 +Revises: 0061 +Create Date: 2026-06-29 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0062" +down_revision: Union[str, None] = "0061" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "gpu_job", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "image_record_id", sa.Integer(), + sa.ForeignKey("image_record.id", ondelete="CASCADE"), nullable=False, + ), + sa.Column("task", sa.String(length=32), nullable=False), + sa.Column( + "status", sa.String(length=16), nullable=False, + server_default="pending", + ), + sa.Column("lease_token", sa.String(length=64), nullable=True), + sa.Column("leased_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("lease_expires_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("attempts", sa.Integer(), nullable=False, server_default="0"), + sa.Column("error", sa.Text(), nullable=True), + sa.Column( + "created_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column( + "updated_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + ) + op.create_index("ix_gpu_job_image_record_id", "gpu_job", ["image_record_id"]) + op.create_index("ix_gpu_job_status", "gpu_job", ["status"]) + + +def downgrade() -> None: + op.drop_index("ix_gpu_job_status", table_name="gpu_job") + op.drop_index("ix_gpu_job_image_record_id", table_name="gpu_job") + op.drop_table("gpu_job") diff --git a/alembic/versions/0063_ccip_match_threshold.py b/alembic/versions/0063_ccip_match_threshold.py new file mode 100644 index 0000000..d841398 --- /dev/null +++ b/alembic/versions/0063_ccip_match_threshold.py @@ -0,0 +1,33 @@ +"""ml_settings.ccip_match_threshold — tunable CCIP character-match cut (#114) + +The v1 matcher used a flat 0.75 cosine; live data showed that over-fires (a +high-reference character matched a scatter of images). 0.85 keeps the confident +single-character matches and drops the noise. Tunable from the GPU agent card. + +Revision ID: 0063 +Revises: 0062 +Create Date: 2026-06-29 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0063" +down_revision: Union[str, None] = "0062" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "ccip_match_threshold", sa.Float(), nullable=False, + server_default="0.85", + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "ccip_match_threshold") diff --git a/alembic/versions/0064_ccip_auto_apply.py b/alembic/versions/0064_ccip_auto_apply.py new file mode 100644 index 0000000..e5323cf --- /dev/null +++ b/alembic/versions/0064_ccip_auto_apply.py @@ -0,0 +1,42 @@ +"""ml_settings: CCIP auto-apply switch + threshold (#114) + +Confident CCIP character matches auto-tag (source='ccip_auto') on a daily sweep, +so identity tags keep flowing without pressing a button. ON by default (opt-out, +like head auto-apply); the high threshold (0.92, above the 0.85 suggest cut) + +single-character references keep it safe, and every auto-tag is reversible. + +Revision ID: 0064 +Revises: 0063 +Create Date: 2026-06-30 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0064" +down_revision: Union[str, None] = "0063" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "ccip_auto_apply_enabled", sa.Boolean(), nullable=False, + server_default=sa.true(), + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "ccip_auto_apply_threshold", sa.Float(), nullable=False, + server_default="0.92", + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "ccip_auto_apply_threshold") + op.drop_column("ml_settings", "ccip_auto_apply_enabled") diff --git a/alembic/versions/0065_embedder_model_name.py b/alembic/versions/0065_embedder_model_name.py new file mode 100644 index 0000000..0a986b3 --- /dev/null +++ b/alembic/versions/0065_embedder_model_name.py @@ -0,0 +1,35 @@ +"""ml_settings: embedder_model_name (#1190 operator model swap) + +The embedder MODEL VERSION was already a setting (and stamps image_record. +siglip_model_version); the HF model NAME was env-only, so an operator couldn't +actually point the pipeline at a different embedder. Storing the name as a +setting makes the model an operator choice: set name + version → re-embed (the +GPU agent) → retrain heads. Default = the current SigLIP so400m. + +Revision ID: 0065 +Revises: 0064 +Create Date: 2026-06-30 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0065" +down_revision: Union[str, None] = "0064" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "embedder_model_name", sa.String(length=128), nullable=False, + server_default="google/siglip-so400m-patch14-384", + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "embedder_model_name") diff --git a/alembic/versions/0066_drop_centroids.py b/alembic/versions/0066_drop_centroids.py new file mode 100644 index 0000000..d75a334 --- /dev/null +++ b/alembic/versions/0066_drop_centroids.py @@ -0,0 +1,57 @@ +"""drop the dead per-tag centroid subsystem (#1189 cleanup) + +The v2 pivot replaced per-tag SigLIP centroids with learned heads + CCIP. +Nothing read the centroids anymore — they were recomputed (on merge + a daily +beat) but never consumed for suggestions or auto-apply. Remove the storage + +its two now-unused settings columns. (The recompute tasks, beat, endpoint, +service, and UI card are removed in the same change.) + +Revision ID: 0066 +Revises: 0065 +Create Date: 2026-06-30 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0066" +down_revision: Union[str, None] = "0065" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.drop_table("tag_reference_embedding") + op.drop_column("ml_settings", "centroid_similarity_threshold") + op.drop_column("ml_settings", "min_reference_images") + + +def downgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "min_reference_images", sa.Integer(), nullable=False, + server_default="5", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "centroid_similarity_threshold", sa.Float(), nullable=False, + server_default="0.55", + ), + ) + op.create_table( + "tag_reference_embedding", + sa.Column("tag_id", sa.Integer(), nullable=False), + sa.Column("embedding", sa.LargeBinary(), nullable=False), + sa.Column("reference_count", sa.Integer(), nullable=False), + sa.Column("model_version", sa.String(length=128), nullable=False), + sa.Column( + "updated_at", sa.DateTime(timezone=True), + server_default=sa.func.now(), nullable=False, + ), + sa.ForeignKeyConstraint(["tag_id"], ["tag.id"], ondelete="CASCADE"), + sa.PrimaryKeyConstraint("tag_id"), + ) diff --git a/alembic/versions/0067_retire_camie_allowlist.py b/alembic/versions/0067_retire_camie_allowlist.py new file mode 100644 index 0000000..e3edd02 --- /dev/null +++ b/alembic/versions/0067_retire_camie_allowlist.py @@ -0,0 +1,66 @@ +"""retire the Camie tagger + allowlist bulk-apply (#1189) + +The v2 pivot made heads + CCIP the tag source and head auto-apply the earned +propagation. The Camie tagger ran only to feed the allowlist bulk-apply (its +predictions had no other consumer), and the allowlist was a second, un-earned +auto-apply path parallel to heads. Both are retired — drop their storage. + +(image_prediction = Camie's per-image predictions; tag_allowlist = the bulk- +apply allowlist. Nothing references INTO these tables, so the drop is clean.) + +Revision ID: 0067 +Revises: 0066 +Create Date: 2026-06-30 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0067" +down_revision: Union[str, None] = "0066" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.drop_table("image_prediction") + op.drop_table("tag_allowlist") + + +def downgrade() -> None: + op.create_table( + "tag_allowlist", + sa.Column("tag_id", sa.Integer(), nullable=False), + sa.Column( + "min_confidence", sa.Float(), nullable=False, server_default="0.9" + ), + sa.Column( + "created_at", sa.DateTime(timezone=True), + server_default=sa.func.now(), nullable=False, + ), + sa.ForeignKeyConstraint(["tag_id"], ["tag.id"], ondelete="CASCADE"), + sa.PrimaryKeyConstraint("tag_id"), + sa.CheckConstraint( + "min_confidence >= 0 AND min_confidence <= 1", + name="ck_tag_allowlist_confidence_range", + ), + ) + op.create_table( + "image_prediction", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("image_record_id", sa.Integer(), nullable=False), + sa.Column("raw_name", sa.String(length=255), nullable=False), + sa.Column("category", sa.String(length=32), nullable=False), + sa.Column("score", sa.Float(), nullable=False), + sa.ForeignKeyConstraint( + ["image_record_id"], ["image_record.id"], ondelete="CASCADE" + ), + ) + op.create_index( + "ix_image_prediction_image", "image_prediction", ["image_record_id"] + ) + op.create_index( + "ix_image_prediction_name_score", "image_prediction", + ["raw_name", "score"], + ) diff --git a/alembic/versions/0068_drop_dead_tagger_settings.py b/alembic/versions/0068_drop_dead_tagger_settings.py new file mode 100644 index 0000000..770676d --- /dev/null +++ b/alembic/versions/0068_drop_dead_tagger_settings.py @@ -0,0 +1,80 @@ +"""drop dead tagger/suggestion settings + columns left after Camie retirement (#1199) + +Hygiene follow-up to #1189. These were left inert to bound that change; nothing +reads them now: +- ml_settings: tagger_store_floor + tagger_model_version (only the deleted Camie + tagger used them), suggestion_threshold_character/general (already dead pre- + retirement — scoring uses per-head thresholds), video_min_tag_frames (only the + deleted video-prediction aggregator used it). +- image_record: tagger_model_version (no writer now), centroid_scores (long-dead + JSON cache, no reader). + +Revision ID: 0068 +Revises: 0067 +Create Date: 2026-06-30 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0068" +down_revision: Union[str, None] = "0067" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.drop_column("ml_settings", "suggestion_threshold_character") + op.drop_column("ml_settings", "suggestion_threshold_general") + op.drop_column("ml_settings", "tagger_store_floor") + op.drop_column("ml_settings", "video_min_tag_frames") + op.drop_column("ml_settings", "tagger_model_version") + op.drop_column("image_record", "tagger_model_version") + op.drop_column("image_record", "centroid_scores") + + +def downgrade() -> None: + op.add_column( + "image_record", + sa.Column("centroid_scores", sa.JSON(), nullable=True), + ) + op.add_column( + "image_record", + sa.Column("tagger_model_version", sa.String(length=128), nullable=True), + ) + op.add_column( + "ml_settings", + sa.Column( + "tagger_model_version", sa.String(length=128), nullable=False, + server_default="camie-tagger-v2", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "video_min_tag_frames", sa.Integer(), nullable=False, + server_default="3", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "tagger_store_floor", sa.Float(), nullable=False, + server_default="0.7", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "suggestion_threshold_general", sa.Float(), nullable=False, + server_default="0.7", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "suggestion_threshold_character", sa.Float(), nullable=False, + server_default="0.7", + ), + ) diff --git a/alembic/versions/0069_default_siglip2.py b/alembic/versions/0069_default_siglip2.py new file mode 100644 index 0000000..7bef8b1 --- /dev/null +++ b/alembic/versions/0069_default_siglip2.py @@ -0,0 +1,51 @@ +"""default the embedder to SigLIP 2 — for FRESH installs only (#1203) + +Make SigLIP 2 (so400m, 512px; a 1152-d drop-in) the default embedder. New +installs start on it. An EXISTING library is NOT touched: flipping its stored +embedder version would mark every embedding stale (the scorer is version-gated) +and kill suggestions until a full re-embed+retrain — so an existing instance +switches deliberately via Settings → GPU agent → Embedding model → Re-embed → +Retrain. We detect "fresh" by the absence of any embedded image. + +Revision ID: 0069 +Revises: 0068 +Create Date: 2026-06-30 +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0069" +down_revision: Union[str, None] = "0068" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +_NEW_NAME = "google/siglip2-so400m-patch16-512" +_NEW_VERSION = "siglip2-so400m-patch16-512" +_OLD_NAME = "google/siglip-so400m-patch14-384" +_OLD_VERSION = "siglip-so400m-patch14-384" + + +def upgrade() -> None: + # Fresh install (nothing embedded yet) → adopt SigLIP 2. + op.execute( + f""" + UPDATE ml_settings SET + embedder_model_name = '{_NEW_NAME}', + embedder_model_version = '{_NEW_VERSION}' + WHERE NOT EXISTS ( + SELECT 1 FROM image_record WHERE siglip_embedding IS NOT NULL + ) + """ + ) + op.alter_column("ml_settings", "embedder_model_name", server_default=_NEW_NAME) + op.alter_column( + "ml_settings", "embedder_model_version", server_default=_NEW_VERSION + ) + + +def downgrade() -> None: + op.alter_column("ml_settings", "embedder_model_name", server_default=_OLD_NAME) + op.alter_column( + "ml_settings", "embedder_model_version", server_default=_OLD_VERSION + ) diff --git a/alembic/versions/0070_gpu_job_lease_indexes.py b/alembic/versions/0070_gpu_job_lease_indexes.py new file mode 100644 index 0000000..10ec3f9 --- /dev/null +++ b/alembic/versions/0070_gpu_job_lease_indexes.py @@ -0,0 +1,44 @@ +"""partial indexes so GPU-job leasing stays O(batch), not O(completed) + +The lease claims the lowest-id pending (or expired-leased) jobs. With only a +plain `status` index, `... ORDER BY id LIMIT n` walked the primary-key index from +the start, skipping the entire prefix of already-done/error rows before reaching +pending ones — so leasing slowed to a crawl as `done` piled up (the whole reason +throughput fell off a cliff mid-run and /status stalled). Two partial indexes fix +it: the pending one is id-ordered so the hot path reads just the first n entries, +and the leased-expiry one keeps the crash-recovery reclaim + the orphan sweep +cheap. They cover only the small live slice of the table, so they stay tiny even +as the done/error history grows to millions. + +Revision ID: 0070 +Revises: 0069 +Create Date: 2026-06-30 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0070" +down_revision: Union[str, None] = "0069" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # Hot path: lowest-id pending jobs. Index on id, restricted to pending, so + # `WHERE status='pending' ORDER BY id LIMIT n` is a short index-order scan. + op.create_index( + "ix_gpu_job_pending", "gpu_job", ["id"], + postgresql_where=sa.text("status = 'pending'"), + ) + # Crash-recovery: expired leases, for the lease backstop + recover_orphaned. + op.create_index( + "ix_gpu_job_leased_expires", "gpu_job", ["lease_expires_at"], + postgresql_where=sa.text("status = 'leased'"), + ) + + +def downgrade() -> None: + op.drop_index("ix_gpu_job_leased_expires", table_name="gpu_job") + op.drop_index("ix_gpu_job_pending", table_name="gpu_job") diff --git a/alembic/versions/0071_image_record_earliest_post_date.py b/alembic/versions/0071_image_record_earliest_post_date.py new file mode 100644 index 0000000..b2e8f0c --- /dev/null +++ b/alembic/versions/0071_image_record_earliest_post_date.py @@ -0,0 +1,80 @@ +"""image_record.earliest_post_date: original-publish gallery sort key + index + +Revision ID: 0071 +Revises: 0070 +Create Date: 2026-07-01 + +effective_date (0035) keys off the PRIMARY post — which is often the repost / +download the file actually came from — and falls back to created_at, so the +gallery's default order surfaces download dates rather than when content was +first posted (operator-flagged 2026-07-01). Materialize a second sort key, +earliest_post_date = MIN(post_date) across ALL of an image's provenance posts +(every post it appears in), falling back to created_at only when no linked post +carries a date. Indexed (DESC, id DESC) so the "post date" gallery sort is an +index range scan just like effective_date. + +Backfill mirrors 0035: created_at baseline, then override with the MIN over +image_provenance ⋈ post. New rows get the created_at-equivalent server default; +services/importer.py recomputes it whenever a dated post is linked. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0071" +down_revision: Union[str, None] = "0070" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # Add nullable first so the backfill can populate before NOT NULL. + op.add_column( + "image_record", + sa.Column("earliest_post_date", sa.DateTime(timezone=True), nullable=True), + ) + # Baseline: download date. Set-based (no per-row binds) → immune to the + # 65535 bind-parameter ceiling regardless of library size. + op.execute( + """ + UPDATE image_record + SET earliest_post_date = created_at + """ + ) + # Override with the earliest post_date across EVERY post the image appears + # in (image_provenance is the many-to-many edge; ignore posts with no date). + op.execute( + """ + UPDATE image_record AS ir + SET earliest_post_date = sub.min_date + FROM ( + SELECT ip.image_record_id AS iid, MIN(p.post_date) AS min_date + FROM image_provenance AS ip + JOIN post AS p ON p.id = ip.post_id + WHERE p.post_date IS NOT NULL + GROUP BY ip.image_record_id + ) AS sub + WHERE ir.id = sub.iid + """ + ) + op.alter_column( + "image_record", + "earliest_post_date", + nullable=False, + server_default=sa.text("now()"), + ) + # DESC/DESC matches the gallery's ORDER BY earliest_post_date DESC, id DESC + # so the "post date" scroll is a forward index scan; raw SQL because + # alembic's column list doesn't express per-column DESC cleanly. + op.execute( + "CREATE INDEX ix_image_record_earliest_post_date " + "ON image_record (earliest_post_date DESC, id DESC)" + ) + + +def downgrade() -> None: + op.drop_index( + "ix_image_record_earliest_post_date", table_name="image_record" + ) + op.drop_column("image_record", "earliest_post_date") diff --git a/alembic/versions/0072_gpu_job_triage_status.py b/alembic/versions/0072_gpu_job_triage_status.py new file mode 100644 index 0000000..1dce875 --- /dev/null +++ b/alembic/versions/0072_gpu_job_triage_status.py @@ -0,0 +1,32 @@ +"""gpu_job.triage_status — the probe's verdict on an errored job's FILE + +Failure triage (#125): a periodic sweep probes each errored image's file +(sha256 + decode, verify_integrity's machinery) exactly once and stores the +verdict here — 'defect' (the file is bad: recovery material, excluded from +/retry_errors) or 'file_ok' (failure was operational, safe to retry). NULL +means not yet probed; selecting on NULL is what makes the sweep resumable. +No index: the errored slice the sweep scans is tiny by design (tombstones). + +Revision ID: 0072 +Revises: 0071 +Create Date: 2026-07-02 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0072" +down_revision: Union[str, None] = "0071" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "gpu_job", sa.Column("triage_status", sa.String(16), nullable=True) + ) + + +def downgrade() -> None: + op.drop_column("gpu_job", "triage_status") diff --git a/alembic/versions/0073_drop_tag_eval_run.py b/alembic/versions/0073_drop_tag_eval_run.py new file mode 100644 index 0000000..4aedb38 --- /dev/null +++ b/alembic/versions/0073_drop_tag_eval_run.py @@ -0,0 +1,46 @@ +"""drop tag_eval_run — the head-vs-centroid eval harness is retired + +The eval (#1130) existed to prove the heads tagging spine on the operator's own +data. It did; the operator accepted the system and retired the harness +(2026-07-02) — card, API, task, model and this table all go. The eval's data +loaders + metric helpers live on in services/ml/training_data.py, where the +production heads trainer uses them nightly. + +Revision ID: 0073 +Revises: 0072 +Create Date: 2026-07-02 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from sqlalchemy.dialects import postgresql + +revision: str = "0073" +down_revision: Union[str, None] = "0072" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.drop_index("ix_tag_eval_run_status", table_name="tag_eval_run") + op.drop_table("tag_eval_run") + + +def downgrade() -> None: + # Recreates the shape from 0056 (data is not restorable). + op.create_table( + "tag_eval_run", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("params", postgresql.JSONB(), nullable=False), + sa.Column("status", sa.String(length=16), nullable=False, + server_default="running"), + sa.Column("started_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now()), + sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("report", postgresql.JSONB(), nullable=True), + sa.Column("error", sa.Text(), nullable=True), + sa.Column("last_progress_at", sa.DateTime(timezone=True), + nullable=True), + ) + op.create_index("ix_tag_eval_run_status", "tag_eval_run", ["status"]) diff --git a/alembic/versions/0074_ml_settings_cpu_embed_enabled.py b/alembic/versions/0074_ml_settings_cpu_embed_enabled.py new file mode 100644 index 0000000..48ff8ea --- /dev/null +++ b/alembic/versions/0074_ml_settings_cpu_embed_enabled.py @@ -0,0 +1,35 @@ +"""ml_settings.cpu_embed_enabled — the CPU embed fallback becomes a switch + +B3 (operator 2026-07-02): the ml-worker's only processing role is the CPU +whole-image embed for stacks without a GPU agent. ON by default (a fresh +install works agent-less); agent-equipped stacks that drop the ml-worker +container turn it off so import hooks stop queueing embed work into a queue +nothing consumes — the daily GPU 'embed' backfill covers those images. + +Revision ID: 0074 +Revises: 0073 +Create Date: 2026-07-02 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0074" +down_revision: Union[str, None] = "0073" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "cpu_embed_enabled", sa.Boolean(), nullable=False, + server_default=sa.true(), + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "cpu_embed_enabled") diff --git a/alembic/versions/0075_tag_is_system.py b/alembic/versions/0075_tag_is_system.py new file mode 100644 index 0000000..a6b7e7a --- /dev/null +++ b/alembic/versions/0075_tag_is_system.py @@ -0,0 +1,60 @@ +"""tag.is_system + seed the three hygiene system tags + +Training hygiene (operator 2026-07-03, milestone #128): rough WIPs tagged as a +character poison that character's head and CCIP references; banners/editor +screenshots pollute whole-image similarity. The fix keys on SYSTEM tags the +product ships — not operator configuration — so the seed lives here. + +Seeding ADOPTS an existing same-(name, kind=general) tag (case-insensitive, +matching TagService.rename's collision stance) instead of inserting a +duplicate, so an operator who already tagged `wip` keeps their applications. + +Revision ID: 0075 +Revises: 0074 +Create Date: 2026-07-03 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0075" +down_revision: Union[str, None] = "0074" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +SYSTEM_TAG_NAMES = ("wip", "banner", "editor screenshot") + + +def upgrade() -> None: + op.add_column( + "tag", + sa.Column( + "is_system", sa.Boolean(), nullable=False, + server_default=sa.false(), + ), + ) + conn = op.get_bind() + for name in SYSTEM_TAG_NAMES: + adopted = conn.execute( + sa.text( + "UPDATE tag SET is_system = true " + "WHERE lower(name) = lower(:name) AND kind = 'general'" + ), + {"name": name}, + ) + if adopted.rowcount == 0: + conn.execute( + sa.text( + "INSERT INTO tag (name, kind, is_system) " + "VALUES (:name, 'general', true)" + ), + {"name": name}, + ) + + +def downgrade() -> None: + # The seeded rows survive as ordinary general tags — dropping the flag is + # enough to disarm the mechanism, and deleting rows would orphan any + # operator applications made while the flag existed. + op.drop_column("tag", "is_system") diff --git a/alembic/versions/0076_pixiv_ledgers.py b/alembic/versions/0076_pixiv_ledgers.py new file mode 100644 index 0000000..2655130 --- /dev/null +++ b/alembic/versions/0076_pixiv_ledgers.py @@ -0,0 +1,82 @@ +"""pixiv_seen_media + pixiv_failed_media: per-source ledgers + +Revision ID: 0076 +Revises: 0075 +Create Date: 2026-07-03 + +Pixiv native ingester (milestone #129, gallery-dl → native-core migration). +Mirrors the Patreon (0037/0038) and SubscribeStar (0054) ledger tables: a +seen-ledger so routine walks skip already-ingested media (recovery bypasses +it) and a dead-letter ledger so persistently-failing media stops re-burning +backfill chunks. Pixiv URLs carry no content hash, so `filehash` is always the +synthesized ``:p`` / ``:ugoira`` key — String(128) +matches the siblings. UNIQUE (source_id, filehash) is the upsert key on each. +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0076" +down_revision: Union[str, None] = "0075" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "pixiv_seen_media", + sa.Column("id", sa.Integer, primary_key=True), + sa.Column( + "source_id", + sa.Integer, + sa.ForeignKey("source.id", ondelete="CASCADE"), + nullable=False, + index=True, + ), + sa.Column("filehash", sa.String(128), nullable=False), + sa.Column("post_id", sa.String(64), nullable=True), + sa.Column( + "seen_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.UniqueConstraint( + "source_id", "filehash", name="uq_pixiv_seen_media_source_id" + ), + ) + op.create_table( + "pixiv_failed_media", + sa.Column("id", sa.Integer, primary_key=True), + sa.Column( + "source_id", + sa.Integer, + sa.ForeignKey("source.id", ondelete="CASCADE"), + nullable=False, + index=True, + ), + sa.Column("filehash", sa.String(128), nullable=False), + sa.Column("attempts", sa.Integer, nullable=False, server_default="1"), + sa.Column("last_error", sa.Text, nullable=True), + sa.Column( + "first_failed_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.Column( + "last_failed_at", + sa.DateTime(timezone=True), + nullable=False, + server_default=sa.text("NOW()"), + ), + sa.UniqueConstraint( + "source_id", "filehash", name="uq_pixiv_failed_media_source_id" + ), + ) + + +def downgrade() -> None: + op.drop_table("pixiv_failed_media") + op.drop_table("pixiv_seen_media") diff --git a/alembic/versions/0077_artist_name_not_unique.py b/alembic/versions/0077_artist_name_not_unique.py new file mode 100644 index 0000000..6a09288 --- /dev/null +++ b/alembic/versions/0077_artist_name_not_unique.py @@ -0,0 +1,32 @@ +"""drop uq_artist_name — decouple display name from identity/storage + +Revision ID: 0077 +Revises: 0076 +Create Date: 2026-07-04 + +Artist model fragility fix (milestone #130). One `slug` column was doing +identity + storage-path + display, and BOTH `name` and `slug` were UNIQUE, so +the display name couldn't be edited freely and two genuinely different creators +collided. Decouple: `slug` stays the immutable, unique storage/identity key (the +on-disk path component — untouched here); `name` becomes freely editable, NON- +unique display text. This migration only drops the `uq_artist_name` constraint; +no data moves and no path changes. +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0077" +down_revision: Union[str, None] = "0076" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.drop_constraint("uq_artist_name", "artist", type_="unique") + + +def downgrade() -> None: + # Re-adding the UNIQUE would fail if duplicate names now exist; callers that + # need to reverse this must dedupe names first. + op.create_unique_constraint("uq_artist_name", "artist", ["name"]) diff --git a/alembic/versions/0078_ml_settings_detectors.py b/alembic/versions/0078_ml_settings_detectors.py new file mode 100644 index 0000000..6d04601 --- /dev/null +++ b/alembic/versions/0078_ml_settings_detectors.py @@ -0,0 +1,83 @@ +"""ml_settings crop-proposer / detector config (#134) + +Move the WHERE-to-crop detector config (per-proposer enable + weights + conf, +plus caps + dedupe IoU) into the DB so it's UI-tunable and announced to the GPU +agent in the lease (like the embedder model) — no restart, agent env is now +bootstrap-only. All server_defaults are the working values so existing rows + +fresh installs crop out-of-the-box with all three proposers ON. + +Revision ID: 0078 +Revises: 0077 +Create Date: 2026-07-05 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0078" +down_revision: Union[str, None] = "0077" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +_ANATOMY_DEFAULT = ( + "https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt" +) +_PANEL_DEFAULT = "mosesb/best-comic-panel-detection::best.pt" + + +def upgrade() -> None: + op.add_column("ml_settings", sa.Column( + "detector_person_enabled", sa.Boolean(), nullable=False, + server_default=sa.true())) + op.add_column("ml_settings", sa.Column( + "detector_person_weights", sa.String(512), nullable=False, + server_default="yolo11n.pt")) + op.add_column("ml_settings", sa.Column( + "detector_person_conf", sa.Float(), nullable=False, + server_default=sa.text("0.35"))) + op.add_column("ml_settings", sa.Column( + "detector_anatomy_enabled", sa.Boolean(), nullable=False, + server_default=sa.true())) + op.add_column("ml_settings", sa.Column( + "detector_anatomy_weights", sa.String(512), nullable=False, + server_default=_ANATOMY_DEFAULT)) + op.add_column("ml_settings", sa.Column( + "detector_anatomy_conf", sa.Float(), nullable=False, + server_default=sa.text("0.30"))) + op.add_column("ml_settings", sa.Column( + "detector_panel_enabled", sa.Boolean(), nullable=False, + server_default=sa.true())) + op.add_column("ml_settings", sa.Column( + "detector_panel_weights", sa.String(512), nullable=False, + server_default=_PANEL_DEFAULT)) + op.add_column("ml_settings", sa.Column( + "detector_panel_conf", sa.Float(), nullable=False, + server_default=sa.text("0.30"))) + op.add_column("ml_settings", sa.Column( + "detector_max_figures", sa.Integer(), nullable=False, + server_default=sa.text("8"))) + op.add_column("ml_settings", sa.Column( + "detector_max_components", sa.Integer(), nullable=False, + server_default=sa.text("8"))) + op.add_column("ml_settings", sa.Column( + "detector_max_panels", sa.Integer(), nullable=False, + server_default=sa.text("8"))) + op.add_column("ml_settings", sa.Column( + "detector_max_regions", sa.Integer(), nullable=False, + server_default=sa.text("128"))) + op.add_column("ml_settings", sa.Column( + "detector_dedupe_iou", sa.Float(), nullable=False, + server_default=sa.text("0.85"))) + + +def downgrade() -> None: + for col in ( + "detector_person_enabled", "detector_person_weights", "detector_person_conf", + "detector_anatomy_enabled", "detector_anatomy_weights", "detector_anatomy_conf", + "detector_panel_enabled", "detector_panel_weights", "detector_panel_conf", + "detector_max_figures", "detector_max_components", "detector_max_panels", + "detector_max_regions", "detector_dedupe_iou", + ): + op.drop_column("ml_settings", col) diff --git a/alembic/versions/0079_character_prototypes.py b/alembic/versions/0079_character_prototypes.py new file mode 100644 index 0000000..8ada2f4 --- /dev/null +++ b/alembic/versions/0079_character_prototypes.py @@ -0,0 +1,77 @@ +"""character prototype store (#1317) — precomputed, incremental CCIP references + +New tables character_prototype + ccip_prototype_state, plus MLSettings columns +ccip_ref_signature (cheap global change gate) + ccip_prototype_cap (per-character +reference cap). The reference set the CCIP matcher uses becomes a precomputed +artifact refreshed incrementally off the request path. See milestone 138 / +backend.app.services.ml.character_prototypes. + +Revision ID: 0079 +Revises: 0078 +Create Date: 2026-07-06 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op +from pgvector.sqlalchemy import Vector + +revision: str = "0079" +down_revision: Union[str, None] = "0078" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +# Matches models.image_region.CCIP_DIM (the CCIP figure-embedding width). +_CCIP_DIM = 768 + + +def upgrade() -> None: + op.create_table( + "character_prototype", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "tag_id", sa.Integer(), + sa.ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, + ), + sa.Column("ccip_embedding", Vector(_CCIP_DIM), nullable=False), + sa.Column( + "region_id", sa.Integer(), + sa.ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True, + ), + ) + op.create_index( + "ix_character_prototype_tag_id", "character_prototype", ["tag_id"] + ) + op.create_table( + "ccip_prototype_state", + sa.Column( + "tag_id", sa.Integer(), + sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, + ), + sa.Column("fingerprint", sa.String(64), nullable=False), + sa.Column( + "updated_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + ) + op.add_column( + "ml_settings", + sa.Column("ccip_ref_signature", sa.String(128), nullable=True), + ) + op.add_column( + "ml_settings", + sa.Column( + "ccip_prototype_cap", sa.Integer(), nullable=False, + server_default=sa.text("64"), + ), + ) + + +def downgrade() -> None: + op.drop_column("ml_settings", "ccip_prototype_cap") + op.drop_column("ml_settings", "ccip_ref_signature") + op.drop_table("ccip_prototype_state") + op.drop_index( + "ix_character_prototype_tag_id", table_name="character_prototype" + ) + op.drop_table("character_prototype") diff --git a/alembic/versions/0080_tag_head_train_fingerprint.py b/alembic/versions/0080_tag_head_train_fingerprint.py new file mode 100644 index 0000000..b4bd224 --- /dev/null +++ b/alembic/versions/0080_tag_head_train_fingerprint.py @@ -0,0 +1,31 @@ +"""tag_head.train_fingerprint (#1317 phase 2) — incremental head retraining + +A per-head training-data fingerprint (positive + rejection count/latest-timestamp) +so a manual Retrain refits only the tags whose data changed; the nightly run +ignores it (full reconcile). Nullable — a NULL fingerprint (existing heads) forces +a refit on the first incremental run, then it's stamped. + +Revision ID: 0080 +Revises: 0079 +Create Date: 2026-07-06 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0080" +down_revision: Union[str, None] = "0079" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "tag_head", + sa.Column("train_fingerprint", sa.String(128), nullable=True), + ) + + +def downgrade() -> None: + op.drop_column("tag_head", "train_fingerprint") diff --git a/alembic/versions/0081_stricter_auto_apply_defaults.py b/alembic/versions/0081_stricter_auto_apply_defaults.py new file mode 100644 index 0000000..8030eec --- /dev/null +++ b/alembic/versions/0081_stricter_auto_apply_defaults.py @@ -0,0 +1,43 @@ +"""stricter auto-apply defaults (milestone 139) — cut auto-apply misfires + +head_auto_apply_min_positives 30→50 and ccip_auto_apply_threshold 0.92→0.95 +(operator-asked 2026-07-06). The head graduation precision bar stays 0.97 — the +operator confirmed the general-tag confidence was already well tuned; only the +support floor + the CCIP match confidence are raised. The model defaults change +for fresh installs; here we bump the existing singleton row IFF it is still at +the previous default, so a deliberate operator change is NOT clobbered. + +Revision ID: 0081 +Revises: 0080 +Create Date: 2026-07-06 +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0081" +down_revision: Union[str, None] = "0080" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.execute( + "UPDATE ml_settings SET head_auto_apply_min_positives = 50 " + "WHERE head_auto_apply_min_positives = 30" + ) + op.execute( + "UPDATE ml_settings SET ccip_auto_apply_threshold = 0.95 " + "WHERE ccip_auto_apply_threshold = 0.92" + ) + + +def downgrade() -> None: + op.execute( + "UPDATE ml_settings SET head_auto_apply_min_positives = 30 " + "WHERE head_auto_apply_min_positives = 50" + ) + op.execute( + "UPDATE ml_settings SET ccip_auto_apply_threshold = 0.92 " + "WHERE ccip_auto_apply_threshold = 0.95" + ) diff --git a/alembic/versions/0082_presentation_auto_hide.py b/alembic/versions/0082_presentation_auto_hide.py new file mode 100644 index 0000000..8dc1f3c --- /dev/null +++ b/alembic/versions/0082_presentation_auto_hide.py @@ -0,0 +1,85 @@ +"""presentation-chrome auto-hide (#141) — settings knobs + review table + +MLSettings gains presentation_auto_apply_enabled / _threshold and +presentation_conflict_threshold: banner + editor-screenshot auto-hide on the +sweep with a FLAT threshold (decoupled from content-head graduation), and a +conflict threshold that flags an auto-hide that "also looks like content". + +New table presentation_review records an auto-hidden chrome image that also +scored high on a content head, surfaced in the Hidden view for a keep-hidden / +un-hide decision. Resolved rows are pruned by retention. + +Revision ID: 0082 +Revises: 0081 +Create Date: 2026-07-07 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0082" +down_revision: Union[str, None] = "0081" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "presentation_auto_apply_enabled", sa.Boolean(), nullable=False, + server_default=sa.text("true"), + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "presentation_auto_apply_threshold", sa.Float(), nullable=False, + server_default=sa.text("0.90"), + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "presentation_conflict_threshold", sa.Float(), nullable=False, + server_default=sa.text("0.50"), + ), + ) + op.create_table( + "presentation_review", + sa.Column( + "image_record_id", sa.Integer(), + sa.ForeignKey("image_record.id", ondelete="CASCADE"), + primary_key=True, + ), + sa.Column( + "tag_id", sa.Integer(), + sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, + ), + sa.Column( + "conflict_tag_id", sa.Integer(), + sa.ForeignKey("tag.id", ondelete="SET NULL"), nullable=True, + ), + sa.Column("conflict_score", sa.Float(), nullable=False), + sa.Column( + "created_at", sa.DateTime(timezone=True), nullable=False, + server_default=sa.func.now(), + ), + sa.Column("resolved_at", sa.DateTime(timezone=True), nullable=True), + ) + # The review list queries the unresolved flags (resolved_at IS NULL). + op.create_index( + "ix_presentation_review_resolved_at", "presentation_review", + ["resolved_at"], + ) + + +def downgrade() -> None: + op.drop_index( + "ix_presentation_review_resolved_at", table_name="presentation_review" + ) + op.drop_table("presentation_review") + op.drop_column("ml_settings", "presentation_conflict_threshold") + op.drop_column("ml_settings", "presentation_auto_apply_threshold") + op.drop_column("ml_settings", "presentation_auto_apply_enabled") diff --git a/alembic/versions/0083_post_translation.py b/alembic/versions/0083_post_translation.py new file mode 100644 index 0000000..d491d03 --- /dev/null +++ b/alembic/versions/0083_post_translation.py @@ -0,0 +1,73 @@ +"""post-text translation via Interpreter (milestone 143) — Post columns + settings + +Post gains the translated title/description + the detected source language, +Interpreter engine_version (cache key), and translated_at — filled by the +translate sweep. ImportSettings gains translation_enabled (OFF by default), +interpreter_base_url (EMPTY — the operator sets their own, behind a reverse +proxy), and translation_target_lang (en). + +Revision ID: 0083 +Revises: 0082 +Create Date: 2026-07-07 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0083" +down_revision: Union[str, None] = "0082" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "post", sa.Column("post_title_translated", sa.Text(), nullable=True) + ) + op.add_column( + "post", sa.Column("description_translated", sa.Text(), nullable=True) + ) + op.add_column( + "post", + sa.Column("translated_source_lang", sa.String(8), nullable=True), + ) + op.add_column( + "post", + sa.Column("translation_engine_version", sa.String(128), nullable=True), + ) + op.add_column( + "post", + sa.Column("translated_at", sa.DateTime(timezone=True), nullable=True), + ) + op.add_column( + "import_settings", + sa.Column( + "translation_enabled", sa.Boolean(), nullable=False, + server_default=sa.text("false"), + ), + ) + op.add_column( + "import_settings", + sa.Column( + "interpreter_base_url", sa.Text(), nullable=False, server_default="", + ), + ) + op.add_column( + "import_settings", + sa.Column( + "translation_target_lang", sa.Text(), nullable=False, + server_default="en", + ), + ) + + +def downgrade() -> None: + op.drop_column("import_settings", "translation_target_lang") + op.drop_column("import_settings", "interpreter_base_url") + op.drop_column("import_settings", "translation_enabled") + op.drop_column("post", "translated_at") + op.drop_column("post", "translation_engine_version") + op.drop_column("post", "translated_source_lang") + op.drop_column("post", "description_translated") + op.drop_column("post", "post_title_translated") diff --git a/alembic/versions/0084_translation_strictness_override.py b/alembic/versions/0084_translation_strictness_override.py new file mode 100644 index 0000000..cd514b4 --- /dev/null +++ b/alembic/versions/0084_translation_strictness_override.py @@ -0,0 +1,51 @@ +"""translation strictness setting + per-post translation override (milestone 155) + +ImportSettings gains ``translation_min_confidence`` (the latin-script acceptance +floor, now operator-tunable in the UI; default 0.9 — stricter than the old +hardcoded 0.8, since Interpreter confidently mis-detects short ASCII English at +~0.86). Post gains ``translation_override`` — a sticky per-post choice of +auto / force / original so the operator can force a skipped translation on, or +knock a wrongly-translated one back to the original, and have it survive a +Re-translate-all. + +Revision ID: 0084 +Revises: 0083 +Create Date: 2026-07-10 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0084" +down_revision: Union[str, None] = "0083" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "import_settings", + sa.Column( + "translation_min_confidence", sa.Float(), nullable=False, + server_default=sa.text("0.9"), + ), + ) + op.add_column( + "post", + sa.Column( + "translation_override", sa.String(16), nullable=False, + server_default="auto", + ), + ) + op.create_check_constraint( + "ck_post_translation_override", + "post", + "translation_override IN ('auto', 'force', 'original')", + ) + + +def downgrade() -> None: + op.drop_constraint("ck_post_translation_override", "post", type_="check") + op.drop_column("post", "translation_override") + op.drop_column("import_settings", "translation_min_confidence") diff --git a/alembic/versions/0085_wip_title_tagging.py b/alembic/versions/0085_wip_title_tagging.py new file mode 100644 index 0000000..4d260b1 --- /dev/null +++ b/alembic/versions/0085_wip_title_tagging.py @@ -0,0 +1,35 @@ +"""title-based WIP auto-tagging (task #1458) — ImportSettings toggle + +ImportSettings gains wip_title_tagging_enabled (ON by default): when a freshly +imported post's title explicitly declares work-in-progress ("WIP" / "work in +progress"), the importer applies the `wip` system tag to its images. No new +table — the tag itself is the seeded `wip` system tag (migration 0075) and the +application reuses image_tag with source='wip_title'. + +Revision ID: 0085 +Revises: 0084 +Create Date: 2026-07-12 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0085" +down_revision: Union[str, None] = "0084" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "import_settings", + sa.Column( + "wip_title_tagging_enabled", sa.Boolean(), nullable=False, + server_default=sa.text("true"), + ), + ) + + +def downgrade() -> None: + op.drop_column("import_settings", "wip_title_tagging_enabled") diff --git a/alembic/versions/0086_process_auto_apply_settings.py b/alembic/versions/0086_process_auto_apply_settings.py new file mode 100644 index 0000000..16f03c7 --- /dev/null +++ b/alembic/versions/0086_process_auto_apply_settings.py @@ -0,0 +1,61 @@ +"""process auto-apply settings + review mode (#1464) — system-tag refactor + +The system-tag behavior refactor gives `wip` / `editor screenshot` (the PROCESS +group) their own provisional auto-apply, parallel to the presentation (chrome) +sweep. MLSettings gains three knobs: enabled (OFF by default — a new whole-library +auto-tagger is opt-in), the flat apply threshold, and the ring-loud conflict +threshold. presentation_review gains a `mode` column so one review surface serves +both chrome and process flags (existing rows backfill 'chrome'). server_defaults +so the existing rows fill cleanly. + +Revision ID: 0086 +Revises: 0085 +Create Date: 2026-07-13 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0086" +down_revision: Union[str, None] = "0085" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "ml_settings", + sa.Column( + "process_auto_apply_enabled", sa.Boolean(), nullable=False, + server_default=sa.text("false"), + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "process_auto_apply_threshold", sa.Float(), nullable=False, + server_default="0.90", + ), + ) + op.add_column( + "ml_settings", + sa.Column( + "process_conflict_threshold", sa.Float(), nullable=False, + server_default="0.50", + ), + ) + op.add_column( + "presentation_review", + sa.Column( + "mode", sa.String(16), nullable=False, + server_default="chrome", + ), + ) + + +def downgrade() -> None: + op.drop_column("presentation_review", "mode") + op.drop_column("ml_settings", "process_conflict_threshold") + op.drop_column("ml_settings", "process_auto_apply_threshold") + op.drop_column("ml_settings", "process_auto_apply_enabled") diff --git a/alembic/versions/0087_baseline.py b/alembic/versions/0087_baseline.py deleted file mode 100644 index bf803ae..0000000 --- a/alembic/versions/0087_baseline.py +++ /dev/null @@ -1,872 +0,0 @@ -"""Collapsed baseline — the whole schema in one revision. - -Replaces revisions 0001..0087, which narrated the build-out of this project -and were deleted in milestone 328 step 1. A new install creates the schema in -one step instead of replaying that history. - -WHY THE REVISION ID IS "0087" AND NOT "0001" --------------------------------------------- -It is deliberately the id of the LAST revision this baseline collapses, so an -existing database needs no intervention at all: - - * a fresh install finds current=none, head=0087, runs this file once, and - ends stamped at 0087. - * an existing install is ALREADY at 0087, so `alembic upgrade head` finds - current == head and does nothing. - -The alternative — numbering this 0001 and stamping every existing database — -means running `alembic stamp` against live data, and stamp VALIDATES NOTHING. -It writes a version string whether or not the schema actually matches, so a -wrong baseline would be discovered later, by the next real migration, with no -clean way back. Keeping the id removes that operation instead of making it -safe. Future revisions continue at 0088. - -The one case this makes worse, and it fails LOUDLY rather than silently: a -database still sitting between 0001 and 0086 (i.e. never upgraded to head) -cannot be located in this chain and errors out. Upgrade to 0087 on a -pre-squash build first, then take this one. - -WHAT IS HAND-WRITTEN HERE -------------------------- -Most of this file is `alembic revision --autogenerate` output, but four -things are NOT in SQLAlchemy metadata and the generator cannot produce them. -Each fails differently, and none of them fail at generation time: - - 1. CREATE EXTENSION vector (was 0001) — without it the VECTOR - columns below cannot be created at all. - 2. CREATE EXTENSION tsm_system_rows (was 0004) — used by the random-sample - query path; its absence surfaces only when that query runs. - 3. The HNSW index on image_record.siglip_embedding (was 0036). Raw SQL - because alembic's create_index cannot express `USING hnsw (... - vector_cosine_ops)`. Its absence is the quietest failure of the four: - everything works, similarity search just stops using an index. - 4. `import pgvector.sqlalchemy.vector`. Autogenerate EMITS references to - pgvector.sqlalchemy.vector.VECTOR but does not add the import, so the - generated file dies with NameError on first run. - -The acceptance test for this file is not that it reads correctly — it is -`.forgejo/workflows/baseline.yml`, which builds a database from the old -0001..0087 chain (read out of git) and one from this file, and diffs -pg_dump --schema-only output. That is what proves nothing was missed. - -Revision ID: 0087 -Revises: -Create Date: 2026-08-30 - -""" -from typing import Sequence, Union - -from alembic import op -import sqlalchemy as sa -from sqlalchemy.dialects import postgresql - -# Autogenerate references pgvector.sqlalchemy.vector.VECTOR without importing -# it. Item 4 above. -import pgvector.sqlalchemy.vector - -revision: str = "0087" -down_revision: Union[str, None] = None -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # Extensions FIRST: the VECTOR columns below cannot be created without - # `vector`, so ordering here is load-bearing, not tidiness. - op.execute("CREATE EXTENSION IF NOT EXISTS vector") - op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows") - - op.create_table('app_setting', - sa.Column('key', sa.String(length=64), nullable=False), - sa.Column('value', sa.Text(), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.PrimaryKeyConstraint('key', name=op.f('pk_app_setting')) - ) - op.create_table('artist', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('name', sa.String(length=255), nullable=False), - sa.Column('slug', sa.String(length=255), nullable=False), - sa.Column('notes', sa.Text(), nullable=True), - sa.Column('is_subscription', sa.Boolean(), nullable=False), - sa.Column('auto_check', sa.Boolean(), nullable=False), - sa.Column('check_interval_seconds', sa.Integer(), nullable=True), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.PrimaryKeyConstraint('id', name=op.f('pk_artist')), - sa.UniqueConstraint('slug', name=op.f('uq_artist_slug')) - ) - op.create_table('backup_run', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('kind', sa.String(length=16), nullable=False), - sa.Column('status', sa.String(length=16), nullable=False), - sa.Column('tag', sa.String(length=64), nullable=True), - sa.Column('triggered_by', sa.String(length=32), nullable=False), - sa.Column('started_at', sa.DateTime(timezone=True), nullable=False), - sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('sql_path', sa.Text(), nullable=True), - sa.Column('tar_path', sa.Text(), nullable=True), - sa.Column('size_bytes', sa.BigInteger(), nullable=True), - sa.Column('error', sa.Text(), nullable=True), - sa.Column('manifest', sa.JSON(), server_default='{}', nullable=False), - sa.Column('restored_from_id', sa.Integer(), nullable=True), - sa.ForeignKeyConstraint(['restored_from_id'], ['backup_run.id'], name=op.f('fk_backup_run_restored_from_id_backup_run'), ondelete='SET NULL'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_backup_run')) - ) - op.create_index(op.f('ix_backup_run_finished_at'), 'backup_run', ['finished_at'], unique=False) - op.create_index(op.f('ix_backup_run_kind'), 'backup_run', ['kind'], unique=False) - op.create_index(op.f('ix_backup_run_started_at'), 'backup_run', ['started_at'], unique=False) - op.create_index(op.f('ix_backup_run_status'), 'backup_run', ['status'], unique=False) - op.create_index(op.f('ix_backup_run_tag'), 'backup_run', ['tag'], unique=False) - op.create_table('credential', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('platform', sa.String(length=64), nullable=False), - sa.Column('credential_type', sa.String(length=32), nullable=False), - sa.Column('encrypted_blob', sa.LargeBinary(), nullable=False), - sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('expires_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('last_verified', sa.DateTime(timezone=True), nullable=True), - sa.PrimaryKeyConstraint('id', name=op.f('pk_credential')), - sa.UniqueConstraint('platform', name=op.f('uq_credential_platform')) - ) - op.create_table('head_auto_apply_run', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('dry_run', sa.Boolean(), nullable=False), - sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False), - sa.Column('status', sa.String(length=16), nullable=False), - sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('n_applied', sa.Integer(), nullable=True), - sa.Column('report', postgresql.JSONB(astext_type=sa.Text()), nullable=True), - sa.Column('error', sa.Text(), nullable=True), - sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True), - sa.PrimaryKeyConstraint('id', name=op.f('pk_head_auto_apply_run')) - ) - op.create_index(op.f('ix_head_auto_apply_run_status'), 'head_auto_apply_run', ['status'], unique=False) - op.create_table('head_training_run', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False), - sa.Column('status', sa.String(length=16), nullable=False), - sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('n_trained', sa.Integer(), nullable=True), - sa.Column('n_skipped', sa.Integer(), nullable=True), - sa.Column('error', sa.Text(), nullable=True), - sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True), - sa.PrimaryKeyConstraint('id', name=op.f('pk_head_training_run')) - ) - op.create_index(op.f('ix_head_training_run_status'), 'head_training_run', ['status'], unique=False) - op.create_table('import_batch', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('triggered_by', sa.String(length=32), nullable=False), - sa.Column('source_path', sa.Text(), nullable=False), - sa.Column('scan_mode', sa.String(length=16), nullable=False), - sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('total_files', sa.Integer(), nullable=False), - sa.Column('imported', sa.Integer(), nullable=False), - sa.Column('skipped', sa.Integer(), nullable=False), - sa.Column('failed', sa.Integer(), nullable=False), - sa.Column('attachments', sa.Integer(), nullable=False), - sa.Column('refreshed', sa.Integer(), nullable=False), - sa.Column('status', sa.String(length=16), nullable=False), - sa.PrimaryKeyConstraint('id', name=op.f('pk_import_batch')) - ) - op.create_index(op.f('ix_import_batch_status'), 'import_batch', ['status'], unique=False) - op.create_table('import_settings', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('import_scan_path', sa.Text(), nullable=False), - sa.Column('min_width', sa.Integer(), nullable=False), - sa.Column('min_height', sa.Integer(), nullable=False), - sa.Column('skip_transparent', sa.Boolean(), nullable=False), - sa.Column('transparency_threshold', sa.Float(), nullable=False), - sa.Column('skip_single_color', sa.Boolean(), nullable=False), - sa.Column('single_color_threshold', sa.Float(), nullable=False), - sa.Column('single_color_tolerance', sa.Integer(), nullable=False), - sa.Column('phash_threshold', sa.Integer(), nullable=False), - sa.Column('download_rate_limit_seconds', sa.Float(), nullable=False), - sa.Column('download_validate_files', sa.Boolean(), nullable=False), - sa.Column('download_schedule_default_seconds', sa.Integer(), nullable=False), - sa.Column('download_event_retention_days', sa.Integer(), nullable=False), - sa.Column('download_failure_warning_threshold', sa.Integer(), nullable=False), - sa.Column('backup_db_nightly_enabled', sa.Boolean(), nullable=False), - sa.Column('backup_db_nightly_hour_utc', sa.Integer(), nullable=False), - sa.Column('backup_db_keep_last_n', sa.Integer(), nullable=False), - sa.Column('backup_images_keep_last_n', sa.Integer(), nullable=False), - sa.Column('series_suggest_enabled', sa.Boolean(), nullable=False), - sa.Column('series_suggest_threshold', sa.Float(), nullable=False), - sa.Column('extdl_mega_enabled', sa.Boolean(), server_default='true', nullable=False), - sa.Column('extdl_gdrive_enabled', sa.Boolean(), server_default='true', nullable=False), - sa.Column('extdl_mediafire_enabled', sa.Boolean(), server_default='true', nullable=False), - sa.Column('extdl_dropbox_enabled', sa.Boolean(), server_default='true', nullable=False), - sa.Column('extdl_pixeldrain_enabled', sa.Boolean(), server_default='true', nullable=False), - sa.Column('translation_enabled', sa.Boolean(), server_default='false', nullable=False), - sa.Column('interpreter_base_url', sa.Text(), server_default='', nullable=False), - sa.Column('translation_target_lang', sa.Text(), server_default='en', nullable=False), - sa.Column('translation_min_confidence', sa.Float(), server_default='0.9', nullable=False), - sa.Column('wip_title_tagging_enabled', sa.Boolean(), server_default='true', nullable=False), - sa.Column('wip_soft_title_tagging_enabled', sa.Boolean(), server_default='false', nullable=False), - sa.CheckConstraint('id = 1', name=op.f('ck_import_settings_singleton')), - sa.PrimaryKeyConstraint('id', name=op.f('pk_import_settings')) - ) - op.create_table('library_audit_run', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('rule', sa.String(length=32), nullable=False), - sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False), - sa.Column('status', sa.String(length=16), nullable=False), - sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('scanned_count', sa.Integer(), nullable=False), - sa.Column('matched_count', sa.Integer(), nullable=False), - sa.Column('matched_ids', postgresql.JSONB(astext_type=sa.Text()), nullable=False), - sa.Column('error', sa.Text(), nullable=True), - sa.Column('resume_after_id', sa.Integer(), nullable=False), - sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True), - sa.PrimaryKeyConstraint('id', name=op.f('pk_library_audit_run')) - ) - op.create_index(op.f('ix_library_audit_run_rule'), 'library_audit_run', ['rule'], unique=False) - op.create_index(op.f('ix_library_audit_run_status'), 'library_audit_run', ['status'], unique=False) - op.create_table('ml_settings', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('cpu_embed_enabled', sa.Boolean(), nullable=False), - sa.Column('video_frame_interval_seconds', sa.Float(), nullable=False), - sa.Column('video_max_frames', sa.Integer(), nullable=False), - sa.Column('head_min_positives', sa.Integer(), nullable=False), - sa.Column('head_auto_apply_precision', sa.Float(), nullable=False), - sa.Column('head_auto_apply_enabled', sa.Boolean(), nullable=False), - sa.Column('head_auto_apply_min_positives', sa.Integer(), nullable=False), - sa.Column('ccip_match_threshold', sa.Float(), nullable=False), - sa.Column('ccip_auto_apply_enabled', sa.Boolean(), nullable=False), - sa.Column('ccip_auto_apply_threshold', sa.Float(), nullable=False), - sa.Column('presentation_auto_apply_enabled', sa.Boolean(), nullable=False), - sa.Column('presentation_auto_apply_threshold', sa.Float(), nullable=False), - sa.Column('presentation_conflict_threshold', sa.Float(), nullable=False), - sa.Column('process_auto_apply_enabled', sa.Boolean(), nullable=False), - sa.Column('process_auto_apply_threshold', sa.Float(), nullable=False), - sa.Column('process_conflict_threshold', sa.Float(), nullable=False), - sa.Column('embedder_model_version', sa.String(length=128), nullable=False), - sa.Column('embedder_model_name', sa.String(length=128), nullable=False), - sa.Column('detector_person_enabled', sa.Boolean(), nullable=False), - sa.Column('detector_person_weights', sa.String(length=512), nullable=False), - sa.Column('detector_person_conf', sa.Float(), nullable=False), - sa.Column('detector_anatomy_enabled', sa.Boolean(), nullable=False), - sa.Column('detector_anatomy_weights', sa.String(length=512), nullable=False), - sa.Column('detector_anatomy_conf', sa.Float(), nullable=False), - sa.Column('detector_panel_enabled', sa.Boolean(), nullable=False), - sa.Column('detector_panel_weights', sa.String(length=512), nullable=False), - sa.Column('detector_panel_conf', sa.Float(), nullable=False), - sa.Column('detector_max_figures', sa.Integer(), nullable=False), - sa.Column('detector_max_components', sa.Integer(), nullable=False), - sa.Column('detector_max_panels', sa.Integer(), nullable=False), - sa.Column('detector_max_regions', sa.Integer(), nullable=False), - sa.Column('detector_dedupe_iou', sa.Float(), nullable=False), - sa.Column('ccip_ref_signature', sa.String(length=128), nullable=True), - sa.Column('ccip_prototype_cap', sa.Integer(), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.CheckConstraint('id = 1', name=op.f('ck_ml_settings_singleton')), - sa.PrimaryKeyConstraint('id', name=op.f('pk_ml_settings')) - ) - op.create_table('tag', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('name', sa.String(length=255), nullable=False), - sa.Column('kind', sa.Enum('artist', 'character', 'fandom', 'general', 'series', 'archive', 'post', name='tag_kind'), nullable=False), - sa.Column('fandom_id', sa.Integer(), nullable=True), - sa.Column('is_system', sa.Boolean(), server_default=sa.text('false'), nullable=False), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.CheckConstraint("(fandom_id IS NULL) OR (kind = 'character')", name=op.f('ck_tag_ck_tag_fandom_requires_character')), - sa.ForeignKeyConstraint(['fandom_id'], ['tag.id'], name=op.f('fk_tag_fandom_id_tag'), ondelete='SET NULL'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_tag')) - ) - op.create_index(op.f('ix_tag_fandom_id'), 'tag', ['fandom_id'], unique=False) - op.create_table('task_run', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('celery_task_id', sa.String(length=64), nullable=False), - sa.Column('queue', sa.String(length=32), nullable=False), - sa.Column('task_name', sa.String(length=128), nullable=False), - sa.Column('target_id', sa.Integer(), nullable=True), - sa.Column('started_at', sa.DateTime(timezone=True), nullable=False), - sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('duration_ms', sa.Integer(), nullable=True), - sa.Column('status', sa.String(length=16), nullable=False), - sa.Column('error_type', sa.String(length=128), nullable=True), - sa.Column('error_message', sa.Text(), nullable=True), - sa.Column('retry_count', sa.Integer(), nullable=True), - sa.Column('worker_hostname', sa.String(length=128), nullable=True), - sa.Column('args_summary', sa.String(length=255), nullable=True), - sa.PrimaryKeyConstraint('id', name=op.f('pk_task_run')) - ) - op.create_index(op.f('ix_task_run_celery_task_id'), 'task_run', ['celery_task_id'], unique=False) - op.create_index(op.f('ix_task_run_finished_at'), 'task_run', ['finished_at'], unique=False) - op.create_index(op.f('ix_task_run_queue'), 'task_run', ['queue'], unique=False) - op.create_index(op.f('ix_task_run_started_at'), 'task_run', ['started_at'], unique=False) - op.create_index(op.f('ix_task_run_status'), 'task_run', ['status'], unique=False) - op.create_index(op.f('ix_task_run_task_name'), 'task_run', ['task_name'], unique=False) - op.create_table('artist_visit', - sa.Column('artist_id', sa.Integer(), nullable=False), - sa.Column('last_viewed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_artist_visit_artist_id_artist'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('artist_id', name=op.f('pk_artist_visit')) - ) - op.create_table('ccip_prototype_state', - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('fingerprint', sa.String(length=64), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_ccip_prototype_state_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_ccip_prototype_state')) - ) - op.create_table('head_metric', - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('n_misfires', sa.Integer(), nullable=False), - sa.Column('n_underfires', sa.Integer(), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metric_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_head_metric')) - ) - op.create_table('head_metrics_snapshot', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('name', sa.String(length=255), nullable=False), - sa.Column('snapshot_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('n_auto_applied', sa.Integer(), nullable=False), - sa.Column('n_misfires', sa.Integer(), nullable=False), - sa.Column('n_underfires', sa.Integer(), nullable=False), - sa.Column('ap', sa.Float(), nullable=True), - sa.Column('precision_cv', sa.Float(), nullable=True), - sa.Column('recall', sa.Float(), nullable=True), - sa.Column('n_pos', sa.Integer(), nullable=True), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metrics_snapshot_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_head_metrics_snapshot')) - ) - op.create_index(op.f('ix_head_metrics_snapshot_snapshot_at'), 'head_metrics_snapshot', ['snapshot_at'], unique=False) - op.create_index(op.f('ix_head_metrics_snapshot_tag_id'), 'head_metrics_snapshot', ['tag_id'], unique=False) - op.create_table('source', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('artist_id', sa.Integer(), nullable=False), - sa.Column('platform', sa.String(length=64), nullable=False), - sa.Column('url', sa.Text(), nullable=False), - sa.Column('enabled', sa.Boolean(), nullable=False), - sa.Column('config_overrides', sa.JSON(), nullable=True), - sa.Column('last_checked_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('last_error', sa.Text(), nullable=True), - sa.Column('error_type', sa.String(length=32), nullable=True), - sa.Column('check_interval_override', sa.Integer(), nullable=True), - sa.Column('consecutive_failures', sa.Integer(), nullable=False), - sa.Column('backfill_runs_remaining', sa.Integer(), server_default='0', nullable=False), - sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_source_artist_id_artist'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_source')) - ) - op.create_index(op.f('ix_source_artist_id'), 'source', ['artist_id'], unique=False) - op.create_index(op.f('ix_source_error_type'), 'source', ['error_type'], unique=False) - op.create_table('tag_alias', - sa.Column('alias_string', sa.String(length=255), nullable=False), - sa.Column('alias_category', sa.String(length=32), nullable=False), - sa.Column('canonical_tag_id', sa.Integer(), nullable=False), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['canonical_tag_id'], ['tag.id'], name=op.f('fk_tag_alias_canonical_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('alias_string', 'alias_category', name=op.f('pk_tag_alias')) - ) - op.create_index(op.f('ix_tag_alias_canonical_tag_id'), 'tag_alias', ['canonical_tag_id'], unique=False) - op.create_table('tag_head', - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('embedding_version', sa.String(length=128), nullable=False), - sa.Column('weights', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=False), - sa.Column('bias', sa.Float(), nullable=False), - sa.Column('suggest_threshold', sa.Float(), nullable=False), - sa.Column('auto_apply_threshold', sa.Float(), nullable=True), - sa.Column('n_pos', sa.Integer(), nullable=False), - sa.Column('n_neg', sa.Integer(), nullable=False), - sa.Column('ap', sa.Float(), nullable=False), - sa.Column('precision_cv', sa.Float(), nullable=False), - sa.Column('recall', sa.Float(), nullable=False), - sa.Column('trained_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('train_fingerprint', sa.String(length=128), nullable=True), - sa.Column('metrics', postgresql.JSONB(astext_type=sa.Text()), nullable=True), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_head_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_tag_head')) - ) - op.create_table('patreon_failed_media', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=False), - sa.Column('filehash', sa.String(length=128), nullable=False), - sa.Column('attempts', sa.Integer(), nullable=False), - sa.Column('last_error', sa.Text(), nullable=True), - sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_patreon_failed_media_source_id_source'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_patreon_failed_media')), - sa.UniqueConstraint('source_id', 'filehash', name='uq_patreon_failed_media_source_id') - ) - op.create_index(op.f('ix_patreon_failed_media_source_id'), 'patreon_failed_media', ['source_id'], unique=False) - op.create_table('patreon_seen_media', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=False), - sa.Column('filehash', sa.String(length=128), nullable=False), - sa.Column('post_id', sa.String(length=64), nullable=True), - sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_patreon_seen_media_source_id_source'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_patreon_seen_media')), - sa.UniqueConstraint('source_id', 'filehash', name='uq_patreon_seen_media_source_id') - ) - op.create_index(op.f('ix_patreon_seen_media_source_id'), 'patreon_seen_media', ['source_id'], unique=False) - op.create_table('pixiv_failed_media', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=False), - sa.Column('filehash', sa.String(length=128), nullable=False), - sa.Column('attempts', sa.Integer(), nullable=False), - sa.Column('last_error', sa.Text(), nullable=True), - sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_pixiv_failed_media_source_id_source'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_pixiv_failed_media')), - sa.UniqueConstraint('source_id', 'filehash', name='uq_pixiv_failed_media_source_id') - ) - op.create_index(op.f('ix_pixiv_failed_media_source_id'), 'pixiv_failed_media', ['source_id'], unique=False) - op.create_table('pixiv_seen_media', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=False), - sa.Column('filehash', sa.String(length=128), nullable=False), - sa.Column('post_id', sa.String(length=64), nullable=True), - sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_pixiv_seen_media_source_id_source'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_pixiv_seen_media')), - sa.UniqueConstraint('source_id', 'filehash', name='uq_pixiv_seen_media_source_id') - ) - op.create_index(op.f('ix_pixiv_seen_media_source_id'), 'pixiv_seen_media', ['source_id'], unique=False) - op.create_table('post', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=True), - sa.Column('artist_id', sa.Integer(), nullable=False), - sa.Column('external_post_id', sa.String(length=128), nullable=False), - sa.Column('post_url', sa.Text(), nullable=True), - sa.Column('post_title', sa.Text(), nullable=True), - sa.Column('post_date', sa.DateTime(timezone=True), nullable=True), - sa.Column('raw_metadata', sa.JSON(), nullable=True), - sa.Column('description', sa.Text(), nullable=True), - sa.Column('attachment_count', sa.Integer(), nullable=True), - sa.Column('post_title_translated', sa.Text(), nullable=True), - sa.Column('description_translated', sa.Text(), nullable=True), - sa.Column('translated_source_lang', sa.String(length=8), nullable=True), - sa.Column('translation_engine_version', sa.String(length=128), nullable=True), - sa.Column('translated_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('translation_override', sa.String(length=16), server_default='auto', nullable=False), - sa.Column('downloaded_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.CheckConstraint("translation_override IN ('auto', 'force', 'original')", name=op.f('ck_post_ck_post_translation_override')), - sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_artist_id_artist'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_post_source_id_source'), ondelete='SET NULL'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_post')), - sa.UniqueConstraint('source_id', 'external_post_id', name='uq_post_source_external_id') - ) - op.create_index(op.f('ix_post_artist_id'), 'post', ['artist_id'], unique=False) - op.create_index(op.f('ix_post_source_id'), 'post', ['source_id'], unique=False) - op.create_table('subscribestar_failed_media', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=False), - sa.Column('filehash', sa.String(length=128), nullable=False), - sa.Column('attempts', sa.Integer(), nullable=False), - sa.Column('last_error', sa.Text(), nullable=True), - sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_subscribestar_failed_media_source_id_source'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_subscribestar_failed_media')), - sa.UniqueConstraint('source_id', 'filehash', name='uq_subscribestar_failed_media_source_id') - ) - op.create_index(op.f('ix_subscribestar_failed_media_source_id'), 'subscribestar_failed_media', ['source_id'], unique=False) - op.create_table('subscribestar_seen_media', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=False), - sa.Column('filehash', sa.String(length=128), nullable=False), - sa.Column('post_id', sa.String(length=64), nullable=True), - sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_subscribestar_seen_media_source_id_source'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_subscribestar_seen_media')), - sa.UniqueConstraint('source_id', 'filehash', name='uq_subscribestar_seen_media_source_id') - ) - op.create_index(op.f('ix_subscribestar_seen_media_source_id'), 'subscribestar_seen_media', ['source_id'], unique=False) - op.create_table('download_event', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=False), - sa.Column('post_id', sa.Integer(), nullable=True), - sa.Column('status', sa.String(length=32), nullable=False), - sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('bytes_downloaded', sa.BigInteger(), nullable=False), - sa.Column('files_count', sa.Integer(), nullable=False), - sa.Column('error', sa.Text(), nullable=True), - sa.Column('metadata', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'{}'::jsonb"), nullable=False), - sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_download_event_post_id_post'), ondelete='SET NULL'), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_download_event_source_id_source'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_download_event')) - ) - op.create_index(op.f('ix_download_event_post_id'), 'download_event', ['post_id'], unique=False) - op.create_index(op.f('ix_download_event_source_id'), 'download_event', ['source_id'], unique=False) - op.create_table('image_record', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('path', sa.Text(), nullable=False), - sa.Column('sha256', sa.String(length=64), nullable=False), - sa.Column('phash', sa.String(length=32), nullable=True), - sa.Column('size_bytes', sa.BigInteger(), nullable=False), - sa.Column('mime', sa.String(length=64), nullable=False), - sa.Column('width', sa.Integer(), nullable=True), - sa.Column('height', sa.Integer(), nullable=True), - sa.Column('duration_seconds', sa.Float(), nullable=True), - sa.Column('integrity_status', sa.String(length=24), nullable=False), - sa.Column('thumbnail_path', sa.Text(), nullable=True), - sa.Column('source_url', sa.Text(), nullable=True), - sa.Column('source_filehash', sa.String(length=32), nullable=True), - sa.Column('origin', sa.Enum('downloaded', 'imported_filesystem', 'uploaded', name='origin_enum'), nullable=False), - sa.Column('primary_post_id', sa.Integer(), nullable=True), - sa.Column('artist_id', sa.Integer(), nullable=True), - sa.Column('siglip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=True), - sa.Column('siglip_model_version', sa.String(length=128), nullable=True), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('effective_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('earliest_post_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_image_record_artist_id_artist'), ondelete='SET NULL'), - sa.ForeignKeyConstraint(['primary_post_id'], ['post.id'], name=op.f('fk_image_record_primary_post_id_post'), ondelete='SET NULL'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_image_record')), - sa.UniqueConstraint('path', name=op.f('uq_image_record_path')) - ) - op.create_index(op.f('ix_image_record_artist_id'), 'image_record', ['artist_id'], unique=False) - op.create_index(op.f('ix_image_record_integrity_status'), 'image_record', ['integrity_status'], unique=False) - op.create_index(op.f('ix_image_record_phash'), 'image_record', ['phash'], unique=False) - op.create_index(op.f('ix_image_record_primary_post_id'), 'image_record', ['primary_post_id'], unique=False) - op.create_index(op.f('ix_image_record_sha256'), 'image_record', ['sha256'], unique=True) - op.create_index(op.f('ix_image_record_source_filehash'), 'image_record', ['source_filehash'], unique=False) - op.create_table('post_attachment', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('post_id', sa.Integer(), nullable=True), - sa.Column('artist_id', sa.Integer(), nullable=True), - sa.Column('sha256', sa.String(length=64), nullable=False), - sa.Column('path', sa.Text(), nullable=False), - sa.Column('original_filename', sa.Text(), nullable=False), - sa.Column('ext', sa.String(length=32), nullable=False), - sa.Column('mime', sa.String(length=128), nullable=True), - sa.Column('size_bytes', sa.BigInteger(), nullable=False), - sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_attachment_artist_id_artist'), ondelete='SET NULL'), - sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_post_attachment_post_id_post'), ondelete='SET NULL'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_post_attachment')) - ) - op.create_index(op.f('ix_post_attachment_artist_id'), 'post_attachment', ['artist_id'], unique=False) - op.create_index(op.f('ix_post_attachment_post_id'), 'post_attachment', ['post_id'], unique=False) - op.create_index(op.f('ix_post_attachment_sha256'), 'post_attachment', ['sha256'], unique=False) - op.create_index('uq_post_attachment_null_post_sha', 'post_attachment', ['sha256'], unique=True, postgresql_where=sa.text('post_id IS NULL')) - op.create_index('uq_post_attachment_post_sha', 'post_attachment', ['post_id', 'sha256'], unique=True, postgresql_where=sa.text('post_id IS NOT NULL')) - op.create_table('series_suggestion', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('post_id', sa.Integer(), nullable=False), - sa.Column('series_tag_id', sa.Integer(), nullable=False), - sa.Column('score', sa.Float(), nullable=False), - sa.Column('signals', sa.JSON(), nullable=True), - sa.Column('status', sa.String(length=16), server_default='pending', nullable=False), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_series_suggestion_post_id_post'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_suggestion_series_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_series_suggestion')), - sa.UniqueConstraint('post_id', 'series_tag_id', name='uq_series_suggestion_post_series') - ) - op.create_index(op.f('ix_series_suggestion_post_id'), 'series_suggestion', ['post_id'], unique=False) - op.create_index(op.f('ix_series_suggestion_series_tag_id'), 'series_suggestion', ['series_tag_id'], unique=False) - op.create_index(op.f('ix_series_suggestion_status'), 'series_suggestion', ['status'], unique=False) - op.create_table('external_link', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('post_id', sa.Integer(), nullable=False), - sa.Column('artist_id', sa.Integer(), nullable=True), - sa.Column('host', sa.String(length=16), nullable=False), - sa.Column('url', sa.Text(), nullable=False), - sa.Column('label', sa.Text(), nullable=True), - sa.Column('status', sa.String(length=16), server_default='pending', nullable=False), - sa.Column('attempts', sa.Integer(), server_default=sa.text('0'), nullable=False), - sa.Column('last_error', sa.Text(), nullable=True), - sa.Column('attachment_id', sa.Integer(), nullable=True), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('completed_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('duration_seconds', sa.Float(), nullable=True), - sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_external_link_artist_id_artist'), ondelete='SET NULL'), - sa.ForeignKeyConstraint(['attachment_id'], ['post_attachment.id'], name=op.f('fk_external_link_attachment_id_post_attachment'), ondelete='SET NULL'), - sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_external_link_post_id_post'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_external_link')) - ) - op.create_index(op.f('ix_external_link_artist_id'), 'external_link', ['artist_id'], unique=False) - op.create_index(op.f('ix_external_link_post_id'), 'external_link', ['post_id'], unique=False) - op.create_index('ix_external_link_status', 'external_link', ['status'], unique=False) - op.create_index('uq_external_link_post_url', 'external_link', ['post_id', 'url'], unique=True) - op.create_table('gpu_job', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('image_record_id', sa.Integer(), nullable=False), - sa.Column('task', sa.String(length=32), nullable=False), - sa.Column('status', sa.String(length=16), nullable=False), - sa.Column('lease_token', sa.String(length=64), nullable=True), - sa.Column('leased_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('lease_expires_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('attempts', sa.Integer(), nullable=False), - sa.Column('error', sa.Text(), nullable=True), - sa.Column('triage_status', sa.String(length=16), nullable=True), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_gpu_job_image_record_id_image_record'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_gpu_job')) - ) - op.create_index(op.f('ix_gpu_job_image_record_id'), 'gpu_job', ['image_record_id'], unique=False) - op.create_index('ix_gpu_job_leased_expires', 'gpu_job', ['lease_expires_at'], unique=False, postgresql_where=sa.text("status = 'leased'")) - op.create_index('ix_gpu_job_pending', 'gpu_job', ['id'], unique=False, postgresql_where=sa.text("status = 'pending'")) - op.create_index(op.f('ix_gpu_job_status'), 'gpu_job', ['status'], unique=False) - op.create_table('image_provenance', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('image_record_id', sa.Integer(), nullable=False), - sa.Column('post_id', sa.Integer(), nullable=False), - sa.Column('source_id', sa.Integer(), nullable=True), - sa.Column('from_attachment_id', sa.Integer(), nullable=True), - sa.Column('captured_metadata', sa.JSON(), nullable=True), - sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['from_attachment_id'], ['post_attachment.id'], name=op.f('fk_image_provenance_from_attachment_id_post_attachment'), ondelete='SET NULL'), - sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_provenance_image_record_id_image_record'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_image_provenance_post_id_post'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_image_provenance_source_id_source'), ondelete='SET NULL'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_image_provenance')), - sa.UniqueConstraint('image_record_id', 'post_id', name='uq_image_provenance_image_post') - ) - op.create_index(op.f('ix_image_provenance_from_attachment_id'), 'image_provenance', ['from_attachment_id'], unique=False) - op.create_index(op.f('ix_image_provenance_image_record_id'), 'image_provenance', ['image_record_id'], unique=False) - op.create_index(op.f('ix_image_provenance_post_id'), 'image_provenance', ['post_id'], unique=False) - op.create_index(op.f('ix_image_provenance_source_id'), 'image_provenance', ['source_id'], unique=False) - op.create_table('image_region', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('image_record_id', sa.Integer(), nullable=False), - sa.Column('kind', sa.String(length=16), nullable=False), - sa.Column('frame_time', sa.Float(), nullable=True), - sa.Column('rx', sa.Float(), nullable=False), - sa.Column('ry', sa.Float(), nullable=False), - sa.Column('rw', sa.Float(), nullable=False), - sa.Column('rh', sa.Float(), nullable=False), - sa.Column('score', sa.Float(), nullable=True), - sa.Column('detector_version', sa.String(length=64), nullable=True), - sa.Column('crop_version', sa.String(length=64), nullable=True), - sa.Column('embedding_version', sa.String(length=128), nullable=True), - sa.Column('ccip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=768), nullable=True), - sa.Column('siglip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=True), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_region_image_record_id_image_record'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_image_region')) - ) - op.create_index(op.f('ix_image_region_image_record_id'), 'image_region', ['image_record_id'], unique=False) - op.create_table('image_tag', - sa.Column('image_record_id', sa.Integer(), nullable=False), - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('source', sa.String(length=32), nullable=False), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_tag_image_record_id_image_record'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_image_tag_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_image_tag')) - ) - op.create_table('import_task', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('batch_id', sa.Integer(), nullable=False), - sa.Column('source_path', sa.Text(), nullable=False), - sa.Column('task_type', sa.String(length=16), nullable=False), - sa.Column('status', sa.String(length=16), nullable=False), - sa.Column('recovery_count', sa.Integer(), nullable=False), - sa.Column('refetched', sa.Boolean(), nullable=False), - sa.Column('result_image_id', sa.Integer(), nullable=True), - sa.Column('error', sa.Text(), nullable=True), - sa.Column('size_bytes', sa.BigInteger(), nullable=True), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('started_at', sa.DateTime(timezone=True), nullable=True), - sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), - sa.ForeignKeyConstraint(['batch_id'], ['import_batch.id'], name=op.f('fk_import_task_batch_id_import_batch'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['result_image_id'], ['image_record.id'], name=op.f('fk_import_task_result_image_id_image_record'), ondelete='SET NULL'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_import_task')) - ) - op.create_index(op.f('ix_import_task_batch_id'), 'import_task', ['batch_id'], unique=False) - op.create_index(op.f('ix_import_task_status'), 'import_task', ['status'], unique=False) - op.create_table('presentation_review', - sa.Column('image_record_id', sa.Integer(), nullable=False), - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('conflict_tag_id', sa.Integer(), nullable=True), - sa.Column('conflict_score', sa.Float(), nullable=False), - sa.Column('mode', sa.String(length=16), server_default='chrome', nullable=False), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('resolved_at', sa.DateTime(timezone=True), nullable=True), - sa.ForeignKeyConstraint(['conflict_tag_id'], ['tag.id'], name=op.f('fk_presentation_review_conflict_tag_id_tag'), ondelete='SET NULL'), - sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_presentation_review_image_record_id_image_record'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_presentation_review_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_presentation_review')) - ) - op.create_table('series_page', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('series_tag_id', sa.Integer(), nullable=False), - sa.Column('image_id', sa.Integer(), nullable=False), - sa.Column('status', sa.String(length=16), server_default='placed', nullable=False), - sa.Column('page_number', sa.Integer(), nullable=True), - sa.Column('stated_page', sa.Integer(), nullable=True), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['image_id'], ['image_record.id'], name=op.f('fk_series_page_image_id_image_record'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_page_series_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_series_page')), - sa.UniqueConstraint('image_id', name=op.f('uq_series_page_image_id')) - ) - op.create_index(op.f('ix_series_page_series_tag_id'), 'series_page', ['series_tag_id'], unique=False) - op.create_table('tag_positive_confirmation', - sa.Column('image_record_id', sa.Integer(), nullable=False), - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('confirmed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_tag_positive_confirmation_image_record_id_image_record'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_positive_confirmation_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_positive_confirmation')) - ) - op.create_index(op.f('ix_tag_positive_confirmation_tag_id'), 'tag_positive_confirmation', ['tag_id'], unique=False) - op.create_table('tag_suggestion_rejection', - sa.Column('image_record_id', sa.Integer(), nullable=False), - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('rejected_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_tag_suggestion_rejection_image_record_id_image_record'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_suggestion_rejection_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_suggestion_rejection')) - ) - op.create_index(op.f('ix_tag_suggestion_rejection_tag_id'), 'tag_suggestion_rejection', ['tag_id'], unique=False) - op.create_table('character_prototype', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('tag_id', sa.Integer(), nullable=False), - sa.Column('ccip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=768), nullable=False), - sa.Column('region_id', sa.Integer(), nullable=True), - sa.ForeignKeyConstraint(['region_id'], ['image_region.id'], name=op.f('fk_character_prototype_region_id_image_region'), ondelete='SET NULL'), - sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_character_prototype_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_character_prototype')) - ) - op.create_index(op.f('ix_character_prototype_tag_id'), 'character_prototype', ['tag_id'], unique=False) - op.create_table('series_chapter', - sa.Column('id', sa.Integer(), nullable=False), - sa.Column('series_tag_id', sa.Integer(), nullable=False), - sa.Column('anchor_page_id', sa.Integer(), nullable=False), - sa.Column('title', sa.Text(), nullable=True), - sa.Column('stated_part', sa.Integer(), nullable=True), - sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), - sa.ForeignKeyConstraint(['anchor_page_id'], ['series_page.id'], name=op.f('fk_series_chapter_anchor_page_id_series_page'), ondelete='CASCADE'), - sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_chapter_series_tag_id_tag'), ondelete='CASCADE'), - sa.PrimaryKeyConstraint('id', name=op.f('pk_series_chapter')), - sa.UniqueConstraint('anchor_page_id', name=op.f('uq_series_chapter_anchor_page_id')) - ) - op.create_index(op.f('ix_series_chapter_series_tag_id'), 'series_chapter', ['series_tag_id'], unique=False) - - # The HNSW index, item 3 above. Must match the query's cosine-distance - # operator class or the planner will not use it. - op.execute( - "CREATE INDEX ix_image_record_siglip_hnsw " - "ON image_record USING hnsw (siglip_embedding vector_cosine_ops)" - ) - - -def downgrade() -> None: - # Dropping image_record takes its indexes with it, so the HNSW index needs - # no separate drop. The extensions are deliberately left in place: they are - # database-scoped and something else may be using them. - op.drop_index(op.f('ix_series_chapter_series_tag_id'), table_name='series_chapter') - op.drop_table('series_chapter') - op.drop_index(op.f('ix_character_prototype_tag_id'), table_name='character_prototype') - op.drop_table('character_prototype') - op.drop_index(op.f('ix_tag_suggestion_rejection_tag_id'), table_name='tag_suggestion_rejection') - op.drop_table('tag_suggestion_rejection') - op.drop_index(op.f('ix_tag_positive_confirmation_tag_id'), table_name='tag_positive_confirmation') - op.drop_table('tag_positive_confirmation') - op.drop_index(op.f('ix_series_page_series_tag_id'), table_name='series_page') - op.drop_table('series_page') - op.drop_table('presentation_review') - op.drop_index(op.f('ix_import_task_status'), table_name='import_task') - op.drop_index(op.f('ix_import_task_batch_id'), table_name='import_task') - op.drop_table('import_task') - op.drop_table('image_tag') - op.drop_index(op.f('ix_image_region_image_record_id'), table_name='image_region') - op.drop_table('image_region') - op.drop_index(op.f('ix_image_provenance_source_id'), table_name='image_provenance') - op.drop_index(op.f('ix_image_provenance_post_id'), table_name='image_provenance') - op.drop_index(op.f('ix_image_provenance_image_record_id'), table_name='image_provenance') - op.drop_index(op.f('ix_image_provenance_from_attachment_id'), table_name='image_provenance') - op.drop_table('image_provenance') - op.drop_index(op.f('ix_gpu_job_status'), table_name='gpu_job') - op.drop_index('ix_gpu_job_pending', table_name='gpu_job', postgresql_where=sa.text("status = 'pending'")) - op.drop_index('ix_gpu_job_leased_expires', table_name='gpu_job', postgresql_where=sa.text("status = 'leased'")) - op.drop_index(op.f('ix_gpu_job_image_record_id'), table_name='gpu_job') - op.drop_table('gpu_job') - op.drop_index('uq_external_link_post_url', table_name='external_link') - op.drop_index('ix_external_link_status', table_name='external_link') - op.drop_index(op.f('ix_external_link_post_id'), table_name='external_link') - op.drop_index(op.f('ix_external_link_artist_id'), table_name='external_link') - op.drop_table('external_link') - op.drop_index(op.f('ix_series_suggestion_status'), table_name='series_suggestion') - op.drop_index(op.f('ix_series_suggestion_series_tag_id'), table_name='series_suggestion') - op.drop_index(op.f('ix_series_suggestion_post_id'), table_name='series_suggestion') - op.drop_table('series_suggestion') - op.drop_index('uq_post_attachment_post_sha', table_name='post_attachment', postgresql_where=sa.text('post_id IS NOT NULL')) - op.drop_index('uq_post_attachment_null_post_sha', table_name='post_attachment', postgresql_where=sa.text('post_id IS NULL')) - op.drop_index(op.f('ix_post_attachment_sha256'), table_name='post_attachment') - op.drop_index(op.f('ix_post_attachment_post_id'), table_name='post_attachment') - op.drop_index(op.f('ix_post_attachment_artist_id'), table_name='post_attachment') - op.drop_table('post_attachment') - op.drop_index(op.f('ix_image_record_source_filehash'), table_name='image_record') - op.drop_index(op.f('ix_image_record_sha256'), table_name='image_record') - op.drop_index(op.f('ix_image_record_primary_post_id'), table_name='image_record') - op.drop_index(op.f('ix_image_record_phash'), table_name='image_record') - op.drop_index(op.f('ix_image_record_integrity_status'), table_name='image_record') - op.drop_index(op.f('ix_image_record_artist_id'), table_name='image_record') - op.drop_table('image_record') - op.drop_index(op.f('ix_download_event_source_id'), table_name='download_event') - op.drop_index(op.f('ix_download_event_post_id'), table_name='download_event') - op.drop_table('download_event') - op.drop_index(op.f('ix_subscribestar_seen_media_source_id'), table_name='subscribestar_seen_media') - op.drop_table('subscribestar_seen_media') - op.drop_index(op.f('ix_subscribestar_failed_media_source_id'), table_name='subscribestar_failed_media') - op.drop_table('subscribestar_failed_media') - op.drop_index(op.f('ix_post_source_id'), table_name='post') - op.drop_index(op.f('ix_post_artist_id'), table_name='post') - op.drop_table('post') - op.drop_index(op.f('ix_pixiv_seen_media_source_id'), table_name='pixiv_seen_media') - op.drop_table('pixiv_seen_media') - op.drop_index(op.f('ix_pixiv_failed_media_source_id'), table_name='pixiv_failed_media') - op.drop_table('pixiv_failed_media') - op.drop_index(op.f('ix_patreon_seen_media_source_id'), table_name='patreon_seen_media') - op.drop_table('patreon_seen_media') - op.drop_index(op.f('ix_patreon_failed_media_source_id'), table_name='patreon_failed_media') - op.drop_table('patreon_failed_media') - op.drop_table('tag_head') - op.drop_index(op.f('ix_tag_alias_canonical_tag_id'), table_name='tag_alias') - op.drop_table('tag_alias') - op.drop_index(op.f('ix_source_error_type'), table_name='source') - op.drop_index(op.f('ix_source_artist_id'), table_name='source') - op.drop_table('source') - op.drop_index(op.f('ix_head_metrics_snapshot_tag_id'), table_name='head_metrics_snapshot') - op.drop_index(op.f('ix_head_metrics_snapshot_snapshot_at'), table_name='head_metrics_snapshot') - op.drop_table('head_metrics_snapshot') - op.drop_table('head_metric') - op.drop_table('ccip_prototype_state') - op.drop_table('artist_visit') - op.drop_index(op.f('ix_task_run_task_name'), table_name='task_run') - op.drop_index(op.f('ix_task_run_status'), table_name='task_run') - op.drop_index(op.f('ix_task_run_started_at'), table_name='task_run') - op.drop_index(op.f('ix_task_run_queue'), table_name='task_run') - op.drop_index(op.f('ix_task_run_finished_at'), table_name='task_run') - op.drop_index(op.f('ix_task_run_celery_task_id'), table_name='task_run') - op.drop_table('task_run') - op.drop_index(op.f('ix_tag_fandom_id'), table_name='tag') - op.drop_table('tag') - op.drop_table('ml_settings') - op.drop_index(op.f('ix_library_audit_run_status'), table_name='library_audit_run') - op.drop_index(op.f('ix_library_audit_run_rule'), table_name='library_audit_run') - op.drop_table('library_audit_run') - op.drop_table('import_settings') - op.drop_index(op.f('ix_import_batch_status'), table_name='import_batch') - op.drop_table('import_batch') - op.drop_index(op.f('ix_head_training_run_status'), table_name='head_training_run') - op.drop_table('head_training_run') - op.drop_index(op.f('ix_head_auto_apply_run_status'), table_name='head_auto_apply_run') - op.drop_table('head_auto_apply_run') - op.drop_table('credential') - op.drop_index(op.f('ix_backup_run_tag'), table_name='backup_run') - op.drop_index(op.f('ix_backup_run_status'), table_name='backup_run') - op.drop_index(op.f('ix_backup_run_started_at'), table_name='backup_run') - op.drop_index(op.f('ix_backup_run_kind'), table_name='backup_run') - op.drop_index(op.f('ix_backup_run_finished_at'), table_name='backup_run') - op.drop_table('backup_run') - op.drop_table('artist') - op.drop_table('app_setting') diff --git a/alembic/versions/0087_wip_soft_title_tagging.py b/alembic/versions/0087_wip_soft_title_tagging.py new file mode 100644 index 0000000..58478e1 --- /dev/null +++ b/alembic/versions/0087_wip_soft_title_tagging.py @@ -0,0 +1,33 @@ +"""soft WIP title tier toggle (#1474) — ImportSettings.wip_soft_title_tagging_enabled + +The soft tier also tags sketch/doodle/scribble titles, but with a provisional source +that never trains the head. OFF by default (a lower-precision tier is opt-in). +server_default so the existing singleton row (id=1) fills cleanly. + +Revision ID: 0087 +Revises: 0086 +Create Date: 2026-07-13 +""" +from typing import Sequence, Union + +import sqlalchemy as sa +from alembic import op + +revision: str = "0087" +down_revision: Union[str, None] = "0086" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.add_column( + "import_settings", + sa.Column( + "wip_soft_title_tagging_enabled", sa.Boolean(), nullable=False, + server_default=sa.text("false"), + ), + ) + + +def downgrade() -> None: + op.drop_column("import_settings", "wip_soft_title_tagging_enabled") diff --git a/backend/app/utils/artist_backfill.py b/backend/app/utils/artist_backfill.py new file mode 100644 index 0000000..3979b24 --- /dev/null +++ b/backend/app/utils/artist_backfill.py @@ -0,0 +1,44 @@ +"""Literal SQL for the FC-2d-vii-c artist backfill / artist-tag delete. + +Intentionally pure string constants — NO model/slug imports, NO logic — +so migration 0008 and its test share one drift-proof source of truth. +Backfill steps are ordered primary -> provenance -> artist-tag and each +only touches rows still NULL (idempotent, first match wins). The +artist-tag step matches Artist.name = Tag.name: the importer always +created both from the same artist_name string. +""" + +BACKFILL_PRIMARY_SQL = """ +UPDATE image_record AS ir +SET artist_id = s.artist_id +FROM post p +JOIN source s ON s.id = p.source_id +WHERE ir.primary_post_id = p.id + AND ir.artist_id IS NULL +""" + +BACKFILL_PROVENANCE_SQL = """ +UPDATE image_record AS ir +SET artist_id = s.artist_id +FROM ( + SELECT DISTINCT ON (ip.image_record_id) + ip.image_record_id, src.artist_id + FROM image_provenance ip + JOIN source src ON src.id = ip.source_id + ORDER BY ip.image_record_id, ip.id +) AS s +WHERE ir.id = s.image_record_id + AND ir.artist_id IS NULL +""" + +BACKFILL_TAG_SQL = """ +UPDATE image_record AS ir +SET artist_id = a.id +FROM image_tag it +JOIN tag t ON t.id = it.tag_id AND t.kind = 'artist' +JOIN artist a ON a.name = t.name +WHERE it.image_record_id = ir.id + AND ir.artist_id IS NULL +""" + +DELETE_ARTIST_TAGS_SQL = "DELETE FROM tag WHERE kind = 'artist'" diff --git a/tests/test_migration_0002.py b/tests/test_migration_0002.py new file mode 100644 index 0000000..1786731 --- /dev/null +++ b/tests/test_migration_0002.py @@ -0,0 +1,58 @@ +"""Smoke test for migration 0002: confirms model classes import and the +tag-kind uniqueness rule shape is correct. +""" + +from backend.app.models import ( + Base, + ImportBatch, + ImportSettings, + ImportTask, + Tag, + TagKind, +) + + +def test_new_tables_registered(): + expected = {"import_batch", "import_task", "import_settings"} + assert expected.issubset(Base.metadata.tables.keys()) + + +def test_tag_has_kind_and_fandom_id(): + cols = {c.name for c in Tag.__table__.columns} + assert "kind" in cols + assert "fandom_id" in cols + assert "namespace" not in cols + + +def test_tag_kind_enum_values(): + # Current TagKind enum after alembic 0023 dropped meta + rating + # (operator-retired 2026-05-26). `artist` is still in the enum + # for backward-compat with historical rows, though new artist + # tags don't get created (Artist row is canonical per FC-2d-vii-c). + expected = { + "artist", + "character", + "fandom", + "general", + "series", + "archive", + "post", + } + assert {k.value for k in TagKind} == expected + + +def test_image_record_has_integrity_status(): + from backend.app.models import ImageRecord + cols = {c.name for c in ImageRecord.__table__.columns} + assert "integrity_status" in cols + + +def test_import_task_has_state_columns(): + cols = {c.name for c in ImportTask.__table__.columns} + for required in ("batch_id", "source_path", "task_type", "status", "result_image_id"): + assert required in cols + + +def test_import_settings_singleton_constraint(): + constraints = {c.name for c in ImportSettings.__table__.constraints} + assert "ck_import_settings_singleton" in constraints diff --git a/tests/test_migration_0003.py b/tests/test_migration_0003.py new file mode 100644 index 0000000..1752120 --- /dev/null +++ b/tests/test_migration_0003.py @@ -0,0 +1,46 @@ +"""Smoke test for migration 0003: model classes import, schema shape correct.""" + +from backend.app.models import ( + Base, + ImageRecord, + MLSettings, + TagAlias, + TagSuggestionRejection, +) + + +def test_new_tables_registered(): + expected = { + "tag_suggestion_rejection", + "tag_alias", + "ml_settings", + } + assert expected.issubset(Base.metadata.tables.keys()) + + +def test_image_record_columns_renamed(): + cols = {c.name for c in ImageRecord.__table__.columns} + # Legacy tagger columns are all gone: tagger_predictions/wd14_* dropped in + # 0046, tagger_model_version + centroid_scores dropped in 0068 (#1199, Camie + # retirement). The SigLIP embedding columns are the live ML fields. + assert "siglip_embedding" in cols + assert "siglip_model_version" in cols + assert "tagger_model_version" not in cols + assert "centroid_scores" not in cols + assert "tagger_predictions" not in cols + assert "wd14_predictions" not in cols + + +def test_tag_alias_composite_pk(): + pk_cols = {c.name for c in TagAlias.__table__.primary_key.columns} + assert pk_cols == {"alias_string", "alias_category"} + + +def test_ml_settings_singleton_constraint(): + names = {c.name for c in MLSettings.__table__.constraints} + assert "ck_ml_settings_singleton" in names + + +def test_tag_suggestion_rejection_pk(): + pk_cols = {c.name for c in TagSuggestionRejection.__table__.primary_key.columns} + assert pk_cols == {"image_record_id", "tag_id"} diff --git a/tests/test_migration_0004.py b/tests/test_migration_0004.py new file mode 100644 index 0000000..7975d0c --- /dev/null +++ b/tests/test_migration_0004.py @@ -0,0 +1,27 @@ +"""Integration: the tsm_system_rows extension is installed by migration 0004. + +Needs a real Postgres (CI does not provision one), so integration-marked. +""" + +import pytest +from sqlalchemy import text + +pytestmark = pytest.mark.integration + + +@pytest.mark.asyncio +async def test_tsm_system_rows_extension_present(db): + row = ( + await db.execute( + text("SELECT 1 FROM pg_extension WHERE extname = 'tsm_system_rows'") + ) + ).first() + assert row is not None + + +@pytest.mark.asyncio +async def test_system_rows_sampling_is_usable(db): + # Should parse and execute even on an empty table. + await db.execute( + text("SELECT * FROM image_record TABLESAMPLE SYSTEM_ROWS(1)") + ) diff --git a/tests/test_migration_0007.py b/tests/test_migration_0007.py new file mode 100644 index 0000000..6f716f4 --- /dev/null +++ b/tests/test_migration_0007.py @@ -0,0 +1,48 @@ +"""FC-2d-iv: post.description + post.attachment_count round-trip.""" + +from datetime import UTC, datetime + +import pytest + +from backend.app.models import Artist, Post, Source + +pytestmark = pytest.mark.integration + + +async def _post(db, **post_kwargs): + artist = Artist(name="Nadia", slug="nadia") + db.add(artist) + await db.flush() + src = Source(artist_id=artist.id, platform="web", url="http://x") + db.add(src) + await db.flush() + post = Post( + source_id=src.id, artist_id=artist.id, external_post_id="p1", + post_date=datetime(2026, 3, 1, tzinfo=UTC), + **post_kwargs, + ) + db.add(post) + await db.flush() + return post.id + + +def test_post_has_new_columns(): + cols = {c.name for c in Post.__table__.columns} + assert "description" in cols + assert "attachment_count" in cols + + +@pytest.mark.asyncio +async def test_description_and_attachment_count_round_trip(db): + pid = await _post(db, description="

hi

", attachment_count=3) + row = await db.get(Post, pid) + assert row.description == "

hi

" + assert row.attachment_count == 3 + + +@pytest.mark.asyncio +async def test_new_fields_default_null(db): + pid = await _post(db) + row = await db.get(Post, pid) + assert row.description is None + assert row.attachment_count is None diff --git a/tests/test_migration_0008.py b/tests/test_migration_0008.py new file mode 100644 index 0000000..0d7552a --- /dev/null +++ b/tests/test_migration_0008.py @@ -0,0 +1,137 @@ +"""FC-2d-vii-c: image_record.artist_id + backfill + artist-tag delete.""" + +from datetime import UTC, datetime, timedelta + +import pytest +from sqlalchemy import func, select, text + +from backend.app.models import ( + Artist, + ImageProvenance, + ImageRecord, + Post, + Source, + Tag, + TagKind, +) +from backend.app.models.tag import image_tag +from backend.app.utils.artist_backfill import ( + BACKFILL_PRIMARY_SQL, + BACKFILL_PROVENANCE_SQL, + BACKFILL_TAG_SQL, + DELETE_ARTIST_TAGS_SQL, +) + +pytestmark = pytest.mark.integration + + +def test_image_record_has_artist_id_column(): + assert "artist_id" in {c.name for c in ImageRecord.__table__.columns} + + +async def _img(db, n): + rec = ImageRecord( + path=f"/images/bf/{n}.jpg", sha256=f"bf{n:062d}", + size_bytes=1, mime="image/jpeg", width=1, height=1, + origin="imported_filesystem", integrity_status="unknown", + ) + rec.created_at = datetime.now(UTC) - timedelta(minutes=n) + db.add(rec) + await db.flush() + return rec + + +async def _artist_source(db, name, slug): + a = Artist(name=name, slug=slug) + db.add(a) + await db.flush() + s = Source(artist_id=a.id, platform="patreon", + url=f"https://p.test/{slug}") + db.add(s) + await db.flush() + return a, s + + +async def _run_backfill(db): + await db.execute(text(BACKFILL_PRIMARY_SQL)) + await db.execute(text(BACKFILL_PROVENANCE_SQL)) + await db.execute(text(BACKFILL_TAG_SQL)) + + +@pytest.mark.asyncio +async def test_backfill_primary_post(db): + rec = await _img(db, 1) + a, s = await _artist_source(db, "Alice", "alice") + post = Post(source_id=s.id, artist_id=a.id, external_post_id="1") + db.add(post) + await db.flush() + rec.primary_post_id = post.id + await db.flush() + await _run_backfill(db) + got = await db.scalar( + select(ImageRecord.artist_id).where(ImageRecord.id == rec.id) + ) + assert got == a.id + + +@pytest.mark.asyncio +async def test_backfill_provenance_fallback(db): + rec = await _img(db, 1) + a, s = await _artist_source(db, "Bob", "bob") + post = Post(source_id=s.id, artist_id=a.id, external_post_id="2") + db.add(post) + await db.flush() + db.add(ImageProvenance(image_record_id=rec.id, post_id=post.id, + source_id=s.id)) + await db.flush() + await _run_backfill(db) + got = await db.scalar( + select(ImageRecord.artist_id).where(ImageRecord.id == rec.id) + ) + assert got == a.id + + +@pytest.mark.asyncio +async def test_backfill_artist_tag_by_name(db): + rec = await _img(db, 1) + a = Artist(name="Carol", slug="carol") + db.add(a) + await db.flush() + tag = Tag(name="Carol", kind=TagKind.artist) + db.add(tag) + await db.flush() + await db.execute(image_tag.insert().values( + image_record_id=rec.id, tag_id=tag.id, source="auto")) + await db.flush() + await _run_backfill(db) + got = await db.scalar( + select(ImageRecord.artist_id).where(ImageRecord.id == rec.id) + ) + assert got == a.id + + +@pytest.mark.asyncio +async def test_no_signal_stays_null(db): + rec = await _img(db, 1) + await _run_backfill(db) + got = await db.scalar( + select(ImageRecord.artist_id).where(ImageRecord.id == rec.id) + ) + assert got is None + + +@pytest.mark.asyncio +async def test_delete_removes_only_artist_tags(db): + artist_tag = Tag(name="Dave", kind=TagKind.artist) + general_tag = Tag(name="forest", kind=TagKind.general) + db.add_all([artist_tag, general_tag]) + await db.flush() + await db.execute(text(DELETE_ARTIST_TAGS_SQL)) + remaining = await db.scalar( + select(func.count()).select_from(Tag).where(Tag.kind == TagKind.artist) + ) + assert remaining == 0 + survived = await db.scalar( + select(func.count()).select_from(Tag).where(Tag.kind == TagKind.general) + ) + assert survived >= 1 diff --git a/tests/test_migration_0009.py b/tests/test_migration_0009.py new file mode 100644 index 0000000..f2a0d1f --- /dev/null +++ b/tests/test_migration_0009.py @@ -0,0 +1,37 @@ +"""FC-2d-iii: post_attachment table + import_batch.attachments column.""" + +import pytest + +from backend.app.models import ImportBatch, PostAttachment + +pytestmark = pytest.mark.integration + + +def test_post_attachment_columns(): + cols = {c.name for c in PostAttachment.__table__.columns} + assert { + "id", "post_id", "artist_id", "sha256", "path", + "original_filename", "ext", "mime", "size_bytes", "captured_at", + } <= cols + + +def test_import_batch_has_attachments_counter(): + assert "attachments" in {c.name for c in ImportBatch.__table__.columns} + + +@pytest.mark.asyncio +async def test_post_attachment_roundtrip(db): + from backend.app.models import Artist + + a = Artist(name="Zed", slug="zed") + db.add(a) + await db.flush() + att = PostAttachment( + post_id=None, artist_id=a.id, sha256="z" + "0" * 63, + path="/images/attachments/z00/z.zip", original_filename="pack.zip", + ext=".zip", mime="application/zip", size_bytes=123, + ) + db.add(att) + await db.flush() + got = await db.get(PostAttachment, att.id) + assert got.original_filename == "pack.zip" and got.post_id is None diff --git a/tests/test_migration_0010.py b/tests/test_migration_0010.py new file mode 100644 index 0000000..a261e6c --- /dev/null +++ b/tests/test_migration_0010.py @@ -0,0 +1,35 @@ +import pytest +from sqlalchemy.exc import IntegrityError + +from backend.app.models import Artist, Source + +pytestmark = pytest.mark.integration + + +@pytest.mark.asyncio +async def test_duplicate_artist_platform_url_rejected(db): + artist = Artist(name="Alice", slug="alice") + db.add(artist) + await db.flush() + db.add(Source( + artist_id=artist.id, platform="patreon", + url="https://patreon.com/alice", enabled=True, + )) + await db.flush() + db.add(Source( + artist_id=artist.id, platform="patreon", + url="https://patreon.com/alice", enabled=True, + )) + with pytest.raises(IntegrityError): + await db.flush() + + +@pytest.mark.asyncio +async def test_same_url_under_different_artist_ok(db): + a = Artist(name="A", slug="a") + b = Artist(name="B", slug="b") + db.add_all([a, b]) + await db.flush() + db.add(Source(artist_id=a.id, platform="patreon", url="https://x/y", enabled=True)) + db.add(Source(artist_id=b.id, platform="patreon", url="https://x/y", enabled=True)) + await db.flush() # must NOT raise diff --git a/tests/test_migration_0011.py b/tests/test_migration_0011.py new file mode 100644 index 0000000..17e9702 --- /dev/null +++ b/tests/test_migration_0011.py @@ -0,0 +1,32 @@ +import pytest +from sqlalchemy import inspect, text + +pytestmark = pytest.mark.integration + + +@pytest.mark.asyncio +async def test_credential_has_credential_type_not_kind(db): + cols = (await db.run_sync( + lambda sync_session: [c["name"] for c in inspect(sync_session.bind).get_columns("credential")] + )) + assert "credential_type" in cols + assert "kind" not in cols + assert "status" not in cols + assert "last_verified" in cols + + +@pytest.mark.asyncio +async def test_credential_round_trip(db): + from backend.app.models import Credential + + db.add(Credential( + platform="patreon", + credential_type="cookies", + encrypted_blob=b"\x00\x01\x02", + )) + await db.flush() + row = (await db.execute( + text("SELECT credential_type, last_verified FROM credential WHERE platform='patreon'") + )).one() + assert row.credential_type == "cookies" + assert row.last_verified is None diff --git a/tests/test_migration_0012.py b/tests/test_migration_0012.py new file mode 100644 index 0000000..a993ab4 --- /dev/null +++ b/tests/test_migration_0012.py @@ -0,0 +1,32 @@ +import pytest +from sqlalchemy import select + +from backend.app.models import AppSetting + +pytestmark = pytest.mark.integration + + +@pytest.mark.asyncio +async def test_app_setting_table_round_trip(db): + db.add(AppSetting(key="extension_api_key", value="abc123")) + await db.flush() + row = (await db.execute( + select(AppSetting).where(AppSetting.key == "extension_api_key") + )).scalar_one() + assert row.value == "abc123" + assert row.updated_at is not None + + +@pytest.mark.asyncio +async def test_app_setting_upsert(db): + db.add(AppSetting(key="k", value="v1")) + await db.flush() + row = (await db.execute( + select(AppSetting).where(AppSetting.key == "k") + )).scalar_one() + row.value = "v2" + await db.flush() + again = (await db.execute( + select(AppSetting.value).where(AppSetting.key == "k") + )).scalar_one() + assert again == "v2" diff --git a/tests/test_migration_0013.py b/tests/test_migration_0013.py new file mode 100644 index 0000000..1a18384 --- /dev/null +++ b/tests/test_migration_0013.py @@ -0,0 +1,31 @@ +import pytest +from sqlalchemy import inspect, select + +from backend.app.models import ImportSettings + +pytestmark = pytest.mark.integration + + +@pytest.mark.asyncio +async def test_download_event_has_metadata(db): + cols = await db.run_sync( + lambda s: {c["name"]: c for c in inspect(s.bind).get_columns("download_event")} + ) + assert "metadata" in cols + assert cols["metadata"]["nullable"] is False + + +@pytest.mark.asyncio +async def test_import_settings_has_downloader_fields(db): + cols = await db.run_sync( + lambda s: {c["name"]: c for c in inspect(s.bind).get_columns("import_settings")} + ) + assert "download_rate_limit_seconds" in cols + assert "download_validate_files" in cols + + +@pytest.mark.asyncio +async def test_import_settings_defaults(db): + row = (await db.execute(select(ImportSettings).where(ImportSettings.id == 1))).scalar_one() + assert row.download_rate_limit_seconds == 3.0 + assert row.download_validate_files is True