diff --git a/alembic/versions/0001_initial_unified_schema.py b/alembic/versions/0001_initial_unified_schema.py deleted file mode 100644 index 0580b45..0000000 --- a/alembic/versions/0001_initial_unified_schema.py +++ /dev/null @@ -1,277 +0,0 @@ -"""initial unified schema - -Revision ID: 0001 -Revises: -Create Date: 2026-05-13 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from pgvector.sqlalchemy import Vector - -revision: str = "0001" -down_revision: Union[str, None] = None -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.execute("CREATE EXTENSION IF NOT EXISTS vector") - - op.create_table( - "artist", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("name", sa.String(length=255), nullable=False), - sa.Column("slug", sa.String(length=255), nullable=False), - sa.Column("notes", sa.Text(), nullable=True), - sa.Column("is_subscription", sa.Boolean(), nullable=False, server_default=sa.false()), - sa.Column("auto_check", sa.Boolean(), nullable=False, server_default=sa.true()), - sa.Column("check_interval_seconds", sa.Integer(), nullable=True), - sa.Column( - "created_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.PrimaryKeyConstraint("id", name="pk_artist"), - sa.UniqueConstraint("name", name="uq_artist_name"), - sa.UniqueConstraint("slug", name="uq_artist_slug"), - ) - - op.create_table( - "source", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("artist_id", sa.Integer(), nullable=False), - sa.Column("platform", sa.String(length=64), nullable=False), - sa.Column("url", sa.Text(), nullable=False), - sa.Column("enabled", sa.Boolean(), nullable=False, server_default=sa.true()), - sa.Column("config_overrides", sa.JSON(), nullable=True), - sa.Column("last_checked_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("last_error", sa.Text(), nullable=True), - sa.Column("check_interval_override", sa.Integer(), nullable=True), - sa.ForeignKeyConstraint( - ["artist_id"], ["artist.id"], name="fk_source_artist_id_artist", ondelete="CASCADE" - ), - sa.PrimaryKeyConstraint("id", name="pk_source"), - ) - op.create_index("ix_source_artist_id", "source", ["artist_id"]) - - op.create_table( - "credential", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("platform", sa.String(length=64), nullable=False), - sa.Column("kind", sa.String(length=32), nullable=False), - sa.Column("encrypted_blob", sa.LargeBinary(), nullable=False), - sa.Column("status", sa.String(length=32), nullable=False, server_default="active"), - sa.Column( - "captured_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.Column("expires_at", sa.DateTime(timezone=True), nullable=True), - sa.PrimaryKeyConstraint("id", name="pk_credential"), - sa.UniqueConstraint("platform", name="uq_credential_platform"), - ) - - op.create_table( - "post", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("source_id", sa.Integer(), nullable=False), - sa.Column("external_post_id", sa.String(length=128), nullable=False), - sa.Column("post_url", sa.Text(), nullable=True), - sa.Column("post_title", sa.Text(), nullable=True), - sa.Column("post_date", sa.DateTime(timezone=True), nullable=True), - sa.Column("raw_metadata", sa.JSON(), nullable=True), - sa.Column( - "downloaded_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["source_id"], ["source.id"], name="fk_post_source_id_source", ondelete="CASCADE" - ), - sa.PrimaryKeyConstraint("id", name="pk_post"), - sa.UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"), - ) - op.create_index("ix_post_source_id", "post", ["source_id"]) - - op.create_table( - "image_record", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("path", sa.Text(), nullable=False), - sa.Column("sha256", sa.String(length=64), nullable=False), - sa.Column("phash", sa.String(length=32), nullable=True), - sa.Column("size_bytes", sa.BigInteger(), nullable=False), - sa.Column("mime", sa.String(length=64), nullable=False), - sa.Column("width", sa.Integer(), nullable=True), - sa.Column("height", sa.Integer(), nullable=True), - sa.Column("thumbnail_path", sa.Text(), nullable=True), - sa.Column( - "origin", - sa.Enum( - "downloaded", - "imported_filesystem", - "uploaded", - name="origin_enum", - ), - nullable=False, - ), - sa.Column("primary_post_id", sa.Integer(), nullable=True), - sa.Column("wd14_predictions", sa.JSON(), nullable=True), - sa.Column("wd14_model_version", sa.String(length=128), nullable=True), - sa.Column("siglip_embedding", Vector(1152), nullable=True), - sa.Column("siglip_model_version", sa.String(length=128), nullable=True), - sa.Column("centroid_scores", sa.JSON(), nullable=True), - sa.Column( - "created_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.Column( - "updated_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["primary_post_id"], - ["post.id"], - name="fk_image_record_primary_post_id_post", - ondelete="SET NULL", - ), - sa.PrimaryKeyConstraint("id", name="pk_image_record"), - sa.UniqueConstraint("path", name="uq_image_record_path"), - sa.UniqueConstraint("sha256", name="uq_image_record_sha256"), - ) - op.create_index("ix_image_record_sha256", "image_record", ["sha256"]) - op.create_index("ix_image_record_phash", "image_record", ["phash"]) - op.create_index("ix_image_record_primary_post_id", "image_record", ["primary_post_id"]) - - op.create_table( - "image_provenance", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("image_record_id", sa.Integer(), nullable=False), - sa.Column("post_id", sa.Integer(), nullable=False), - sa.Column("source_id", sa.Integer(), nullable=False), - sa.Column("captured_metadata", sa.JSON(), nullable=True), - sa.Column( - "captured_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["image_record_id"], - ["image_record.id"], - name="fk_image_provenance_image_record_id_image_record", - ondelete="CASCADE", - ), - sa.ForeignKeyConstraint( - ["post_id"], - ["post.id"], - name="fk_image_provenance_post_id_post", - ondelete="CASCADE", - ), - sa.ForeignKeyConstraint( - ["source_id"], - ["source.id"], - name="fk_image_provenance_source_id_source", - ondelete="CASCADE", - ), - sa.PrimaryKeyConstraint("id", name="pk_image_provenance"), - ) - op.create_index("ix_image_provenance_image_record_id", "image_provenance", ["image_record_id"]) - op.create_index("ix_image_provenance_post_id", "image_provenance", ["post_id"]) - op.create_index("ix_image_provenance_source_id", "image_provenance", ["source_id"]) - - op.create_table( - "tag", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("name", sa.String(length=255), nullable=False), - sa.Column("namespace", sa.String(length=64), nullable=True), - sa.Column( - "created_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.PrimaryKeyConstraint("id", name="pk_tag"), - sa.UniqueConstraint("name", name="uq_tag_name"), - ) - op.create_index("ix_tag_name", "tag", ["name"]) - op.create_index("ix_tag_namespace", "tag", ["namespace"]) - - op.create_table( - "image_tag", - sa.Column("image_record_id", sa.Integer(), nullable=False), - sa.Column("tag_id", sa.Integer(), nullable=False), - sa.Column("source", sa.String(length=32), nullable=False, server_default="manual"), - sa.Column( - "created_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["image_record_id"], - ["image_record.id"], - name="fk_image_tag_image_record_id_image_record", - ondelete="CASCADE", - ), - sa.ForeignKeyConstraint( - ["tag_id"], ["tag.id"], name="fk_image_tag_tag_id_tag", ondelete="CASCADE" - ), - sa.PrimaryKeyConstraint("image_record_id", "tag_id", name="pk_image_tag"), - ) - - op.create_table( - "download_event", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("source_id", sa.Integer(), nullable=False), - sa.Column("post_id", sa.Integer(), nullable=True), - sa.Column("status", sa.String(length=32), nullable=False), - sa.Column( - "started_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("bytes_downloaded", sa.BigInteger(), nullable=False, server_default="0"), - sa.Column("files_count", sa.Integer(), nullable=False, server_default="0"), - sa.Column("error", sa.Text(), nullable=True), - sa.ForeignKeyConstraint( - ["source_id"], - ["source.id"], - name="fk_download_event_source_id_source", - ondelete="CASCADE", - ), - sa.ForeignKeyConstraint( - ["post_id"], - ["post.id"], - name="fk_download_event_post_id_post", - ondelete="SET NULL", - ), - sa.PrimaryKeyConstraint("id", name="pk_download_event"), - ) - op.create_index("ix_download_event_source_id", "download_event", ["source_id"]) - op.create_index("ix_download_event_post_id", "download_event", ["post_id"]) - - -def downgrade() -> None: - op.drop_table("download_event") - op.drop_table("image_tag") - op.drop_table("tag") - op.drop_table("image_provenance") - op.drop_table("image_record") - op.execute("DROP TYPE IF EXISTS origin_enum") - op.drop_table("post") - op.drop_table("credential") - op.drop_table("source") - op.drop_table("artist") - op.execute("DROP EXTENSION IF EXISTS vector") diff --git a/alembic/versions/0002_fc2a_tag_kinds_and_import_tasks.py b/alembic/versions/0002_fc2a_tag_kinds_and_import_tasks.py deleted file mode 100644 index b9dac2c..0000000 --- a/alembic/versions/0002_fc2a_tag_kinds_and_import_tasks.py +++ /dev/null @@ -1,208 +0,0 @@ -"""fc2a: tag kinds, import_task, import_batch, integrity_status - -Revision ID: 0002 -Revises: 0001 -Create Date: 2026-05-14 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0002" -down_revision: Union[str, None] = "0001" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - -TAG_KINDS = ( - "artist", - "character", - "fandom", - "general", - "series", - "archive", - "post", - "meta", - "rating", -) - - -def upgrade() -> None: - # --- Tag kind enum + fandom_id --- - tag_kind = sa.Enum(*TAG_KINDS, name="tag_kind") - tag_kind.create(op.get_bind(), checkfirst=True) - - op.add_column( - "tag", - sa.Column("kind", tag_kind, nullable=False, server_default="general"), - ) - op.add_column( - "tag", - sa.Column("fandom_id", sa.Integer(), nullable=True), - ) - op.create_foreign_key( - "fk_tag_fandom_id_tag", - "tag", - "tag", - ["fandom_id"], - ["id"], - ondelete="SET NULL", - ) - - # Drop the old global uniqueness on name; add kind+fandom-aware uniqueness. - op.drop_constraint("uq_tag_name", "tag", type_="unique") - op.drop_index("ix_tag_name", table_name="tag") - op.execute( - """ - CREATE UNIQUE INDEX uq_tag_name_kind_fandom - ON tag (name, kind, COALESCE(fandom_id, 0)) - """ - ) - - # CHECK: fandom_id is only allowed for character kind. - op.create_check_constraint( - "ck_tag_fandom_requires_character", - "tag", - "(fandom_id IS NULL) OR (kind = 'character')", - ) - - # Drop the old namespace column — superseded by kind. - op.drop_index("ix_tag_namespace", table_name="tag") - op.drop_column("tag", "namespace") - - # --- ImportBatch --- - op.create_table( - "import_batch", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("triggered_by", sa.String(length=32), nullable=False), - sa.Column("source_path", sa.Text(), nullable=False), - sa.Column("scan_mode", sa.String(length=16), nullable=False), - sa.Column( - "started_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("total_files", sa.Integer(), nullable=False, server_default="0"), - sa.Column("imported", sa.Integer(), nullable=False, server_default="0"), - sa.Column("skipped", sa.Integer(), nullable=False, server_default="0"), - sa.Column("failed", sa.Integer(), nullable=False, server_default="0"), - sa.Column("status", sa.String(length=16), nullable=False, server_default="running"), - sa.PrimaryKeyConstraint("id", name="pk_import_batch"), - ) - op.create_index("ix_import_batch_status", "import_batch", ["status"]) - - # --- ImportTask --- - op.create_table( - "import_task", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("batch_id", sa.Integer(), nullable=False), - sa.Column("source_path", sa.Text(), nullable=False), - sa.Column("task_type", sa.String(length=16), nullable=False), - sa.Column("status", sa.String(length=16), nullable=False, server_default="pending"), - sa.Column("result_image_id", sa.Integer(), nullable=True), - sa.Column("error", sa.Text(), nullable=True), - sa.Column("size_bytes", sa.BigInteger(), nullable=True), - sa.Column( - "created_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.func.now(), - ), - sa.Column("started_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.ForeignKeyConstraint( - ["batch_id"], - ["import_batch.id"], - name="fk_import_task_batch_id_import_batch", - ondelete="CASCADE", - ), - sa.ForeignKeyConstraint( - ["result_image_id"], - ["image_record.id"], - name="fk_import_task_result_image_id_image_record", - ondelete="SET NULL", - ), - sa.PrimaryKeyConstraint("id", name="pk_import_task"), - ) - op.create_index("ix_import_task_batch_id", "import_task", ["batch_id"]) - op.create_index("ix_import_task_status", "import_task", ["status"]) - op.create_index( - "ix_import_task_created_at_desc", - "import_task", - [sa.text("created_at DESC")], - ) - - # --- ImportSettings (single-row table) --- - op.create_table( - "import_settings", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("import_scan_path", sa.Text(), nullable=False, server_default="/import"), - sa.Column("min_width", sa.Integer(), nullable=False, server_default="0"), - sa.Column("min_height", sa.Integer(), nullable=False, server_default="0"), - sa.Column( - "skip_transparent", sa.Boolean(), nullable=False, server_default=sa.false() - ), - sa.Column( - "transparency_threshold", - sa.Float(), - nullable=False, - server_default="0.9", - ), - sa.Column( - "skip_single_color", sa.Boolean(), nullable=False, server_default=sa.false() - ), - sa.Column( - "single_color_threshold", - sa.Float(), - nullable=False, - server_default="0.95", - ), - sa.Column("single_color_tolerance", sa.Integer(), nullable=False, server_default="30"), - sa.PrimaryKeyConstraint("id", name="pk_import_settings"), - sa.CheckConstraint("id = 1", name="ck_import_settings_singleton"), - ) - # Seed the single row immediately so callers can always SELECT id=1. - op.execute("INSERT INTO import_settings (id) VALUES (1)") - - # --- ImageRecord additions --- - op.add_column( - "image_record", - sa.Column( - "integrity_status", - sa.String(length=24), - nullable=False, - server_default="unknown", - ), - ) - op.create_index( - "ix_image_record_integrity_status", - "image_record", - ["integrity_status"], - ) - - -def downgrade() -> None: - op.drop_index("ix_image_record_integrity_status", table_name="image_record") - op.drop_column("image_record", "integrity_status") - - op.drop_table("import_settings") - op.drop_index("ix_import_task_created_at_desc", table_name="import_task") - op.drop_index("ix_import_task_status", table_name="import_task") - op.drop_index("ix_import_task_batch_id", table_name="import_task") - op.drop_table("import_task") - op.drop_index("ix_import_batch_status", table_name="import_batch") - op.drop_table("import_batch") - - op.drop_constraint("ck_tag_fandom_requires_character", "tag", type_="check") - op.execute("DROP INDEX uq_tag_name_kind_fandom") - op.add_column("tag", sa.Column("namespace", sa.String(length=64), nullable=True)) - op.create_index("ix_tag_namespace", "tag", ["namespace"]) - op.create_index("ix_tag_name", "tag", ["name"], unique=False) - op.create_unique_constraint("uq_tag_name", "tag", ["name"]) - op.drop_constraint("fk_tag_fandom_id_tag", "tag", type_="foreignkey") - op.drop_column("tag", "fandom_id") - op.drop_column("tag", "kind") - sa.Enum(name="tag_kind").drop(op.get_bind(), checkfirst=True) diff --git a/alembic/versions/0003_fc2b_ml_pipeline.py b/alembic/versions/0003_fc2b_ml_pipeline.py deleted file mode 100644 index 584bffe..0000000 --- a/alembic/versions/0003_fc2b_ml_pipeline.py +++ /dev/null @@ -1,172 +0,0 @@ -"""fc2b: ML pipeline — allowlist, aliases, centroids, ml_settings - -Revision ID: 0003 -Revises: 0002 -Create Date: 2026-05-15 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from pgvector.sqlalchemy import Vector - -revision: str = "0003" -down_revision: Union[str, None] = "0002" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # 3.1 rename wd14_* -> tagger_* - op.alter_column("image_record", "wd14_predictions", new_column_name="tagger_predictions") - op.alter_column( - "image_record", "wd14_model_version", new_column_name="tagger_model_version" - ) - - # 3.2 tag_allowlist - op.create_table( - "tag_allowlist", - sa.Column("tag_id", sa.Integer(), nullable=False), - sa.Column( - "min_confidence", sa.Float(), nullable=False, server_default="0.95" - ), - sa.Column( - "added_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["tag_id"], ["tag.id"], name="fk_tag_allowlist_tag_id_tag", - ondelete="CASCADE", - ), - sa.PrimaryKeyConstraint("tag_id", name="pk_tag_allowlist"), - sa.CheckConstraint( - "min_confidence > 0 AND min_confidence <= 1", - name="ck_tag_allowlist_confidence_range", - ), - ) - - # 3.3 tag_suggestion_rejection - op.create_table( - "tag_suggestion_rejection", - sa.Column("image_record_id", sa.Integer(), nullable=False), - sa.Column("tag_id", sa.Integer(), nullable=False), - sa.Column( - "rejected_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["image_record_id"], ["image_record.id"], - name="fk_tsr_image_record_id_image_record", ondelete="CASCADE", - ), - sa.ForeignKeyConstraint( - ["tag_id"], ["tag.id"], name="fk_tsr_tag_id_tag", ondelete="CASCADE", - ), - sa.PrimaryKeyConstraint( - "image_record_id", "tag_id", name="pk_tag_suggestion_rejection" - ), - ) - op.create_index( - "ix_tag_suggestion_rejection_tag", "tag_suggestion_rejection", ["tag_id"] - ) - - # 3.4 tag_alias - op.create_table( - "tag_alias", - sa.Column("alias_string", sa.String(length=255), nullable=False), - sa.Column("alias_category", sa.String(length=32), nullable=False), - sa.Column("canonical_tag_id", sa.Integer(), nullable=False), - sa.Column( - "created_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["canonical_tag_id"], ["tag.id"], - name="fk_tag_alias_canonical_tag_id_tag", ondelete="CASCADE", - ), - sa.PrimaryKeyConstraint( - "alias_string", "alias_category", name="pk_tag_alias" - ), - ) - op.create_index("ix_tag_alias_canonical", "tag_alias", ["canonical_tag_id"]) - - # 3.5 tag_reference_embedding (centroids) - op.create_table( - "tag_reference_embedding", - sa.Column("tag_id", sa.Integer(), nullable=False), - sa.Column("embedding", Vector(1152), nullable=False), - sa.Column("reference_count", sa.Integer(), nullable=False), - sa.Column("model_version", sa.String(length=128), nullable=False), - sa.Column( - "updated_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["tag_id"], ["tag.id"], - name="fk_tag_reference_embedding_tag_id_tag", ondelete="CASCADE", - ), - sa.PrimaryKeyConstraint("tag_id", name="pk_tag_reference_embedding"), - ) - - # 3.6 ml_settings singleton - op.create_table( - "ml_settings", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column( - "suggestion_threshold_artist", sa.Float(), nullable=False, - server_default="0.30", - ), - sa.Column( - "suggestion_threshold_character", sa.Float(), nullable=False, - server_default="0.50", - ), - sa.Column( - "suggestion_threshold_copyright", sa.Float(), nullable=False, - server_default="0.50", - ), - sa.Column( - "suggestion_threshold_general", sa.Float(), nullable=False, - server_default="0.95", - ), - sa.Column( - "centroid_similarity_threshold", sa.Float(), nullable=False, - server_default="0.55", - ), - sa.Column( - "min_reference_images", sa.Integer(), nullable=False, - server_default="5", - ), - sa.Column( - "tagger_model_version", sa.String(length=128), nullable=False, - server_default="camie-tagger-v2", - ), - sa.Column( - "embedder_model_version", sa.String(length=128), nullable=False, - server_default="siglip-so400m-patch14-384", - ), - sa.Column( - "updated_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.PrimaryKeyConstraint("id", name="pk_ml_settings"), - sa.CheckConstraint("id = 1", name="ck_ml_settings_singleton"), - ) - op.execute("INSERT INTO ml_settings (id) VALUES (1)") - - -def downgrade() -> None: - op.drop_table("ml_settings") - op.drop_table("tag_reference_embedding") - op.drop_index("ix_tag_alias_canonical", table_name="tag_alias") - op.drop_table("tag_alias") - op.drop_index( - "ix_tag_suggestion_rejection_tag", table_name="tag_suggestion_rejection" - ) - op.drop_table("tag_suggestion_rejection") - op.drop_table("tag_allowlist") - op.alter_column( - "image_record", "tagger_model_version", new_column_name="wd14_model_version" - ) - op.alter_column( - "image_record", "tagger_predictions", new_column_name="wd14_predictions" - ) diff --git a/alembic/versions/0004_fc2c_i_tsm_system_rows.py b/alembic/versions/0004_fc2c_i_tsm_system_rows.py deleted file mode 100644 index e8bd920..0000000 --- a/alembic/versions/0004_fc2c_i_tsm_system_rows.py +++ /dev/null @@ -1,23 +0,0 @@ -"""fc2c-i: enable tsm_system_rows for scalable random sampling - -Revision ID: 0004 -Revises: 0003 -Create Date: 2026-05-15 - -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0004" -down_revision: Union[str, None] = "0003" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows") - - -def downgrade() -> None: - op.execute("DROP EXTENSION IF EXISTS tsm_system_rows") diff --git a/alembic/versions/0005_fc2c_iii_a_series_page.py b/alembic/versions/0005_fc2c_iii_a_series_page.py deleted file mode 100644 index ffe397e..0000000 --- a/alembic/versions/0005_fc2c_iii_a_series_page.py +++ /dev/null @@ -1,50 +0,0 @@ -"""fc2c-iii-a: series_page ordered membership - -Revision ID: 0005 -Revises: 0004 -Create Date: 2026-05-16 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0005" -down_revision: Union[str, None] = "0004" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "series_page", - sa.Column("id", sa.Integer(), nullable=False), - sa.Column("series_tag_id", sa.Integer(), nullable=False), - sa.Column("image_id", sa.Integer(), nullable=False), - sa.Column("page_number", sa.Integer(), nullable=False), - sa.Column( - "created_at", sa.DateTime(timezone=True), - nullable=False, server_default=sa.func.now(), - ), - sa.Column( - "updated_at", sa.DateTime(timezone=True), - nullable=False, server_default=sa.func.now(), - ), - sa.ForeignKeyConstraint( - ["series_tag_id"], ["tag.id"], ondelete="CASCADE" - ), - sa.ForeignKeyConstraint( - ["image_id"], ["image_record.id"], ondelete="CASCADE" - ), - sa.PrimaryKeyConstraint("id"), - sa.UniqueConstraint("image_id", name="uq_series_page_image"), - ) - op.create_index( - "ix_series_page_series_tag_id", "series_page", ["series_tag_id"] - ) - - -def downgrade() -> None: - op.drop_index("ix_series_page_series_tag_id", table_name="series_page") - op.drop_table("series_page") diff --git a/alembic/versions/0006_fc2d_phash_threshold.py b/alembic/versions/0006_fc2d_phash_threshold.py deleted file mode 100644 index 895ed46..0000000 --- a/alembic/versions/0006_fc2d_phash_threshold.py +++ /dev/null @@ -1,30 +0,0 @@ -"""fc2d: import_settings.phash_threshold - -Revision ID: 0006 -Revises: 0005 -Create Date: 2026-05-17 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0006" -down_revision: Union[str, None] = "0005" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "import_settings", - sa.Column( - "phash_threshold", sa.Integer(), - nullable=False, server_default="10", - ), - ) - - -def downgrade() -> None: - op.drop_column("import_settings", "phash_threshold") diff --git a/alembic/versions/0007_fc2d_post_metadata_fields.py b/alembic/versions/0007_fc2d_post_metadata_fields.py deleted file mode 100644 index 24e8ba9..0000000 --- a/alembic/versions/0007_fc2d_post_metadata_fields.py +++ /dev/null @@ -1,31 +0,0 @@ -"""fc2d-iv: post.description + post.attachment_count - -Revision ID: 0007 -Revises: 0006 -Create Date: 2026-05-18 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0007" -down_revision: Union[str, None] = "0006" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "post", sa.Column("description", sa.Text(), nullable=True) - ) - op.add_column( - "post", - sa.Column("attachment_count", sa.Integer(), nullable=True), - ) - - -def downgrade() -> None: - op.drop_column("post", "attachment_count") - op.drop_column("post", "description") diff --git a/alembic/versions/0008_fc2d_vii_c_artist_deconfliction.py b/alembic/versions/0008_fc2d_vii_c_artist_deconfliction.py deleted file mode 100644 index 4019e2d..0000000 --- a/alembic/versions/0008_fc2d_vii_c_artist_deconfliction.py +++ /dev/null @@ -1,52 +0,0 @@ -"""fc2d-vii-c: image_record.artist_id + backfill + drop artist tags - -Revision ID: 0008 -Revises: 0007 -Create Date: 2026-05-18 - -Internal forward-correctness migration (the big legacy-import migration -stays deferred). downgrade() does NOT recreate deleted artist tags; -downgrade is dev-only and the data is reconstructable by re-import. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -from backend.app.utils.artist_backfill import ( - BACKFILL_PRIMARY_SQL, - BACKFILL_PROVENANCE_SQL, - BACKFILL_TAG_SQL, - DELETE_ARTIST_TAGS_SQL, -) - -revision: str = "0008" -down_revision: Union[str, None] = "0007" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "image_record", - sa.Column("artist_id", sa.Integer(), nullable=True), - ) - op.create_foreign_key( - "fk_image_record_artist_id", "image_record", "artist", - ["artist_id"], ["id"], ondelete="SET NULL", - ) - op.create_index( - "ix_image_record_artist_id", "image_record", ["artist_id"], - ) - op.execute(BACKFILL_PRIMARY_SQL) - op.execute(BACKFILL_PROVENANCE_SQL) - op.execute(BACKFILL_TAG_SQL) - op.execute(DELETE_ARTIST_TAGS_SQL) - - -def downgrade() -> None: - op.drop_index("ix_image_record_artist_id", table_name="image_record") - op.drop_constraint( - "fk_image_record_artist_id", "image_record", type_="foreignkey" - ) - op.drop_column("image_record", "artist_id") diff --git a/alembic/versions/0009_fc2d_iii_post_attachment.py b/alembic/versions/0009_fc2d_iii_post_attachment.py deleted file mode 100644 index 4820fa4..0000000 --- a/alembic/versions/0009_fc2d_iii_post_attachment.py +++ /dev/null @@ -1,68 +0,0 @@ -"""fc2d-iii: post_attachment + import_batch.attachments - -Revision ID: 0009 -Revises: 0008 -Create Date: 2026-05-19 - -Internal forward-correctness migration (big legacy-import migration -stays deferred). No backfill — no attachments exist yet. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0009" -down_revision: Union[str, None] = "0008" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "post_attachment", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column( - "post_id", sa.Integer(), - sa.ForeignKey("post.id", ondelete="SET NULL"), nullable=True, - ), - sa.Column( - "artist_id", sa.Integer(), - sa.ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, - ), - sa.Column("sha256", sa.String(64), nullable=False), - sa.Column("path", sa.Text(), nullable=False), - sa.Column("original_filename", sa.Text(), nullable=False), - sa.Column("ext", sa.String(32), nullable=False), - sa.Column("mime", sa.String(128), nullable=True), - sa.Column("size_bytes", sa.BigInteger(), nullable=False), - sa.Column( - "captured_at", sa.DateTime(timezone=True), - server_default=sa.func.now(), nullable=False, - ), - ) - op.create_index( - "ix_post_attachment_sha256", "post_attachment", ["sha256"], - unique=True, - ) - op.create_index( - "ix_post_attachment_post_id", "post_attachment", ["post_id"], - ) - op.create_index( - "ix_post_attachment_artist_id", "post_attachment", ["artist_id"], - ) - op.add_column( - "import_batch", - sa.Column( - "attachments", sa.Integer(), nullable=False, - server_default="0", - ), - ) - - -def downgrade() -> None: - op.drop_column("import_batch", "attachments") - op.drop_index("ix_post_attachment_artist_id", table_name="post_attachment") - op.drop_index("ix_post_attachment_post_id", table_name="post_attachment") - op.drop_index("ix_post_attachment_sha256", table_name="post_attachment") - op.drop_table("post_attachment") diff --git a/alembic/versions/0010_fc3a_source_unique_artist_platform_url.py b/alembic/versions/0010_fc3a_source_unique_artist_platform_url.py deleted file mode 100644 index 10f502b..0000000 --- a/alembic/versions/0010_fc3a_source_unique_artist_platform_url.py +++ /dev/null @@ -1,32 +0,0 @@ -"""fc3a: unique(source.artist_id, source.platform, source.url) - -Revision ID: 0010 -Revises: 0009 -Create Date: 2026-05-20 - -Enforces FC-3a's dedup invariant at the DB level. No backfill — no -existing rows are expected to collide; if they do the migration will -fail loudly (intended). -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0010" -down_revision: Union[str, None] = "0009" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_unique_constraint( - "uq_source_artist_platform_url", - "source", - ["artist_id", "platform", "url"], - ) - - -def downgrade() -> None: - op.drop_constraint( - "uq_source_artist_platform_url", "source", type_="unique" - ) diff --git a/alembic/versions/0011_fc3b_credential_schema_alignment.py b/alembic/versions/0011_fc3b_credential_schema_alignment.py deleted file mode 100644 index 00a31a5..0000000 --- a/alembic/versions/0011_fc3b_credential_schema_alignment.py +++ /dev/null @@ -1,41 +0,0 @@ -"""fc3b: rename credential.kind -> credential_type, drop status, add last_verified - -Revision ID: 0011 -Revises: 0010 -Create Date: 2026-05-20 - -Aligns the credential table with the GallerySubscriber wire-field names -so the existing browser extension can POST to FC unmodified. Greenfield — -no rows exist in production yet, so no data preservation logic is -needed; the rename uses ALTER COLUMN rather than copy-then-drop. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0011" -down_revision: Union[str, None] = "0010" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.alter_column("credential", "kind", new_column_name="credential_type") - op.drop_column("credential", "status") - op.add_column( - "credential", - sa.Column("last_verified", sa.DateTime(timezone=True), nullable=True), - ) - - -def downgrade() -> None: - op.drop_column("credential", "last_verified") - op.add_column( - "credential", - sa.Column( - "status", sa.String(length=32), nullable=False, - server_default="active", - ), - ) - op.alter_column("credential", "credential_type", new_column_name="kind") diff --git a/alembic/versions/0012_fc3b_app_setting.py b/alembic/versions/0012_fc3b_app_setting.py deleted file mode 100644 index 42c06eb..0000000 --- a/alembic/versions/0012_fc3b_app_setting.py +++ /dev/null @@ -1,36 +0,0 @@ -"""fc3b: app_setting key/value table - -Revision ID: 0012 -Revises: 0011 -Create Date: 2026-05-20 - -A simple key/value table for small app settings that don't fit -ImportSettings. Initially seeds only `extension_api_key` (done in -create_app on first boot — not in the migration, to keep it -deterministic and independent of randomness). -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0012" -down_revision: Union[str, None] = "0011" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "app_setting", - sa.Column("key", sa.String(length=64), primary_key=True), - sa.Column("value", sa.Text(), nullable=False), - sa.Column( - "updated_at", sa.DateTime(timezone=True), - nullable=False, server_default=sa.func.now(), - ), - ) - - -def downgrade() -> None: - op.drop_table("app_setting") diff --git a/alembic/versions/0013_fc3c_download_event_metadata.py b/alembic/versions/0013_fc3c_download_event_metadata.py deleted file mode 100644 index c88b3f9..0000000 --- a/alembic/versions/0013_fc3c_download_event_metadata.py +++ /dev/null @@ -1,52 +0,0 @@ -"""fc3c: download_event.metadata + import_settings downloader fields - -Revision ID: 0013 -Revises: 0012 -Create Date: 2026-05-20 - -Additive only. download_event.metadata is the rich JSONB blob FC-3c -populates per run (run_stats, stdout/stderr, quarantined paths, import -summary). import_settings gains two operator-tunable downloader knobs: -download_rate_limit_seconds (gallery-dl extractor.sleep) and -download_validate_files (toggle the magic-byte validator). -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from sqlalchemy.dialects import postgresql - -revision: str = "0013" -down_revision: Union[str, None] = "0012" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "download_event", - sa.Column( - "metadata", postgresql.JSONB, - nullable=False, server_default=sa.text("'{}'::jsonb"), - ), - ) - op.add_column( - "import_settings", - sa.Column( - "download_rate_limit_seconds", sa.Float(), - nullable=False, server_default="3.0", - ), - ) - op.add_column( - "import_settings", - sa.Column( - "download_validate_files", sa.Boolean(), - nullable=False, server_default=sa.true(), - ), - ) - - -def downgrade() -> None: - op.drop_column("import_settings", "download_validate_files") - op.drop_column("import_settings", "download_rate_limit_seconds") - op.drop_column("download_event", "metadata") diff --git a/alembic/versions/0014_fc3d_scheduling.py b/alembic/versions/0014_fc3d_scheduling.py deleted file mode 100644 index 955e956..0000000 --- a/alembic/versions/0014_fc3d_scheduling.py +++ /dev/null @@ -1,58 +0,0 @@ -"""fc3d: scheduling + source health columns - -Revision ID: 0014 -Revises: 0013 -Create Date: 2026-05-21 - -Additive only. source.consecutive_failures (default 0, DownloadService -finalize hook owns the writes). import_settings gains the three -scheduling knobs (global default interval, event retention, failure -warning threshold). -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0014" -down_revision: Union[str, None] = "0013" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "source", - sa.Column( - "consecutive_failures", sa.Integer(), - nullable=False, server_default="0", - ), - ) - op.add_column( - "import_settings", - sa.Column( - "download_schedule_default_seconds", sa.Integer(), - nullable=False, server_default="28800", - ), - ) - op.add_column( - "import_settings", - sa.Column( - "download_event_retention_days", sa.Integer(), - nullable=False, server_default="90", - ), - ) - op.add_column( - "import_settings", - sa.Column( - "download_failure_warning_threshold", sa.Integer(), - nullable=False, server_default="5", - ), - ) - - -def downgrade() -> None: - op.drop_column("import_settings", "download_failure_warning_threshold") - op.drop_column("import_settings", "download_event_retention_days") - op.drop_column("import_settings", "download_schedule_default_seconds") - op.drop_column("source", "consecutive_failures") diff --git a/alembic/versions/0015_fc5_migration_run.py b/alembic/versions/0015_fc5_migration_run.py deleted file mode 100644 index 89d7d89..0000000 --- a/alembic/versions/0015_fc5_migration_run.py +++ /dev/null @@ -1,51 +0,0 @@ -"""fc5: migration_run table - -Revision ID: 0015 -Revises: 0014 -Create Date: 2026-05-22 - -Additive only. New table tracks each invocation of the FC-5 migration -tooling (backup, gs, ir, ml_queue, verify, rollback). kind/status are -plain String(32) — values validated at the API layer per the spec, not -a Postgres ENUM (so adding kinds later doesn't need a schema migration). -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from sqlalchemy.dialects import postgresql - -revision: str = "0015" -down_revision: Union[str, None] = "0014" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "migration_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("kind", sa.String(32), nullable=False, index=True), - sa.Column("status", sa.String(32), nullable=False, index=True), - sa.Column( - "dry_run", sa.Boolean(), nullable=False, server_default=sa.false(), - ), - sa.Column( - "started_at", sa.DateTime(timezone=True), - nullable=False, server_default=sa.func.now(), - ), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column( - "counts", postgresql.JSONB, - nullable=False, server_default=sa.text("'{}'::jsonb"), - ), - sa.Column("error", sa.Text(), nullable=True), - sa.Column( - "metadata", postgresql.JSONB, - nullable=False, server_default=sa.text("'{}'::jsonb"), - ), - ) - - -def downgrade() -> None: - op.drop_table("migration_run") diff --git a/alembic/versions/0016_fc3i_task_run.py b/alembic/versions/0016_fc3i_task_run.py deleted file mode 100644 index 678b1ee..0000000 --- a/alembic/versions/0016_fc3i_task_run.py +++ /dev/null @@ -1,86 +0,0 @@ -"""fc3i: task_run table - -Revision ID: 0016 -Revises: 0015 -Create Date: 2026-05-24 - -Additive only. New table records every Celery task attempt via signal -handlers (backend.app.celery_signals). Status is plain String(16) not -Postgres ENUM (per feedback_check_existing_enums: ENUM columns hard- -fail at INSERT, String columns extend cleanly). - -Composite indexes anticipate the three dashboard panes: -- (queue, started_at desc) — per-lane recent activity -- (status, started_at desc) — recent failures pane -- (task_name, started_at desc) — drill-down by task - -Indexed columns get individual indexes via `index=True` on the model; -the composites below cover the multi-column lookups. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0016" -down_revision: Union[str, None] = "0015" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "task_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("celery_task_id", sa.String(length=64), nullable=False), - sa.Column("queue", sa.String(length=32), nullable=False), - sa.Column("task_name", sa.String(length=128), nullable=False), - sa.Column("target_id", sa.Integer(), nullable=True), - sa.Column("started_at", sa.DateTime(timezone=True), nullable=False), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("duration_ms", sa.Integer(), nullable=True), - sa.Column( - "status", sa.String(length=16), nullable=False, - server_default="running", - ), - sa.Column("error_type", sa.String(length=128), nullable=True), - sa.Column("error_message", sa.Text(), nullable=True), - sa.Column("retry_count", sa.Integer(), nullable=True), - sa.Column("worker_hostname", sa.String(length=128), nullable=True), - sa.Column("args_summary", sa.String(length=255), nullable=True), - ) - - # Single-column indexes (matches Mapped[...].index=True on model). - op.create_index("ix_task_run_celery_task_id", "task_run", ["celery_task_id"]) - op.create_index("ix_task_run_queue", "task_run", ["queue"]) - op.create_index("ix_task_run_task_name", "task_run", ["task_name"]) - op.create_index("ix_task_run_started_at", "task_run", ["started_at"]) - op.create_index("ix_task_run_finished_at", "task_run", ["finished_at"]) - op.create_index("ix_task_run_status", "task_run", ["status"]) - - # Composite indexes for dashboard query patterns. - op.create_index( - "ix_task_run_queue_started", - "task_run", ["queue", sa.text("started_at DESC")], - ) - op.create_index( - "ix_task_run_status_started", - "task_run", ["status", sa.text("started_at DESC")], - ) - op.create_index( - "ix_task_run_name_started", - "task_run", ["task_name", sa.text("started_at DESC")], - ) - - -def downgrade() -> None: - op.drop_index("ix_task_run_name_started", table_name="task_run") - op.drop_index("ix_task_run_status_started", table_name="task_run") - op.drop_index("ix_task_run_queue_started", table_name="task_run") - op.drop_index("ix_task_run_status", table_name="task_run") - op.drop_index("ix_task_run_finished_at", table_name="task_run") - op.drop_index("ix_task_run_started_at", table_name="task_run") - op.drop_index("ix_task_run_task_name", table_name="task_run") - op.drop_index("ix_task_run_queue", table_name="task_run") - op.drop_index("ix_task_run_celery_task_id", table_name="task_run") - op.drop_table("task_run") diff --git a/alembic/versions/0017_fc3h_backup_run.py b/alembic/versions/0017_fc3h_backup_run.py deleted file mode 100644 index 5b6a839..0000000 --- a/alembic/versions/0017_fc3h_backup_run.py +++ /dev/null @@ -1,82 +0,0 @@ -"""fc3h: backup_run table - -Revision ID: 0017 -Revises: 0016 -Create Date: 2026-05-24 - -Additive. New table records every backup/restore attempt with artifact -metadata. Lifecycle tracking lives in task_run from FC-3i; this is -artifact-only (paths, sizes, tag, restore lineage). -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0017" -down_revision: Union[str, None] = "0016" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "backup_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("kind", sa.String(length=16), nullable=False), - sa.Column( - "status", sa.String(length=16), nullable=False, - server_default="pending", - ), - sa.Column("tag", sa.String(length=64), nullable=True), - sa.Column("triggered_by", sa.String(length=32), nullable=False), - sa.Column("started_at", sa.DateTime(timezone=True), nullable=False), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("sql_path", sa.Text(), nullable=True), - sa.Column("tar_path", sa.Text(), nullable=True), - sa.Column("size_bytes", sa.BigInteger(), nullable=True), - sa.Column("error", sa.Text(), nullable=True), - sa.Column( - "manifest", sa.JSON(), nullable=False, server_default="{}", - ), - sa.Column( - "restored_from_id", sa.Integer(), - sa.ForeignKey("backup_run.id", ondelete="SET NULL"), - nullable=True, - ), - ) - - # Single-column indexes (matches Mapped[...].index=True). - op.create_index("ix_backup_run_kind", "backup_run", ["kind"]) - op.create_index("ix_backup_run_status", "backup_run", ["status"]) - op.create_index("ix_backup_run_tag", "backup_run", ["tag"]) - op.create_index("ix_backup_run_started_at", "backup_run", ["started_at"]) - op.create_index("ix_backup_run_finished_at", "backup_run", ["finished_at"]) - - # Composite indexes for dashboard query patterns. - op.create_index( - "ix_backup_run_kind_started", - "backup_run", ["kind", sa.text("started_at DESC")], - ) - op.create_index( - "ix_backup_run_status_finished", - "backup_run", ["status", sa.text("finished_at DESC")], - ) - # Partial index: only tagged rows participate in retention-exempt query. - op.create_index( - "ix_backup_run_tag_partial", - "backup_run", ["tag"], - postgresql_where=sa.text("tag IS NOT NULL"), - ) - - -def downgrade() -> None: - op.drop_index("ix_backup_run_tag_partial", table_name="backup_run") - op.drop_index("ix_backup_run_status_finished", table_name="backup_run") - op.drop_index("ix_backup_run_kind_started", table_name="backup_run") - op.drop_index("ix_backup_run_finished_at", table_name="backup_run") - op.drop_index("ix_backup_run_started_at", table_name="backup_run") - op.drop_index("ix_backup_run_tag", table_name="backup_run") - op.drop_index("ix_backup_run_status", table_name="backup_run") - op.drop_index("ix_backup_run_kind", table_name="backup_run") - op.drop_table("backup_run") diff --git a/alembic/versions/0018_fc3h_backup_settings.py b/alembic/versions/0018_fc3h_backup_settings.py deleted file mode 100644 index 517c8f1..0000000 --- a/alembic/versions/0018_fc3h_backup_settings.py +++ /dev/null @@ -1,62 +0,0 @@ -"""fc3h: backup_* knobs on import_settings - -Revision ID: 0018 -Revises: 0017 -Create Date: 2026-05-24 - -Adds four columns to the singleton import_settings row: - - backup_db_nightly_enabled (default False — opt-in) - - backup_db_nightly_hour_utc (default 3) - - backup_db_keep_last_n (default 14) - - backup_images_keep_last_n (default 3) - -server_default ensures the singleton row is backfilled in place -without an UPDATE statement. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0018" -down_revision: Union[str, None] = "0017" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "import_settings", - sa.Column( - "backup_db_nightly_enabled", sa.Boolean(), - nullable=False, server_default=sa.false(), - ), - ) - op.add_column( - "import_settings", - sa.Column( - "backup_db_nightly_hour_utc", sa.Integer(), - nullable=False, server_default="3", - ), - ) - op.add_column( - "import_settings", - sa.Column( - "backup_db_keep_last_n", sa.Integer(), - nullable=False, server_default="14", - ), - ) - op.add_column( - "import_settings", - sa.Column( - "backup_images_keep_last_n", sa.Integer(), - nullable=False, server_default="3", - ), - ) - - -def downgrade() -> None: - op.drop_column("import_settings", "backup_images_keep_last_n") - op.drop_column("import_settings", "backup_db_keep_last_n") - op.drop_column("import_settings", "backup_db_nightly_hour_utc") - op.drop_column("import_settings", "backup_db_nightly_enabled") diff --git a/alembic/versions/0019_import_batch_refreshed.py b/alembic/versions/0019_import_batch_refreshed.py deleted file mode 100644 index 1770daa..0000000 --- a/alembic/versions/0019_import_batch_refreshed.py +++ /dev/null @@ -1,38 +0,0 @@ -"""import_batch.refreshed counter for deep-scan sidecar re-application - -Revision ID: 0019 -Revises: 0018 -Create Date: 2026-05-25 - -Adds a `refreshed` counter to `import_batch`, mirroring the existing -`imported`/`skipped`/`failed`/`attachments` columns. Deep scan now -re-applies sidecar metadata to already-imported files (the IR feature -that didn't make the FC port the first time); a "refreshed" outcome -increments this counter so the UI can surface "X new, Y refreshed" -instead of the misleading "Scan complete — no new files" message. - -server_default=0 backfills existing rows in place — no UPDATE needed. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0019" -down_revision: Union[str, None] = "0018" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "import_batch", - sa.Column( - "refreshed", sa.Integer(), - nullable=False, server_default=sa.text("0"), - ), - ) - - -def downgrade() -> None: - op.drop_column("import_batch", "refreshed") diff --git a/alembic/versions/0020_library_audit_run.py b/alembic/versions/0020_library_audit_run.py deleted file mode 100644 index 07a8b87..0000000 --- a/alembic/versions/0020_library_audit_run.py +++ /dev/null @@ -1,65 +0,0 @@ -"""fc-cleanup: library_audit_run table for async transparency/single_color audits - -Revision ID: 0020 -Revises: 0019 -Create Date: 2026-05-26 - -The table backs the async audit lifecycle: rule + params snapshot, status -state machine ('running' → 'ready' → 'applied'/'cancelled'/'error'), and -the matched_ids JSONB array that the apply step deletes. Capped at 50k IDs -per row by the scan task (oversize = rule too aggressive, operator narrows -before re-running). -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from sqlalchemy.dialects import postgresql - -revision: str = "0020" -down_revision: Union[str, None] = "0019" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "library_audit_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("rule", sa.String(32), nullable=False), - sa.Column("params", postgresql.JSONB(astext_type=sa.Text()), nullable=False), - sa.Column( - "status", sa.String(16), - nullable=False, server_default="running", - ), - sa.Column( - "started_at", sa.DateTime(timezone=True), - nullable=False, server_default=sa.func.now(), - ), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column( - "scanned_count", sa.Integer(), - nullable=False, server_default="0", - ), - sa.Column( - "matched_count", sa.Integer(), - nullable=False, server_default="0", - ), - sa.Column( - "matched_ids", postgresql.JSONB(astext_type=sa.Text()), - nullable=False, server_default=sa.text("'[]'::jsonb"), - ), - sa.Column("error", sa.Text(), nullable=True), - ) - op.create_index( - "ix_library_audit_run_rule", "library_audit_run", ["rule"], - ) - op.create_index( - "ix_library_audit_run_status", "library_audit_run", ["status"], - ) - - -def downgrade() -> None: - op.drop_index("ix_library_audit_run_status", table_name="library_audit_run") - op.drop_index("ix_library_audit_run_rule", table_name="library_audit_run") - op.drop_table("library_audit_run") diff --git a/alembic/versions/0021_image_provenance_unique.py b/alembic/versions/0021_image_provenance_unique.py deleted file mode 100644 index b9be941..0000000 --- a/alembic/versions/0021_image_provenance_unique.py +++ /dev/null @@ -1,54 +0,0 @@ -"""provenance-race: dedupe + UNIQUE(image_record_id, post_id) on image_provenance - -Revision ID: 0021 -Revises: 0020 -Create Date: 2026-05-26 - -Closes the race in Importer._apply_sidecar's existence-check + INSERT pattern. -Two workers writing for the same (image, post) pair both saw no existing row -and both inserted, leaving duplicates that then broke .scalar_one_or_none() -on every subsequent deep-scan rederive against those images -(MultipleResultsFound). Most plausibly seeded when the 5-min recovery sweep -re-enqueued a still-running long-import task and the second worker collided -with the first inside _apply_sidecar. - -Migration steps: - 1. DELETE all but min(id) per (image_record_id, post_id) pair. Operator's - DB had 2 affected pairs at write-time; harmless no-op if zero. - 2. Add UNIQUE constraint so the importer's new savepoint+IntegrityError - recovery path can trip on collision and re-select, mirroring - uq_source_artist_platform_url and uq_post_source_external_id. -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0021" -down_revision: Union[str, None] = "0020" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.execute( - """ - DELETE FROM image_provenance ip1 - USING image_provenance ip2 - WHERE ip1.image_record_id = ip2.image_record_id - AND ip1.post_id = ip2.post_id - AND ip1.id > ip2.id - """ - ) - op.create_unique_constraint( - "uq_image_provenance_image_post", - "image_provenance", - ["image_record_id", "post_id"], - ) - - -def downgrade() -> None: - op.drop_constraint( - "uq_image_provenance_image_post", - "image_provenance", - type_="unique", - ) diff --git a/alembic/versions/0022_source_per_artist_platform.py b/alembic/versions/0022_source_per_artist_platform.py deleted file mode 100644 index d3e75dd..0000000 --- a/alembic/versions/0022_source_per_artist_platform.py +++ /dev/null @@ -1,223 +0,0 @@ -"""source-collapse: one Source per (artist, platform) — consolidate junk per-post Sources - -Revision ID: 0022 -Revises: 0021 -Create Date: 2026-05-26 - -Closes the operator-flagged 2026-05-26 issue where the filesystem importer -called _find_or_create_source(url=sd.post_url), creating one Source row per -imported post URL. Operator's Atole artist had 406 Source rows where there -should have been 1 (the /cw/Atole subscription Source). - -Source represents a subscription feed (one per artist+platform — the -gallery-dl URL polled by the FC-3 downloader). Posts hang off it. The -filesystem importer was misusing Source as a per-post key. - -Migration steps per (artist_id, platform) group with >1 Source: - 1. Pick canonical — prefer a URL NOT matching '/posts/$' (real - campaign URL like /cw/Atole); else min(id). - 2. PRE-merge any Posts under non-canonical sources whose - external_post_id ALREADY exists under the canonical source. (Same - gallery-dl post imported via two different sidecar paths can plant - two Post rows with identical external_post_id under different - Sources for the same artist.) Repoint ImageProvenance + - ImageRecord.primary_post_id to the canonical-side Post, dedupe - ImageProvenance against alembic 0021's uq, then delete the - non-canonical-side Post. This MUST happen before step 3 — Postgres - fires uq_post_source_external_id row-by-row during the bulk UPDATE - and the merge-after-reparent ordering 500s on first collision - (operator-hit during v26.05.26.1 deploy, 2026-05-26). - 3. Reparent remaining Posts onto canonical (no collisions possible now). - 4. Reparent ImageProvenance.source_id off the non-canonical sources. - 5. Delete the orphan Source rows. - 6. If the canonical Source's URL still looks like a per-post URL (no - campaign URL existed among candidates), rewrite it to - 'sidecar::' so the artist detail page shows - something readable. -""" -from typing import Sequence, Union - -from alembic import op -from sqlalchemy import text - -revision: str = "0022" -down_revision: Union[str, None] = "0021" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - -_POST_URL_RE = r"/posts/[^/]+$" - - -def upgrade() -> None: - conn = op.get_bind() - - # Find (artist_id, platform) groups with > 1 Source row. - groups = conn.execute(text(""" - SELECT artist_id, platform - FROM source - GROUP BY artist_id, platform - HAVING COUNT(*) > 1 - """)).fetchall() - - for artist_id, platform in groups: - rows = conn.execute( - text(""" - SELECT id, url FROM source - WHERE artist_id = :a AND platform = :p - ORDER BY id ASC - """), - {"a": artist_id, "p": platform}, - ).fetchall() - - # Canonical: first row whose URL doesn't look like a per-post URL; - # else min(id). - canonical_id = None - for sid, url in rows: - if not _matches_post_url(url): - canonical_id = sid - break - if canonical_id is None: - canonical_id = rows[0][0] - - other_ids = [sid for sid, _ in rows if sid != canonical_id] - if not other_ids: - continue - - # STEP 2: PRE-merge ALL Posts with duplicate external_post_id - # across the entire (canonical + others) group, BEFORE the bulk - # reparent. Two cases must both be handled: - # (A) canonical has Post X with epid=N; an "other" source has - # Post Y with epid=N → after bulk UPDATE, (canonical, N) - # collides with itself. - # (B) two different "other" sources each have a Post with - # epid=N; canonical has none → after bulk UPDATE, both - # are repointed to (canonical, N) and the second collides. - # The earlier version of this migration only handled (A); the - # operator's deploy 2026-05-26 tripped (B) at line 139. - # Fix: group ALL Posts in the (artist, platform) by epid; for - # any group with count>1, pick the keep (prefer one already - # under canonical; else lowest id) and merge the rest into it. - all_posts = conn.execute( - text(""" - SELECT external_post_id, id, source_id - FROM post - WHERE source_id = :canonical OR source_id = ANY(:others) - ORDER BY external_post_id, id - """), - {"canonical": canonical_id, "others": other_ids}, - ).fetchall() - by_epid: dict = {} - for epid, post_id, src_id in all_posts: - by_epid.setdefault(epid, []).append((post_id, src_id)) - for _epid, posts in by_epid.items(): - if len(posts) <= 1: - continue - # Prefer a Post already under canonical as the keep. - canonical_posts = [p for p in posts if p[1] == canonical_id] - if canonical_posts: - keep_id = canonical_posts[0][0] - else: - keep_id = posts[0][0] # already sorted by id ASC - drop_ids = [p[0] for p in posts if p[0] != keep_id] - for drop_id in drop_ids: - # Pre-delete image_provenance rows under drop_ whose - # image_record_id ALREADY has a provenance under keep — - # the UPDATE below would otherwise repoint them and - # trip uq_image_provenance_image_post (alembic 0021) - # row-by-row before any after-the-fact dedupe could - # run. Operator's v26.05.26.3 deploy 2026-05-26 tripped - # this at line 123. - conn.execute( - text(""" - DELETE FROM image_provenance - WHERE post_id = :drop_ - AND image_record_id IN ( - SELECT image_record_id FROM image_provenance - WHERE post_id = :keep - ) - """), - {"keep": keep_id, "drop_": drop_id}, - ) - # Now safe to repoint the survivors. - conn.execute( - text(""" - UPDATE image_provenance SET post_id = :keep - WHERE post_id = :drop_ - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text(""" - UPDATE image_record SET primary_post_id = :keep - WHERE primary_post_id = :drop_ - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text("DELETE FROM post WHERE id = :drop_"), - {"drop_": drop_id}, - ) - - # STEP 3: Bulk reparent the remaining Posts off the other - # Sources. After step 2, no collisions on - # (canonical, external_post_id) are possible. - conn.execute( - text(""" - UPDATE post SET source_id = :canonical - WHERE source_id = ANY(:others) - """), - {"canonical": canonical_id, "others": other_ids}, - ) - - # STEP 4: Reparent ImageProvenance.source_id (denormalized FK). - # No UNIQUE on source_id; safe bulk update. - conn.execute( - text(""" - UPDATE image_provenance SET source_id = :canonical - WHERE source_id = ANY(:others) - """), - {"canonical": canonical_id, "others": other_ids}, - ) - - # STEP 5: Drop the orphan Sources. - conn.execute( - text("DELETE FROM source WHERE id = ANY(:others)"), - {"others": other_ids}, - ) - - # If the canonical's URL still looks per-post (no campaign URL - # existed among the candidates), rewrite to a synthetic anchor so - # the artist detail page renders something readable. - canonical_url = conn.execute( - text("SELECT url FROM source WHERE id = :id"), - {"id": canonical_id}, - ).scalar_one() - if _matches_post_url(canonical_url): - slug = conn.execute( - text("SELECT slug FROM artist WHERE id = :id"), - {"id": artist_id}, - ).scalar_one() - conn.execute( - text(""" - UPDATE source - SET url = :new_url, enabled = false - WHERE id = :id - """), - { - "id": canonical_id, - "new_url": f"sidecar:{platform}:{slug}", - }, - ) - - -def downgrade() -> None: - # Lossy migration — orphan Sources deleted, Posts reparented, Posts - # merged. No safe downgrade. If you need to roll back the schema - # invariant, fork from 0021 and re-run filesystem imports. - pass - - -def _matches_post_url(url: str) -> bool: - """True if url ends with /posts/ (gallery-dl-style per-post URL).""" - import re - return bool(re.search(_POST_URL_RE, url or "")) diff --git a/alembic/versions/0023_drop_meta_rating_tag_kinds.py b/alembic/versions/0023_drop_meta_rating_tag_kinds.py deleted file mode 100644 index fc65ea1..0000000 --- a/alembic/versions/0023_drop_meta_rating_tag_kinds.py +++ /dev/null @@ -1,99 +0,0 @@ -"""drop meta + rating tag kinds — operator-retired 2026-05-26 - -Revision ID: 0023 -Revises: 0022 -Create Date: 2026-05-26 - -Operator decided meta + rating aren't valid tag kinds for FC. Per-row -behavior: DELETE existing rows (operator chose "clean break" over -"convert to general"). All cascading FKs (image_tag, tag_alias, -tag_allowlist, tag_reference_embedding, tag_suggestion_rejection, -series_page) use ondelete="CASCADE" so a single DELETE on tag cleans -the related rows in one go. - -After the data cleanup, recreate the tag_kind ENUM without 'meta' / -'rating' (Postgres has no `ALTER TYPE ... DROP VALUE`; standard -rename-create-cast-drop dance). The server default 'general' is -dropped before the type swap and restored after. -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0023" -down_revision: Union[str, None] = "0022" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # 1. Delete tags of the retired kinds. CASCADE handles related tables. - op.execute("DELETE FROM tag WHERE kind IN ('meta', 'rating')") - - # 2. Drop the CHECK constraint that references the enum's literal - # values. Postgres can't resolve `kind = 'character'` across the - # type swap below — the literal would bind to the new tag_kind - # but the column is on tag_kind_old, producing - # "operator does not exist: tag_kind = tag_kind_old". - # (Operator-hit during the v26.05.26.5 deploy attempt; ck was - # originally added by alembic 0002.) Recreated post-swap. - op.drop_constraint( - "ck_tag_fandom_requires_character", "tag", type_="check" - ) - - # 3. Drop the server default — ALTER COLUMN TYPE can't carry it - # across the type swap below. - op.execute("ALTER TABLE tag ALTER COLUMN kind DROP DEFAULT") - - # 4. Recreate the tag_kind enum without meta/rating. - op.execute("ALTER TYPE tag_kind RENAME TO tag_kind_old") - op.execute( - "CREATE TYPE tag_kind AS ENUM (" - "'artist', 'character', 'fandom', 'general', " - "'series', 'archive', 'post'" - ")" - ) - op.execute( - "ALTER TABLE tag " - "ALTER COLUMN kind TYPE tag_kind " - "USING kind::text::tag_kind" - ) - op.execute("DROP TYPE tag_kind_old") - - # 5. Restore the server default. - op.execute("ALTER TABLE tag ALTER COLUMN kind SET DEFAULT 'general'") - - # 6. Restore the CHECK constraint (now bound to the new tag_kind). - op.create_check_constraint( - "ck_tag_fandom_requires_character", - "tag", - "(fandom_id IS NULL) OR (kind = 'character')", - ) - - -def downgrade() -> None: - # Add the values back to the enum so old code can boot. The deleted - # tag rows are gone permanently — no safe restore. - op.drop_constraint( - "ck_tag_fandom_requires_character", "tag", type_="check" - ) - op.execute("ALTER TABLE tag ALTER COLUMN kind DROP DEFAULT") - op.execute("ALTER TYPE tag_kind RENAME TO tag_kind_old") - op.execute( - "CREATE TYPE tag_kind AS ENUM (" - "'artist', 'character', 'fandom', 'general', " - "'series', 'archive', 'post', 'meta', 'rating'" - ")" - ) - op.execute( - "ALTER TABLE tag " - "ALTER COLUMN kind TYPE tag_kind " - "USING kind::text::tag_kind" - ) - op.execute("DROP TYPE tag_kind_old") - op.execute("ALTER TABLE tag ALTER COLUMN kind SET DEFAULT 'general'") - op.create_check_constraint( - "ck_tag_fandom_requires_character", - "tag", - "(fandom_id IS NULL) OR (kind = 'character')", - ) diff --git a/alembic/versions/0024_backfill_post_title_from_description.py b/alembic/versions/0024_backfill_post_title_from_description.py deleted file mode 100644 index 2b1385c..0000000 --- a/alembic/versions/0024_backfill_post_title_from_description.py +++ /dev/null @@ -1,80 +0,0 @@ -"""backfill post.post_title from description first-line — 2026-05-27 - -Revision ID: 0024 -Revises: 0023 -Create Date: 2026-05-27 - -SubscribeStar gallery-dl always writes `title: ""` and embeds the leading -sentence inside `content` HTML. FC's sidecar parser was leaving -post_title NULL for every SubscribeStar post since FC-3 shipped. The -parser fix (sidecar._first_line_text fallback) now synthesizes a title -at parse time; this migration applies the same logic retroactively to -existing rows. - -Operator-flagged 2026-05-27 after inspecting -/mnt/Data/Patreon/Cheunart/subscribestar/ sidecars. - -Idempotent: only touches rows where post_title IS NULL or empty AND -description IS NOT NULL. Re-running the migration is a no-op. -""" -from __future__ import annotations - -import re -from typing import Sequence, Union - -from alembic import op -from sqlalchemy import text - -revision: str = "0024" -down_revision: Union[str, None] = "0023" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -_TAG_RE = re.compile(r"<[^>]+>") -_WS_RE = re.compile(r"\s+") - - -def _first_line_text(body: str, limit: int = 120) -> str | None: - """Mirror of sidecar._first_line_text. Kept inline so the migration - doesn't carry a runtime import dependency from app code that may - have moved by the time the migration is replayed years from now.""" - if not body: - return None - text_ = _TAG_RE.sub(" ", body) - text_ = text_.replace("\xa0", " ") - for line in text_.splitlines(): - line = _WS_RE.sub(" ", line).strip() - if line: - if len(line) > limit: - return line[: limit - 1].rstrip() + "…" - return line - return None - - -def upgrade() -> None: - bind = op.get_bind() - rows = bind.execute( - text( - "SELECT id, description FROM post " - "WHERE (post_title IS NULL OR post_title = '') " - "AND description IS NOT NULL AND description <> ''" - ) - ).fetchall() - updated = 0 - for row in rows: - derived = _first_line_text(row.description) - if not derived: - continue - bind.execute( - text("UPDATE post SET post_title = :t WHERE id = :id"), - {"t": derived, "id": row.id}, - ) - updated += 1 - print(f"0024: backfilled post_title on {updated} row(s)") - - -def downgrade() -> None: - # No safe restore — we can't tell which post_titles were derived vs - # genuinely present. Leave the column alone on rollback. - pass diff --git a/alembic/versions/0025_fix_subscribestar_post_ids.py b/alembic/versions/0025_fix_subscribestar_post_ids.py deleted file mode 100644 index b44f430..0000000 --- a/alembic/versions/0025_fix_subscribestar_post_ids.py +++ /dev/null @@ -1,288 +0,0 @@ -"""sidecar-audit followup: correct external_post_id + post_url across all platforms - -Revision ID: 0025 -Revises: 0024 -Create Date: 2026-05-27 - -Closes the operator-flagged 2026-05-27 sidecar audit findings. Three -data-correctness bugs across non-Patreon platforms had been silently -corrupting Posts since FC-3 shipped; the parser fix (sidecar.py, same -commit) addresses new imports. This migration cleans up existing rows. - -Per-platform actions: - - subscribestar — gallery-dl wrote the per-attachment id in `id` and - the actual post id in `post_id`. FC's parser picked `id`, so every - multi-image SubscribeStar post was fragmented into N Post rows. - 1. For each SubscribeStar Post, read its sidecar (via the related - ImageRecord's on-disk path), pull `post_id`, overwrite - external_post_id and post_url. - 2. Merge groups of Posts under one source that now share an - external_post_id (fragments of the same actual post). Same - ImageProvenance pre-delete + repoint dance as alembic 0022. - - hentaifoundry — sidecars have NO `url` field; `src` is the image - URL. FC's parser stored post_url=NULL. Read each HF Post's sidecar - for `user` + `index`, derive the canonical /pictures/user// - permalink. external_post_id (= `index`) was already correct. - - discord — gallery-dl wrote the CDN attachment URL in `url`. FC's - parser stored that as post_url. Read each Discord Post's sidecar - for the server/channel/message triple, derive the proper - discord.com/channels/.../ permalink. external_post_id (= - `message_id`) was already correct. - - pixiv — pure-SQL backfill: replace any `i.pximg.net`-style URL on - Post.post_url with the derived `/artworks/` permalink. Pixiv - external_post_id (= `id`) was already correct; no sidecar IO - needed. - -Idempotent: re-running on already-corrected data is a no-op (skips -rows whose derived value matches what's already stored). - -Posts whose related ImageRecord paths don't resolve on disk (orphaned -filesystem state) are skipped with a count in the migration output — -those will be picked up by a future deep-scan. -""" -from __future__ import annotations - -import json -import re -from pathlib import Path -from typing import Sequence, Union - -from alembic import op -from sqlalchemy import text - -revision: str = "0025" -down_revision: Union[str, None] = "0024" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -# Mirror of sidecar._NUMBERING_PREFIX. Kept inline so the migration is -# self-contained (the operator's banked rule: -# reference_postgres_enum_swap_drop_checks.md says migrations shouldn't -# import from runtime app code). -_NUMBERING_PREFIX = re.compile(r"^\d+_(.+)$") - - -def _find_sidecar(media_path: Path) -> Path | None: - """gallery-dl writes the sidecar under the unprefixed stem - (`HOLLOW-ICHIGO.json`) while the media file gets a NN_ ordering - prefix (`01_HOLLOW-ICHIGO.png`). Try in order: - 1. .json next to the media - 2. .json next to the media (full-name variant) - 3. strip the NN_ prefix from the stem, then .json - """ - if not media_path: - return None - cand = media_path.with_suffix(".json") - if cand.is_file(): - return cand - cand = media_path.parent / f"{media_path.name}.json" - if cand.is_file(): - return cand - m = _NUMBERING_PREFIX.match(media_path.stem) - if m: - cand = media_path.parent / f"{m.group(1)}.json" - if cand.is_file(): - return cand - return None - - -def _str_id(v) -> str | None: - """str() a JSON scalar id; reject bool (JSON booleans are ints in - Python's eyes but they aren't valid sidecar ids).""" - if isinstance(v, bool): - return None - if isinstance(v, (str, int)) and str(v).strip(): - return str(v).strip() - return None - - -def _str_field(v) -> str | None: - if isinstance(v, str) and v.strip(): - return v.strip() - return None - - -def upgrade() -> None: - conn = op.get_bind() - - # ── PART 1: Per-platform corrections requiring filesystem IO ───── - # SubscribeStar, HentaiFoundry, Discord all need fields from the - # sidecar to construct the right post_url. We walk each Post's - # related ImageRecord.path to find the sidecar, read it, derive, - # and update. - targets = conn.execute(text(""" - SELECT p.id, p.external_post_id, p.post_url, s.platform - FROM post p - JOIN source s ON s.id = p.source_id - WHERE s.platform IN ('subscribestar', 'hentaifoundry', 'discord') - """)).fetchall() - - stats: dict[str, dict[str, int]] = { - plat: {"read": 0, "updated": 0, "no_sidecar": 0} - for plat in ("subscribestar", "hentaifoundry", "discord") - } - for post_row in targets: - plat = post_row.platform - path = _first_attachment_path(conn, post_row.id) - if not path: - stats[plat]["no_sidecar"] += 1 - continue - sidecar = _find_sidecar(Path(path)) - if sidecar is None: - stats[plat]["no_sidecar"] += 1 - continue - try: - data = json.loads(sidecar.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError): - stats[plat]["no_sidecar"] += 1 - continue - stats[plat]["read"] += 1 - - new_epid = post_row.external_post_id - new_url = None - if plat == "subscribestar": - pid = _str_id(data.get("post_id")) - if pid: - new_epid = pid - new_url = f"https://www.subscribestar.com/posts/{pid}" - elif plat == "hentaifoundry": - user = _str_field(data.get("user")) or _str_field(data.get("artist")) - idx = _str_id(data.get("index")) - if user and idx: - new_url = f"https://www.hentai-foundry.com/pictures/user/{user}/{idx}" - elif plat == "discord": - sid = _str_id(data.get("server_id")) - cid = _str_id(data.get("channel_id")) - mid = _str_id(data.get("message_id")) - if sid and cid and mid: - new_url = f"https://discord.com/channels/{sid}/{cid}/{mid}" - - # Idempotent: skip if nothing changed. - if new_epid == post_row.external_post_id and new_url == post_row.post_url: - continue - conn.execute( - text(""" - UPDATE post - SET external_post_id = :epid, post_url = :url - WHERE id = :id - """), - {"epid": new_epid, "url": new_url, "id": post_row.id}, - ) - stats[plat]["updated"] += 1 - - for plat, s in stats.items(): - print( - f"0025: {plat} — read {s['read']} sidecars, " - f"updated {s['updated']} Posts, " - f"{s['no_sidecar']} Posts had no resolvable sidecar" - ) - - # ── PART 2: Merge SubscribeStar fragments now sharing epid ─────── - # After Part 1, each group of Posts under one source with the SAME - # new external_post_id is a fragment-set of the same actual post. - # Merge to one canonical row. Pre-handle the same ImageProvenance - # collision pattern as alembic 0022 (uq_image_provenance_image_post). - fragment_groups = conn.execute(text(""" - SELECT p.source_id, p.external_post_id, - ARRAY_AGG(p.id ORDER BY p.id ASC) AS post_ids - FROM post p - JOIN source s ON s.id = p.source_id - WHERE s.platform = 'subscribestar' - AND p.external_post_id IS NOT NULL - GROUP BY p.source_id, p.external_post_id - HAVING COUNT(*) > 1 - """)).fetchall() - - merged = 0 - for grp in fragment_groups: - post_ids = list(grp.post_ids) - keep_id, *drop_ids = post_ids - for drop_id in drop_ids: - # Pre-DELETE colliding ImageProvenance under drop_ that - # already exist under keep (alembic 0022 banked the pattern). - conn.execute( - text(""" - DELETE FROM image_provenance - WHERE post_id = :drop_ - AND image_record_id IN ( - SELECT image_record_id FROM image_provenance - WHERE post_id = :keep - ) - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text(""" - UPDATE image_provenance SET post_id = :keep - WHERE post_id = :drop_ - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text(""" - UPDATE image_record SET primary_post_id = :keep - WHERE primary_post_id = :drop_ - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text(""" - UPDATE post_attachment SET post_id = :keep - WHERE post_id = :drop_ - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text("DELETE FROM post WHERE id = :drop_"), - {"drop_": drop_id}, - ) - merged += 1 - print(f"0025: subscribestar — merged {merged} duplicate Post fragments") - - # ── PART 3: Pixiv post_url backfill (pure SQL) ─────────────────── - # Pixiv's external_post_id is already correct (gallery-dl's `id` is - # the post id). Only post_url needs derivation: replace anything - # under i.pximg.net (the file URL) with the /artworks/ permalink. - pixiv_updated = conn.execute(text(""" - UPDATE post p - SET post_url = 'https://www.pixiv.net/artworks/' || p.external_post_id - FROM source s - WHERE p.source_id = s.id - AND s.platform = 'pixiv' - AND p.external_post_id IS NOT NULL - AND (p.post_url IS NULL - OR p.post_url LIKE 'https://i.pximg.net/%' - OR p.post_url LIKE 'http://i.pximg.net/%') - """)).rowcount - print(f"0025: pixiv — backfilled post_url on {pixiv_updated} Posts") - - -def _first_attachment_path(conn, post_id: int) -> str | None: - """Return any ImageRecord.path attached to this post (via - ImageProvenance). Lowest-id row keeps the migration deterministic - so re-running on the same DB picks the same sidecar.""" - row = conn.execute( - text(""" - SELECT ir.path - FROM image_provenance ip - JOIN image_record ir ON ir.id = ip.image_record_id - WHERE ip.post_id = :pid - ORDER BY ip.id ASC - LIMIT 1 - """), - {"pid": post_id}, - ).first() - return row[0] if row else None - - -def downgrade() -> None: - # Lossy: external_post_id values were overwritten with the correct - # post_id; original per-attachment ids weren't preserved. Post-merge - # also deleted drop rows. No safe restore. To roll back the schema - # invariant, fork from 0024 and re-run sidecar imports. - pass diff --git a/alembic/versions/0026_import_task_recovery_count_refetched.py b/alembic/versions/0026_import_task_recovery_count_refetched.py deleted file mode 100644 index ccbc3da..0000000 --- a/alembic/versions/0026_import_task_recovery_count_refetched.py +++ /dev/null @@ -1,53 +0,0 @@ -"""import_task.recovery_count + refetched — poison-pill circuit breaker - -Revision ID: 0026 -Revises: 0025 -Create Date: 2026-05-28 - -Backs the import-task resilience work (operator-flagged 2026-05-28): - -- recovery_count: how many times recover_interrupted_tasks has - re-queued this row from a stuck 'processing' state. A row that - hard-crashes the worker (OOM / segfault on a corrupt or oversized - input) leaves no terminal flip, so the sweep re-queues it — and - without a cap it would loop forever, re-crashing the worker each - time. After MAX_RECOVERY_ATTEMPTS the sweep marks it 'failed' with a - diagnostic instead. - -- refetched: whether a one-shot re-download has already been attempted - for this task's file. Bounds the Layer-2 re-fetch remediation to a - single attempt so source-side corruption doesn't loop. - -Both default to 0 / false; additive, no backfill needed. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0026" -down_revision: Union[str, None] = "0025" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "import_task", - sa.Column( - "recovery_count", sa.Integer(), nullable=False, - server_default="0", - ), - ) - op.add_column( - "import_task", - sa.Column( - "refetched", sa.Boolean(), nullable=False, - server_default=sa.false(), - ), - ) - - -def downgrade() -> None: - op.drop_column("import_task", "refetched") - op.drop_column("import_task", "recovery_count") diff --git a/alembic/versions/0027_drop_migration_run.py b/alembic/versions/0027_drop_migration_run.py deleted file mode 100644 index 454481b..0000000 --- a/alembic/versions/0027_drop_migration_run.py +++ /dev/null @@ -1,50 +0,0 @@ -"""drop migration_run — one-and-done GS/IR migration tooling removed - -Revision ID: 0027 -Revises: 0026 -Create Date: 2026-05-29 - -The GS/IR migration tooling (services/migrators, /api/migrate, the -run_migration task, LegacyMigrationCard, and the MigrationRun model) was -removed after the migration cutover completed. This drops its now-orphaned -run-log table. Downgrade recreates the table (mirrors the old model) so the -migration is reversible. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from sqlalchemy.dialects.postgresql import JSONB - -revision: str = "0027" -down_revision: Union[str, None] = "0026" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.drop_table("migration_run") - - -def downgrade() -> None: - op.create_table( - "migration_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("kind", sa.String(length=32), nullable=False), - sa.Column("status", sa.String(length=32), nullable=False), - sa.Column("dry_run", sa.Boolean(), nullable=False, server_default=sa.false()), - sa.Column( - "started_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column( - "counts", JSONB(), nullable=False, server_default=sa.text("'{}'::jsonb"), - ), - sa.Column("error", sa.Text(), nullable=True), - sa.Column( - "metadata", JSONB(), nullable=False, server_default=sa.text("'{}'::jsonb"), - ), - ) - op.create_index("ix_migration_run_kind", "migration_run", ["kind"]) - op.create_index("ix_migration_run_status", "migration_run", ["status"]) diff --git a/alembic/versions/0028_collapse_sidecar_synthetics_into_real_sources.py b/alembic/versions/0028_collapse_sidecar_synthetics_into_real_sources.py deleted file mode 100644 index 5ec9267..0000000 --- a/alembic/versions/0028_collapse_sidecar_synthetics_into_real_sources.py +++ /dev/null @@ -1,190 +0,0 @@ -"""collapse-sidecar-synthetic: repoint Posts/ImageProvenance/DownloadEvents -from `sidecar::` synthetic Source anchors onto the real -Source for the same (artist, platform) when one exists, then delete the -synthetic. - -Revision ID: 0028 -Revises: 0027 -Create Date: 2026-05-31 - -Background: alembic 0022 (2026-05-26) consolidated the old per-post-URL -Source rows into one canonical Source per (artist, platform). When NO -real campaign URL was salvageable among the candidates, it rewrote the -canonical row to url='sidecar::' enabled=false as a -disabled anchor for any Posts already attached. - -That was fine while it was the only Source for that artist+platform. -But: the unique constraint on Source is (artist_id, platform, url), not -(artist_id, platform). When the operator later added the real -subscription via the UI / extension / etc., a SECOND row landed — -the real one — with id > the synthetic. Both coexisted. - -Two follow-on problems surfaced 2026-05-31: - - 1. The Subscriptions UI listed both rows. The synthetic was disabled - so the scheduler never polled it, but it looked like a phantom - subscription. (Fixed in same commit by SourceService.list filter.) - 2. importer._source_for_sidecar picked Source by `ORDER BY id ASC - LIMIT 1`, so EVERY gallery-dl download since the real Source was - added attached its Post to the SYNTHETIC anchor, not the real - Source. (Fixed in same commit by preferring non-sidecar URLs.) - -This migration is the data half of the cleanup: for every (artist, -platform) with both a synthetic AND a real Source, repoint the -synthetic's children (Posts, ImageProvenance, DownloadEvents) onto the -real Source and delete the synthetic. Reuses the same epid/provenance -collision dance from alembic 0022 because the same uniqueness -constraints fire row-by-row during bulk UPDATEs. - -Lone synthetic anchors — those where no real Source for the same -(artist, platform) exists (e.g., filesystem-imported artist with no -subscription added) — are LEFT INTACT. They anchor real imported -content; deleting them would CASCADE-delete the Posts the operator -imported. The SourceService.list filter hides them from the UI; the -operator can delete them by hand if they want the underlying imports -gone. -""" -from typing import Sequence, Union - -from alembic import op -from sqlalchemy import text - -revision: str = "0028" -down_revision: Union[str, None] = "0027" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - conn = op.get_bind() - - # Find (artist_id, platform) groups where BOTH a sidecar synthetic - # and at least one real Source exist. - groups = conn.execute(text(""" - SELECT artist_id, platform - FROM source - GROUP BY artist_id, platform - HAVING bool_or(url LIKE 'sidecar:%') - AND bool_or(url NOT LIKE 'sidecar:%') - """)).fetchall() - - for artist_id, platform in groups: - rows = conn.execute( - text(""" - SELECT id, url FROM source - WHERE artist_id = :a AND platform = :p - ORDER BY id ASC - """), - {"a": artist_id, "p": platform}, - ).fetchall() - - synthetic_ids = [sid for sid, url in rows if url.startswith("sidecar:")] - real_rows = [(sid, url) for sid, url in rows if not url.startswith("sidecar:")] - if not synthetic_ids or not real_rows: - continue # belt+suspenders; the GROUP BY already filtered - - # Canonical real: lowest-id non-sidecar Source. - canonical_id = real_rows[0][0] - - # STEP A: PRE-merge Post collisions on (canonical, external_post_id). - # Mirror alembic 0022's pre-merge logic — when synth has Post X - # epid=N and real has Post Y epid=N, the bulk UPDATE below would - # trip uq_post_source_external_id row-by-row. Group all Posts - # under (canonical + synthetics) by epid; for any group >1, - # pick a keep (prefer one already under canonical, else lowest - # id) and merge the rest into it. - all_posts = conn.execute( - text(""" - SELECT external_post_id, id, source_id - FROM post - WHERE source_id = :canonical OR source_id = ANY(:synths) - ORDER BY external_post_id, id - """), - {"canonical": canonical_id, "synths": synthetic_ids}, - ).fetchall() - by_epid: dict = {} - for epid, post_id, src_id in all_posts: - by_epid.setdefault(epid, []).append((post_id, src_id)) - for _epid, posts in by_epid.items(): - if len(posts) <= 1: - continue - canonical_side = [p for p in posts if p[1] == canonical_id] - keep_id = canonical_side[0][0] if canonical_side else posts[0][0] - drop_ids = [p[0] for p in posts if p[0] != keep_id] - for drop_id in drop_ids: - # Pre-delete image_provenance rows under drop_ whose - # image_record_id already has provenance under keep — - # avoids tripping uq_image_provenance_image_post (0021) - # row-by-row during the repoint UPDATE. - conn.execute( - text(""" - DELETE FROM image_provenance - WHERE post_id = :drop_ - AND image_record_id IN ( - SELECT image_record_id FROM image_provenance - WHERE post_id = :keep - ) - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text(""" - UPDATE image_provenance SET post_id = :keep - WHERE post_id = :drop_ - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text(""" - UPDATE image_record SET primary_post_id = :keep - WHERE primary_post_id = :drop_ - """), - {"keep": keep_id, "drop_": drop_id}, - ) - conn.execute( - text("DELETE FROM post WHERE id = :drop_"), - {"drop_": drop_id}, - ) - - # STEP B: Bulk reparent the remaining Posts off the synthetics. - conn.execute( - text(""" - UPDATE post SET source_id = :canonical - WHERE source_id = ANY(:synths) - """), - {"canonical": canonical_id, "synths": synthetic_ids}, - ) - - # STEP C: Reparent ImageProvenance.source_id (denormalized FK; - # no UNIQUE on source_id, safe bulk). - conn.execute( - text(""" - UPDATE image_provenance SET source_id = :canonical - WHERE source_id = ANY(:synths) - """), - {"canonical": canonical_id, "synths": synthetic_ids}, - ) - - # STEP D: Reparent any DownloadEvent.source_id. Synthetics are - # enabled=false so the scheduler never created events for them; - # this is belt+suspenders for any rows planted by manual force - # or older code paths. - conn.execute( - text(""" - UPDATE download_event SET source_id = :canonical - WHERE source_id = ANY(:synths) - """), - {"canonical": canonical_id, "synths": synthetic_ids}, - ) - - # STEP E: Drop the now-empty synthetics. - conn.execute( - text("DELETE FROM source WHERE id = ANY(:synths)"), - {"synths": synthetic_ids}, - ) - - -def downgrade() -> None: - # Lossy migration — synthetic Sources deleted, Posts repointed and - # potentially merged. No safe downgrade. - pass diff --git a/alembic/versions/0029_drop_artist_copyright_ml_thresholds.py b/alembic/versions/0029_drop_artist_copyright_ml_thresholds.py deleted file mode 100644 index e6c044b..0000000 --- a/alembic/versions/0029_drop_artist_copyright_ml_thresholds.py +++ /dev/null @@ -1,71 +0,0 @@ -"""drop artist + copyright ml thresholds; lower general default to 0.50 - -Revision ID: 0029 -Revises: 0028 -Create Date: 2026-06-01 - -Operator-flagged 2026-06-01: the view modal's Suggestions panel hides -most general-category predictions because the default threshold is -0.95. Lowering the default to 0.50 (matches character) so general -suggestions surface more aggressively; the value remains tunable in -Settings → ML. - -Same change retires two ML suggestion categories whose Tag.kind -surfaces are unused: - -- `artist`: retired in FC-2d-vii-c — artist identity is acquisition- - derived (image_record.artist_id), never ML-inferred. The threshold - column was a leftover from before that retirement. -- `copyright`: retired 2026-06-01 — the app uses `fandom` for the - franchise/copyright concept (per TagsView.vue's doc comment); no - Tag rows of kind=copyright exist, and the threshold column never - fed anything user-visible. - -Both columns are dropped from ml_settings; the existing row's -suggestion_threshold_general value is bumped from 0.95 to 0.50 iff -it's still at the old default, so deployed installs pick up the new -UX without overriding any operator tuning. -""" -from typing import Sequence, Union - -from alembic import op -from sqlalchemy import text - -revision: str = "0029" -down_revision: Union[str, None] = "0028" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # Bump the general threshold for installs still at the old default. - op.execute(text( - "UPDATE ml_settings " - "SET suggestion_threshold_general = 0.50 " - "WHERE id = 1 AND suggestion_threshold_general = 0.95" - )) - op.drop_column("ml_settings", "suggestion_threshold_artist") - op.drop_column("ml_settings", "suggestion_threshold_copyright") - - -def downgrade() -> None: - # Restore the columns with their prior defaults. The bump from - # 0.95 → 0.50 isn't reversible without remembering whether the - # operator had explicitly set 0.95 (unlikely — that was just the - # default) so we leave the current general value as-is. - from sqlalchemy import Column, Float - - op.add_column( - "ml_settings", - Column( - "suggestion_threshold_artist", - Float, nullable=False, server_default="0.30", - ), - ) - op.add_column( - "ml_settings", - Column( - "suggestion_threshold_copyright", - Float, nullable=False, server_default="0.50", - ), - ) diff --git a/alembic/versions/0030_nullable_post_source_id_denorm_artist_id.py b/alembic/versions/0030_nullable_post_source_id_denorm_artist_id.py deleted file mode 100644 index c37e499..0000000 --- a/alembic/versions/0030_nullable_post_source_id_denorm_artist_id.py +++ /dev/null @@ -1,145 +0,0 @@ -"""nullable post.source_id + denormalized post.artist_id; retire sidecar synthetics - -Revision ID: 0030 -Revises: 0029 -Create Date: 2026-06-01 - -Operator-asked 2026-06-01 after the Dymkens orphan investigation: the -sidecar synthetic Source pattern (`sidecar::` rows -with enabled=false) was technically correct but misled the operator -into thinking they had phantom subscriptions. The synthetics existed -solely to satisfy `Post.source_id NOT NULL` for filesystem-imported -content with no real subscription. - -This migration makes the data model honest: - -1. **Post gets a denormalized `artist_id` column** so artist filters - work without traversing `Post → Source.artist_id`. Backfilled from - the existing Source linkage, then NOT NULL'd. -2. **`Post.source_id` becomes nullable**, FK ondelete `CASCADE` → `SET - NULL`. Deleting a Source detaches its Posts instead of destroying - imported content (semantically: subscription ends, archive stays). -3. **`ImageProvenance.source_id` becomes nullable** with the same FK - semantic change. -4. **Sidecar synthetic Sources are deleted** — first NULL out the - FKs from Post + ImageProvenance pointing at them (so the implicit - CASCADE doesn't fire), then delete. DownloadEvent FK is unchanged - (still CASCADE'd, NOT NULL'd) — synthetics have `enabled=false` - so no events exist for them. - -Uniqueness handling: the existing `uq_post_source_external_id` -(source_id, external_post_id) keeps working for source-bound Posts -(Postgres treats NULL != NULL so NULL-source rows aren't deduped by -it). A second partial unique index covers the NULL-source case on -(artist_id, external_post_id) so filesystem-imported posts still -dedupe within an artist. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from sqlalchemy import text - -revision: str = "0030" -down_revision: Union[str, None] = "0029" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - conn = op.get_bind() - - # Step 1: add Post.artist_id, initially nullable for backfill. - # FK naming follows the Base.metadata naming_convention - # (fk___) — alembic 0001 set this up. - op.add_column( - "post", - sa.Column("artist_id", sa.Integer, nullable=True), - ) - op.create_foreign_key( - "fk_post_artist_id_artist", "post", "artist", - ["artist_id"], ["id"], ondelete="CASCADE", - ) - - # Step 2: backfill from Source.artist_id (every existing Post has a - # Source today, so every row gets populated). - conn.execute(text(""" - UPDATE post p - SET artist_id = s.artist_id - FROM source s - WHERE p.source_id = s.id AND p.artist_id IS NULL - """)) - - # Sanity: count any remaining NULLs. Should be zero pre-this-migration. - remaining = conn.execute(text( - "SELECT COUNT(*) FROM post WHERE artist_id IS NULL" - )).scalar_one() - if remaining: - raise RuntimeError( - f"alembic 0030: {remaining} post rows have no resolvable " - f"artist_id after backfill. Investigate before continuing." - ) - - # Step 3: enforce NOT NULL + add index for artist-filter queries. - op.alter_column("post", "artist_id", nullable=False) - op.create_index("ix_post_artist_id", "post", ["artist_id"]) - - # Step 4: relax post.source_id + flip FK to SET NULL. The original FK - # name from alembic 0001 is `fk_post_source_id_source` per the - # NAMING_CONVENTION in models/base.py. - op.alter_column("post", "source_id", nullable=True) - op.drop_constraint("fk_post_source_id_source", "post", type_="foreignkey") - op.create_foreign_key( - "fk_post_source_id_source", "post", "source", - ["source_id"], ["id"], ondelete="SET NULL", - ) - - # Step 5: relax image_provenance.source_id + flip FK to SET NULL. - op.alter_column("image_provenance", "source_id", nullable=True) - op.drop_constraint( - "fk_image_provenance_source_id_source", "image_provenance", - type_="foreignkey", - ) - op.create_foreign_key( - "fk_image_provenance_source_id_source", "image_provenance", "source", - ["source_id"], ["id"], ondelete="SET NULL", - ) - - # Step 6: partial unique index on (artist_id, external_post_id) for - # NULL-source Posts. The existing uq_post_source_external_id keeps - # guarding source-bound rows; NULL-source rows now dedupe within - # an artist. - op.execute( - "CREATE UNIQUE INDEX uq_post_artist_external_id_null_source " - "ON post (artist_id, external_post_id) " - "WHERE source_id IS NULL" - ) - - # Step 7: retire sidecar synthetic Sources. NULL out the references - # FIRST (the new FK is SET NULL so CASCADE wouldn't fire anyway, but - # being explicit makes the intent clear). Then delete the synthetic - # source rows. Any DownloadEvent rows under synthetics CASCADE-die - # with the source — synthetics have enabled=false so there shouldn't - # be any in practice. - conn.execute(text(""" - UPDATE post - SET source_id = NULL - WHERE source_id IN (SELECT id FROM source WHERE url LIKE 'sidecar:%') - """)) - conn.execute(text(""" - UPDATE image_provenance - SET source_id = NULL - WHERE source_id IN (SELECT id FROM source WHERE url LIKE 'sidecar:%') - """)) - deleted = conn.execute(text( - "DELETE FROM source WHERE url LIKE 'sidecar:%' RETURNING id" - )).rowcount - print(f"alembic 0030: deleted {deleted} sidecar synthetic source rows") - - -def downgrade() -> None: - # Lossy migration — the deleted sidecar synthetics can't be - # restored from the orphan post.source_id / image_provenance.source_id - # values, and the partial unique index encodes a constraint that - # NULL-source Posts may now exist. No safe downgrade. - pass diff --git a/alembic/versions/0031_source_backfill_runs_remaining.py b/alembic/versions/0031_source_backfill_runs_remaining.py deleted file mode 100644 index ed140fc..0000000 --- a/alembic/versions/0031_source_backfill_runs_remaining.py +++ /dev/null @@ -1,45 +0,0 @@ -"""source.backfill_runs_remaining: sticky deep-scan mode - -Revision ID: 0031 -Revises: 0030 -Create Date: 2026-06-01 - -Tick vs backfill mode for subscription downloads. When -`backfill_runs_remaining > 0`, the next N download runs use -`skip: True` + 30-min timeout (walk full history). When 0, runs use -`skip: "exit:20"` + 14.5-min timeout (catch-up mode, exits early once -20 contiguous archived items are seen). - -Operator-flagged 2026-06-01 (Knuxy run #38887): a creator with ~550 -archived posts saturates the 870s catch-up timeout even when there is -no new content, because gallery-dl's default `skip: True` keeps walking. -Tick mode short-circuits that; backfill mode is the explicit opt-in for -deep history scans. - -Default 0 (all existing subscriptions start in tick mode). -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0031" -down_revision: Union[str, None] = "0030" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "source", - sa.Column( - "backfill_runs_remaining", - sa.Integer, - nullable=False, - server_default="0", - ), - ) - - -def downgrade() -> None: - op.drop_column("source", "backfill_runs_remaining") diff --git a/alembic/versions/0032_source_error_type.py b/alembic/versions/0032_source_error_type.py deleted file mode 100644 index e264b35..0000000 --- a/alembic/versions/0032_source_error_type.py +++ /dev/null @@ -1,41 +0,0 @@ -"""source.error_type: surface ErrorType taxonomy in FailingSourcesCard - -Revision ID: 0032 -Revises: 0031 -Create Date: 2026-06-02 - -Audit 2026-06-02: the backend computes 13 ErrorType categories (auth_error, -rate_limited, not_found, access_denied, validation_failed, etc.) and -stamps each one on DownloadEvent.metadata, but the Source row only carried -the free-text last_error. Operators couldn't bulk-triage failing sources -("all auth_error → rotate cookies, all rate_limited → just wait") without -opening Logs per row. - -This column receives the last error_type from _update_source_health -and gets cleared on a successful run. Nullable + indexed so the failing- -sources rollup can filter/group cheaply. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0032" -down_revision: Union[str, None] = "0031" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "source", - sa.Column("error_type", sa.String(length=32), nullable=True), - ) - op.create_index( - "ix_source_error_type", "source", ["error_type"], - ) - - -def downgrade() -> None: - op.drop_index("ix_source_error_type", table_name="source") - op.drop_column("source", "error_type") diff --git a/alembic/versions/0033_suggestion_threshold_default_070.py b/alembic/versions/0033_suggestion_threshold_default_070.py deleted file mode 100644 index 652cf44..0000000 --- a/alembic/versions/0033_suggestion_threshold_default_070.py +++ /dev/null @@ -1,48 +0,0 @@ -"""suggestion_threshold default 0.50 → 0.70 - -Revision ID: 0033 -Revises: 0032 -Create Date: 2026-06-02 - -Operator-flagged 2026-06-02 — the 0.50 default (set on 2026-06-01) is -too noisy in practice; raise to 0.70 for both suggestion categories. - -Only conditionally updates singletons whose current value is still the -2026-06-01 default (0.50). Operators who deliberately tuned their row -to some other value (0.55, 0.65, 0.80, etc. via the Settings UI) keep -their pick — the migration only catches the unchanged-default case. -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0033" -down_revision: Union[str, None] = "0032" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.execute( - "UPDATE ml_settings " - "SET suggestion_threshold_character = 0.70 " - "WHERE id = 1 AND suggestion_threshold_character = 0.50" - ) - op.execute( - "UPDATE ml_settings " - "SET suggestion_threshold_general = 0.70 " - "WHERE id = 1 AND suggestion_threshold_general = 0.50" - ) - - -def downgrade() -> None: - op.execute( - "UPDATE ml_settings " - "SET suggestion_threshold_character = 0.50 " - "WHERE id = 1 AND suggestion_threshold_character = 0.70" - ) - op.execute( - "UPDATE ml_settings " - "SET suggestion_threshold_general = 0.50 " - "WHERE id = 1 AND suggestion_threshold_general = 0.70" - ) diff --git a/alembic/versions/0034_artist_visit.py b/alembic/versions/0034_artist_visit.py deleted file mode 100644 index a2234a6..0000000 --- a/alembic/versions/0034_artist_visit.py +++ /dev/null @@ -1,53 +0,0 @@ -"""artist_visit: per-artist last-viewed timestamp for the "+N new" badge - -Revision ID: 0034 -Revises: 0033 -Create Date: 2026-06-03 - -Powers the artists-directory "+N new since last visit" badge + ArtistView -banner. Single row per artist (no user_id yet — rule #47 multi-user ACL -is aspirational; widens to (user_id, artist_id) PK when User lands). - -Seed every existing artist with `last_viewed_at = NOW()` so the badge -starts at 0 across the board — no noisy "you have 5000 unseen images" -on first deploy. New artists auto-get a row via -`ArtistService.find_or_create`. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0034" -down_revision: Union[str, None] = "0033" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "artist_visit", - sa.Column( - "artist_id", - sa.Integer, - sa.ForeignKey("artist.id", ondelete="CASCADE"), - primary_key=True, - ), - sa.Column( - "last_viewed_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - ) - # Seed: every existing artist starts "fully caught up". Without this, - # every operator with N artists would see N badges (worth of every - # image ever imported) on first deploy. - op.execute( - "INSERT INTO artist_visit (artist_id, last_viewed_at) " - "SELECT id, NOW() FROM artist" - ) - - -def downgrade() -> None: - op.drop_table("artist_visit") diff --git a/alembic/versions/0035_image_record_effective_date.py b/alembic/versions/0035_image_record_effective_date.py deleted file mode 100644 index 586cf51..0000000 --- a/alembic/versions/0035_image_record_effective_date.py +++ /dev/null @@ -1,70 +0,0 @@ -"""image_record.effective_date: materialized gallery sort key + index - -Revision ID: 0035 -Revises: 0034 -Create Date: 2026-06-04 - -The gallery ordered/cursored on COALESCE(post.post_date, -image_record.created_at) across the Post outer join. That expression spans -two tables, so no index can serve it — every /scroll sorted a large slice -of the library, and the frontend fired ten of them serially per initial -load. Materialize the value into image_record.effective_date and index -(effective_date DESC, id DESC) so the cursor scroll is an index range scan. - -Backfill = COALESCE(primary post's post_date, created_at) so existing rows -keep their exact ordering. New rows get the created_at-equivalent server -default; services/importer.py overrides it with the post's date when a -primary post with a date is linked. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0035" -down_revision: Union[str, None] = "0034" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # Add nullable first so the backfill can populate before NOT NULL. - op.add_column( - "image_record", - sa.Column("effective_date", sa.DateTime(timezone=True), nullable=True), - ) - # Pure set-based UPDATEs (no per-row params) — immune to the 65535 - # bind-parameter ceiling regardless of library size. - op.execute( - """ - UPDATE image_record AS ir - SET effective_date = COALESCE(p.post_date, ir.created_at) - FROM post AS p - WHERE ir.primary_post_id = p.id - """ - ) - op.execute( - """ - UPDATE image_record - SET effective_date = created_at - WHERE effective_date IS NULL - """ - ) - op.alter_column( - "image_record", - "effective_date", - nullable=False, - server_default=sa.text("now()"), - ) - # DESC/DESC matches the gallery's ORDER BY effective_date DESC, id DESC - # so the scroll is a forward index scan; raw SQL because alembic's - # column list doesn't express per-column DESC cleanly. - op.execute( - "CREATE INDEX ix_image_record_effective_date " - "ON image_record (effective_date DESC, id DESC)" - ) - - -def downgrade() -> None: - op.drop_index("ix_image_record_effective_date", table_name="image_record") - op.drop_column("image_record", "effective_date") diff --git a/alembic/versions/0036_siglip_embedding_hnsw_index.py b/alembic/versions/0036_siglip_embedding_hnsw_index.py deleted file mode 100644 index a8c1251..0000000 --- a/alembic/versions/0036_siglip_embedding_hnsw_index.py +++ /dev/null @@ -1,41 +0,0 @@ -"""image_record.siglip_embedding: HNSW cosine index for "more like this" - -Revision ID: 0036 -Revises: 0035 -Create Date: 2026-06-04 - -Gallery Phase 3 (visual similarity search) ranks images by -`siglip_embedding.cosine_distance(source_embedding)`. Without an index that's -a sequential scan computing a 1152-dim distance for every row — fine at small -scale, but it grows linearly with the library. Add an HNSW index with -`vector_cosine_ops` so the top-N nearest search is sub-50ms ANN. - -1152 dims is under pgvector's 2000-dim HNSW limit, so HNSW (no training, -better recall than IVFFlat) is the right choice. ONE-TIME COST: building the -index over the existing embeddings (~57k vectors on the operator's library) -locks image_record for ~30-60s during this migration on deploy — acceptable -for a single-operator homelab. NULL embeddings (videos / not-yet-embedded -rows) are simply not indexed. -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0036" -down_revision: Union[str, None] = "0035" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # Raw SQL: alembic's create_index doesn't express the `USING hnsw (... - # vector_cosine_ops)` access-method + opclass cleanly. Must match the - # query's cosine_distance operator class to be usable by the planner. - op.execute( - "CREATE INDEX ix_image_record_siglip_hnsw " - "ON image_record USING hnsw (siglip_embedding vector_cosine_ops)" - ) - - -def downgrade() -> None: - op.drop_index("ix_image_record_siglip_hnsw", table_name="image_record") diff --git a/alembic/versions/0037_patreon_seen_media.py b/alembic/versions/0037_patreon_seen_media.py deleted file mode 100644 index 255484e..0000000 --- a/alembic/versions/0037_patreon_seen_media.py +++ /dev/null @@ -1,53 +0,0 @@ -"""patreon_seen_media: per-source ledger of already-ingested Patreon media - -Revision ID: 0037 -Revises: 0036 -Create Date: 2026-06-05 - -Native Patreon ingester (build step 2a). Replaces gallery-dl's -archive.sqlite3 with our own queryable table. The downloader upserts one -row per (source, media) so routine walks skip media we've already -processed; a future "recovery" mode bypasses the ledger to re-walk. - -`filehash` is a 32-hex Patreon CDN MD5, OR a video sentinel of the form -``video::`` — hence String(128). The unique -constraint on (source_id, filehash) is the dedup upsert key. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0037" -down_revision: Union[str, None] = "0036" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "patreon_seen_media", - sa.Column("id", sa.Integer, primary_key=True), - sa.Column( - "source_id", - sa.Integer, - sa.ForeignKey("source.id", ondelete="CASCADE"), - nullable=False, - index=True, - ), - sa.Column("filehash", sa.String(128), nullable=False), - sa.Column("post_id", sa.String(64), nullable=True), - sa.Column( - "seen_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.UniqueConstraint( - "source_id", "filehash", name="uq_patreon_seen_media_source_id" - ), - ) - - -def downgrade() -> None: - op.drop_table("patreon_seen_media") diff --git a/alembic/versions/0038_patreon_failed_media.py b/alembic/versions/0038_patreon_failed_media.py deleted file mode 100644 index e907ae1..0000000 --- a/alembic/versions/0038_patreon_failed_media.py +++ /dev/null @@ -1,58 +0,0 @@ -"""patreon_failed_media: per-source dead-letter ledger for failing Patreon media - -Revision ID: 0038 -Revises: 0037 -Create Date: 2026-06-06 - -Plan #705 (#7). Media that keeps failing to download/validate (404'd CDN, -deleted post, geo-blocked Mux, persistently-corrupt bytes) gets recorded here -with an attempt counter; once it crosses the dead-letter threshold the ingester -skips it on routine walks (recovery still re-attempts). A clean download clears -the row. UNIQUE (source_id, filehash) is the upsert key (same media key the -seen-ledger uses). -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0038" -down_revision: Union[str, None] = "0037" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "patreon_failed_media", - sa.Column("id", sa.Integer, primary_key=True), - sa.Column( - "source_id", - sa.Integer, - sa.ForeignKey("source.id", ondelete="CASCADE"), - nullable=False, - index=True, - ), - sa.Column("filehash", sa.String(128), nullable=False), - sa.Column("attempts", sa.Integer, nullable=False, server_default="1"), - sa.Column("last_error", sa.Text, nullable=True), - sa.Column( - "first_failed_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.Column( - "last_failed_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.UniqueConstraint( - "source_id", "filehash", name="uq_patreon_failed_media_source_id" - ), - ) - - -def downgrade() -> None: - op.drop_table("patreon_failed_media") diff --git a/alembic/versions/0039_library_audit_resume.py b/alembic/versions/0039_library_audit_resume.py deleted file mode 100644 index 6cfb8f9..0000000 --- a/alembic/versions/0039_library_audit_resume.py +++ /dev/null @@ -1,40 +0,0 @@ -"""library_audit_run: resume cursor + progress timestamp for chunked scans - -Revision ID: 0039 -Revises: 0038 -Create Date: 2026-06-07 - -scan_library_for_rule used to run one 2h pass that timed out on large libraries -and monopolized the concurrency-1 maintenance queue (operator-flagged). It now -runs short time-boxed chunks that re-enqueue: `resume_after_id` persists the -keyset cursor so the next chunk continues where it left off, and -`last_progress_at` lets the recovery sweep tell a progressing multi-chunk audit -from a genuinely stuck one. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0039" -down_revision: Union[str, None] = "0038" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "library_audit_run", - sa.Column( - "resume_after_id", sa.Integer, nullable=False, server_default="0" - ), - ) - op.add_column( - "library_audit_run", - sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True), - ) - - -def downgrade() -> None: - op.drop_column("library_audit_run", "last_progress_at") - op.drop_column("library_audit_run", "resume_after_id") diff --git a/alembic/versions/0040_series_chapters.py b/alembic/versions/0040_series_chapters.py deleted file mode 100644 index a0808df..0000000 --- a/alembic/versions/0040_series_chapters.py +++ /dev/null @@ -1,108 +0,0 @@ -"""series chapters: chapter layer over series_page (FC-6.1) - -Revision ID: 0040 -Revises: 0039 -Create Date: 2026-06-07 - -A series (Tag kind='series') gains an ordered chapter layer. Reading order -becomes (series_chapter.chapter_number, series_page.page_number). Every existing -series is backfilled into a single auto-chapter (chapter_number=1) holding its -current flat pages, so no data is lost and the old flat ordering is preserved. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0040" -down_revision: Union[str, None] = "0039" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "series_chapter", - sa.Column("id", sa.Integer, primary_key=True), - sa.Column( - "series_tag_id", - sa.Integer, - sa.ForeignKey("tag.id", ondelete="CASCADE"), - nullable=False, - ), - sa.Column("chapter_number", sa.Integer, nullable=False), - sa.Column("title", sa.Text, nullable=True), - sa.Column( - "is_placeholder", sa.Boolean, nullable=False, server_default="false" - ), - sa.Column("stated_page_start", sa.Integer, nullable=True), - sa.Column("stated_page_end", sa.Integer, nullable=True), - sa.Column( - "created_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("now()"), - ), - sa.Column( - "updated_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("now()"), - ), - ) - op.create_index( - "ix_series_chapter_series_tag_id", "series_chapter", ["series_tag_id"] - ) - - # New columns on series_page; chapter_id starts nullable so we can backfill. - op.add_column( - "series_page", sa.Column("chapter_id", sa.Integer, nullable=True) - ) - op.add_column( - "series_page", sa.Column("stated_page", sa.Integer, nullable=True) - ) - - conn = op.get_bind() - # One auto-chapter per existing series (any series_tag_id present in pages). - conn.execute( - sa.text( - "INSERT INTO series_chapter " - "(series_tag_id, chapter_number, is_placeholder, created_at, updated_at) " - "SELECT DISTINCT series_tag_id, 1, false, now(), now() " - "FROM series_page" - ) - ) - # Point every existing page at its series' auto-chapter. - conn.execute( - sa.text( - "UPDATE series_page sp " - "SET chapter_id = sc.id " - "FROM series_chapter sc " - "WHERE sc.series_tag_id = sp.series_tag_id" - ) - ) - - # Now lock chapter_id down: NOT NULL + FK (cascade) + index. - op.alter_column("series_page", "chapter_id", nullable=False) - op.create_foreign_key( - "fk_series_page_chapter_id", - "series_page", - "series_chapter", - ["chapter_id"], - ["id"], - ondelete="CASCADE", - ) - op.create_index( - "ix_series_page_chapter_id", "series_page", ["chapter_id"] - ) - - -def downgrade() -> None: - op.drop_index("ix_series_page_chapter_id", table_name="series_page") - op.drop_constraint( - "fk_series_page_chapter_id", "series_page", type_="foreignkey" - ) - op.drop_column("series_page", "stated_page") - op.drop_column("series_page", "chapter_id") - op.drop_index("ix_series_chapter_series_tag_id", table_name="series_chapter") - op.drop_table("series_chapter") diff --git a/alembic/versions/0041_series_suggestions.py b/alembic/versions/0041_series_suggestions.py deleted file mode 100644 index 51b690a..0000000 --- a/alembic/versions/0041_series_suggestions.py +++ /dev/null @@ -1,98 +0,0 @@ -"""series suggestions: assisted-continuation matcher (FC-6.3) - -Revision ID: 0041 -Revises: 0040 -Create Date: 2026-06-07 - -A confirm-only queue of "this post may continue this series" hints, plus two -import_settings knobs (enable + score threshold) for the matcher. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0041" -down_revision: Union[str, None] = "0040" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "series_suggestion", - sa.Column("id", sa.Integer, primary_key=True), - sa.Column( - "post_id", - sa.Integer, - sa.ForeignKey("post.id", ondelete="CASCADE"), - nullable=False, - ), - sa.Column( - "series_tag_id", - sa.Integer, - sa.ForeignKey("tag.id", ondelete="CASCADE"), - nullable=False, - ), - sa.Column("score", sa.Float, nullable=False), - sa.Column("signals", sa.JSON, nullable=True), - sa.Column( - "status", sa.String(16), nullable=False, server_default="pending" - ), - sa.Column( - "created_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("now()"), - ), - sa.Column( - "updated_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("now()"), - ), - sa.UniqueConstraint( - "post_id", "series_tag_id", name="uq_series_suggestion_post_series" - ), - ) - op.create_index( - "ix_series_suggestion_post_id", "series_suggestion", ["post_id"] - ) - op.create_index( - "ix_series_suggestion_series_tag_id", - "series_suggestion", - ["series_tag_id"], - ) - op.create_index( - "ix_series_suggestion_status", "series_suggestion", ["status"] - ) - - op.add_column( - "import_settings", - sa.Column( - "series_suggest_enabled", - sa.Boolean, - nullable=False, - server_default=sa.true(), - ), - ) - op.add_column( - "import_settings", - sa.Column( - "series_suggest_threshold", - sa.Float, - nullable=False, - server_default="0.5", - ), - ) - - -def downgrade() -> None: - op.drop_column("import_settings", "series_suggest_threshold") - op.drop_column("import_settings", "series_suggest_enabled") - op.drop_index("ix_series_suggestion_status", table_name="series_suggestion") - op.drop_index( - "ix_series_suggestion_series_tag_id", table_name="series_suggestion" - ) - op.drop_index("ix_series_suggestion_post_id", table_name="series_suggestion") - op.drop_table("series_suggestion") diff --git a/alembic/versions/0042_series_chapter_stated_part.py b/alembic/versions/0042_series_chapter_stated_part.py deleted file mode 100644 index f898e56..0000000 --- a/alembic/versions/0042_series_chapter_stated_part.py +++ /dev/null @@ -1,32 +0,0 @@ -"""series chapter stated_part: operator-facing Part N label (FC-6.4) - -Revision ID: 0042 -Revises: 0041 -Create Date: 2026-06-07 - -A chapter's positional chapter_number is auto-managed (rewritten 1..N on -reorder/delete), so it can't double as the installment number the operator wants -to type (e.g. a series authored from a post that is Part 2). Add a nullable -stated_part alongside it — the same split as series_page.page_number (order) vs -series_page.stated_page (printed number). Nullable; the UI falls back to -chapter_number when unset. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0042" -down_revision: Union[str, None] = "0041" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "series_chapter", sa.Column("stated_part", sa.Integer, nullable=True) - ) - - -def downgrade() -> None: - op.drop_column("series_chapter", "stated_part") diff --git a/alembic/versions/0043_post_attachment_per_post_unique.py b/alembic/versions/0043_post_attachment_per_post_unique.py deleted file mode 100644 index e8e38ce..0000000 --- a/alembic/versions/0043_post_attachment_per_post_unique.py +++ /dev/null @@ -1,62 +0,0 @@ -"""post_attachment: per-post sha uniqueness (empty-post flood fix) - -Revision ID: 0043 -Revises: 0042 -Create Date: 2026-06-08 - -PostAttachment.sha256 was GLOBALLY unique, so a non-art file the creator attaches -to many posts (a standard pdf/zip/link-card) only ever got ONE row — on the first -post — leaving every later post a bare shell (no image, no attachment). The native -Patreon backfill of Anduo surfaced 1589 such shells (operator-flagged 2026-06-08). - -Switch to PER-POST uniqueness: the on-disk blob stays sha-deduped, but each post -gets its own row. Replace the unique sha256 index with a plain lookup index plus -two partial uniques — (post_id, sha256) for real posts and (sha256) for the -NULL-post filesystem case (still one row per file there). - -Existing data has ≤1 row per sha (the old global unique), so the new partial -uniques can't be violated on upgrade — no data backfill needed here. The bare-post -shells themselves are removed by the separate prune-empty-posts cleanup tool. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0043" -down_revision: Union[str, None] = "0042" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # Drop the global unique index; recreate it as a plain (non-unique) lookup - # index so sha-based reads keep their index (matches the model's index=True). - op.drop_index("ix_post_attachment_sha256", table_name="post_attachment") - op.create_index( - "ix_post_attachment_sha256", "post_attachment", ["sha256"], - ) - op.create_index( - "uq_post_attachment_post_sha", "post_attachment", - ["post_id", "sha256"], unique=True, - postgresql_where=sa.text("post_id IS NOT NULL"), - ) - op.create_index( - "uq_post_attachment_null_post_sha", "post_attachment", - ["sha256"], unique=True, - postgresql_where=sa.text("post_id IS NULL"), - ) - - -def downgrade() -> None: - op.drop_index( - "uq_post_attachment_null_post_sha", table_name="post_attachment" - ) - op.drop_index( - "uq_post_attachment_post_sha", table_name="post_attachment" - ) - op.drop_index("ix_post_attachment_sha256", table_name="post_attachment") - op.create_index( - "ix_post_attachment_sha256", "post_attachment", ["sha256"], - unique=True, - ) diff --git a/alembic/versions/0044_ml_settings_tagger_store_floor.py b/alembic/versions/0044_ml_settings_tagger_store_floor.py deleted file mode 100644 index e019e36..0000000 --- a/alembic/versions/0044_ml_settings_tagger_store_floor.py +++ /dev/null @@ -1,37 +0,0 @@ -"""ml_settings.tagger_store_floor - -The ingest confidence floor below which tagger predictions are not stored, -promoted from the TAGGER_STORE_FLOOR env var to a DB-backed, UI-tunable -setting. Default 0.70 (was an env default of 0.05): the suggestion path -already filters at 0.70 and the centroid/learned path covers low-confidence -preferred tags, so the sub-0.70 tail was redundant weight — it had grown -image_record's TOAST to ~100 GB. See plan-task #764. - -Revision ID: 0044 -Revises: 0043 -Create Date: 2026-06-10 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0044" -down_revision: Union[str, None] = "0043" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "tagger_store_floor", sa.Float(), - nullable=False, server_default="0.7", - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "tagger_store_floor") diff --git a/alembic/versions/0045_image_prediction_table.py b/alembic/versions/0045_image_prediction_table.py deleted file mode 100644 index df11de2..0000000 --- a/alembic/versions/0045_image_prediction_table.py +++ /dev/null @@ -1,69 +0,0 @@ -"""image_prediction table (DDL only — backfill runs as a background task) - -Normalizes the per-image tagger predictions out of the JSON blob into a -queryable table (#768). This migration creates ONLY the table + indexes — it -is pure DDL and commits instantly, so web boots immediately. - -The data backfill from the existing image_record.tagger_predictions JSON is -deliberately NOT done here. Doing it inline made the whole migration one -transaction over the ~100 GB TOAST: nothing committed until the very end, it -was invisible/unmonitorable mid-run, and an early MATERIALIZED-CTE form spilled -the full 100 GB to temp. Instead the backfill is the -backend.app.tasks.admin.backfill_image_predictions_task — batched by id window, -committed per chunk (visible progress + resumable), idempotent -(ON CONFLICT DO NOTHING). Trigger it from Settings → Maintenance once web is up. - -The old image_record.tagger_predictions column is left in place (vestigial) and -dropped in a follow-up once the backfill + code cutover are verified — dropping -it needs an ACCESS EXCLUSIVE lock on the hot image_record table (the 0044 lock -class), so it's deferred to a quiesced-worker window. - -Revision ID: 0045 -Revises: 0044 -Create Date: 2026-06-10 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0045" -down_revision: Union[str, None] = "0044" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "image_prediction", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column( - "image_record_id", sa.Integer(), - sa.ForeignKey("image_record.id", ondelete="CASCADE"), - nullable=False, - ), - sa.Column("raw_name", sa.String(length=255), nullable=False), - sa.Column("category", sa.String(length=64), nullable=False), - sa.Column("score", sa.Float(), nullable=False), - sa.UniqueConstraint( - "image_record_id", "raw_name", name="image_raw_name", - ), - ) - op.create_index( - "ix_image_prediction_image", "image_prediction", ["image_record_id"], - ) - op.create_index( - "ix_image_prediction_name_score", "image_prediction", - ["raw_name", "score"], - ) - # No data backfill here — see the module docstring. The one-time copy from - # image_record.tagger_predictions runs as backfill_image_predictions_task - # (batched, resumable, idempotent), kept out of this transaction so web boots - # without waiting on a ~100 GB pass. - - -def downgrade() -> None: - op.drop_index("ix_image_prediction_name_score", "image_prediction") - op.drop_index("ix_image_prediction_image", "image_prediction") - op.drop_table("image_prediction") diff --git a/alembic/versions/0046_drop_tagger_predictions.py b/alembic/versions/0046_drop_tagger_predictions.py deleted file mode 100644 index 84e543a..0000000 --- a/alembic/versions/0046_drop_tagger_predictions.py +++ /dev/null @@ -1,43 +0,0 @@ -"""drop image_record.tagger_predictions (predictions normalized to image_prediction) - -Final step of #768. The per-tag predictions now live in the image_prediction -table (backfilled from the JSON, read by suggestions + allowlist, written by -tag_and_embed). The old JSON column is dead weight — and it's the ~100 GB of -sub-0.70 score tail that bloated image_record's TOAST and broke DB backups -(#739). Dropping it is a fast catalog change; it does NOT reclaim the disk on -its own — run `VACUUM FULL image_record` (or pg_repack) afterward, off-hours, -to return the space to the OS so backups go small. - -DROP COLUMN needs a brief ACCESS EXCLUSIVE lock on image_record; env.py's -lock_timeout guards it, so quiesce the ml-worker if a tagging run is in flight -(see the migration-lock reference). tagger_model_version is kept — it's the -"has this been tagged / is it current?" signal the backfill sweep reads. - -Revision ID: 0046 -Revises: 0045 -Create Date: 2026-06-11 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0046" -down_revision: Union[str, None] = "0045" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.drop_column("image_record", "tagger_predictions") - - -def downgrade() -> None: - # Re-add the column empty. The JSON data is not restored (it lived only in - # this column); a downgrade would re-tag or backfill from image_prediction - # separately if ever needed. - op.add_column( - "image_record", - sa.Column("tagger_predictions", sa.JSON(), nullable=True), - ) diff --git a/alembic/versions/0047_series_chapter_dividers.py b/alembic/versions/0047_series_chapter_dividers.py deleted file mode 100644 index 5afd074..0000000 --- a/alembic/versions/0047_series_chapter_dividers.py +++ /dev/null @@ -1,175 +0,0 @@ -"""series chapters become cosmetic dividers; pages become one series-global run - -FC-6.x reframe (#789). A series is now ONE flat, series-global ordered run of -pages; chapters stop owning pages and become labeled dividers anchored to the -page that begins them. - -Migration (order matters — series_page.chapter_id cascades, so it must be -dropped BEFORE any chapter row is deleted, or pages would cascade away): - a. Renumber series_page.page_number to a series-global 1..N (ordered by the - OLD (chapter_number, page_number)). - b. Add series_chapter.anchor_page_id and populate it with each chapter's first - page (lowest new page_number). - c. Drop series_page.chapter_id (severs the cascade link). - d. Prune chapters that shouldn't become dividers: empty/placeholder ones (no - anchor) and the redundant unlabeled chapter that would sit at page 1. - e. Reshape series_chapter into the divider: drop chapter_number, - is_placeholder, stated_page_start/end; make anchor_page_id NOT NULL + - UNIQUE + FK→series_page ON DELETE CASCADE. - -Revision ID: 0047 -Revises: 0046 -Create Date: 2026-06-11 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0047" -down_revision: Union[str, None] = "0046" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # a. series-global page numbering, preserving the old reading order. - op.execute( - """ - WITH ordered AS ( - SELECT sp.id, - ROW_NUMBER() OVER ( - PARTITION BY sp.series_tag_id - ORDER BY sc.chapter_number, sp.page_number, sp.id - ) AS rn - FROM series_page sp - JOIN series_chapter sc ON sc.id = sp.chapter_id - ) - UPDATE series_page sp - SET page_number = ordered.rn - FROM ordered - WHERE sp.id = ordered.id - """ - ) - - # b. anchor each existing chapter at its first page (lowest new page_number). - op.add_column( - "series_chapter", - sa.Column("anchor_page_id", sa.Integer(), nullable=True), - ) - op.execute( - """ - WITH firsts AS ( - SELECT DISTINCT ON (sp.chapter_id) - sp.chapter_id, sp.id AS page_id - FROM series_page sp - ORDER BY sp.chapter_id, sp.page_number, sp.id - ) - UPDATE series_chapter sc - SET anchor_page_id = firsts.page_id - FROM firsts - WHERE firsts.chapter_id = sc.id - """ - ) - - # c. sever the ownership link (drops the FK + index with the column) BEFORE - # pruning chapters, so deleting a chapter can't cascade-delete its pages. - op.drop_column("series_page", "chapter_id") - - # d. prune chapters that don't become dividers: placeholders / empty ones - # (no anchor), and the unlabeled chapter that would land redundantly at - # page 1 (the series just starts — no divider needed there). - op.execute( - """ - DELETE FROM series_chapter sc - USING ( - SELECT sc2.id - FROM series_chapter sc2 - LEFT JOIN series_page sp ON sp.id = sc2.anchor_page_id - WHERE sc2.anchor_page_id IS NULL - OR (sp.page_number = 1 - AND sc2.title IS NULL - AND sc2.stated_part IS NULL) - ) gone - WHERE sc.id = gone.id - """ - ) - - # e. reshape into the divider model. - op.drop_column("series_chapter", "chapter_number") - op.drop_column("series_chapter", "is_placeholder") - op.drop_column("series_chapter", "stated_page_start") - op.drop_column("series_chapter", "stated_page_end") - op.alter_column("series_chapter", "anchor_page_id", nullable=False) - op.create_unique_constraint( - "uq_series_chapter_anchor_page", "series_chapter", ["anchor_page_id"] - ) - op.create_foreign_key( - "fk_series_chapter_anchor_page", - "series_chapter", - "series_page", - ["anchor_page_id"], - ["id"], - ondelete="CASCADE", - ) - - -def downgrade() -> None: - # Lossy: dividers can't be reconstructed as owning chapters. Collapse back to - # exactly one chapter per series that owns all its pages in order. - op.add_column( - "series_page", sa.Column("chapter_id", sa.Integer(), nullable=True) - ) - op.drop_constraint( - "fk_series_chapter_anchor_page", "series_chapter", type_="foreignkey" - ) - op.drop_constraint( - "uq_series_chapter_anchor_page", "series_chapter", type_="unique" - ) - op.drop_column("series_chapter", "anchor_page_id") - op.add_column( - "series_chapter", - sa.Column( - "chapter_number", sa.Integer(), nullable=False, server_default="1" - ), - ) - op.add_column( - "series_chapter", - sa.Column( - "is_placeholder", sa.Boolean(), nullable=False, - server_default="false", - ), - ) - op.add_column( - "series_chapter", - sa.Column("stated_page_start", sa.Integer(), nullable=True), - ) - op.add_column( - "series_chapter", - sa.Column("stated_page_end", sa.Integer(), nullable=True), - ) - op.execute("DELETE FROM series_chapter") - op.execute( - """ - INSERT INTO series_chapter (series_tag_id, chapter_number) - SELECT DISTINCT series_tag_id, 1 FROM series_page - """ - ) - op.execute( - """ - UPDATE series_page sp - SET chapter_id = sc.id - FROM series_chapter sc - WHERE sc.series_tag_id = sp.series_tag_id - """ - ) - op.alter_column("series_page", "chapter_id", nullable=False) - op.create_foreign_key( - "fk_series_page_chapter", - "series_page", - "series_chapter", - ["chapter_id"], - ["id"], - ondelete="CASCADE", - ) diff --git a/alembic/versions/0048_series_page_pending_status.py b/alembic/versions/0048_series_page_pending_status.py deleted file mode 100644 index 25944a5..0000000 --- a/alembic/versions/0048_series_page_pending_status.py +++ /dev/null @@ -1,45 +0,0 @@ -"""series_page pending staging: status + nullable page_number (#789 Phase 2) - -Pages added from a post no longer append straight into the run — they land -'pending' with a NULL page_number, staged grouped by their source post so the -operator can drop junk (text-free alts, bumpers) and place the keepers into the -sequence. A page only gets a series-global page_number once it's 'placed'. - -Revision ID: 0048 -Revises: 0047 -Create Date: 2026-06-11 - -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0048" -down_revision: Union[str, None] = "0047" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "series_page", - sa.Column( - "status", sa.String(length=16), nullable=False, - server_default="placed", - ), - ) - op.alter_column( - "series_page", "page_number", - existing_type=sa.Integer(), nullable=True, - ) - - -def downgrade() -> None: - # Lossy: pending pages are unsorted staging rows with no order — drop them. - op.execute("DELETE FROM series_page WHERE status = 'pending'") - op.alter_column( - "series_page", "page_number", - existing_type=sa.Integer(), nullable=False, - ) - op.drop_column("series_page", "status") diff --git a/alembic/versions/0049_external_link_table.py b/alembic/versions/0049_external_link_table.py deleted file mode 100644 index 373c807..0000000 --- a/alembic/versions/0049_external_link_table.py +++ /dev/null @@ -1,90 +0,0 @@ -"""external_link table — off-platform file-host links found in post bodies - -Creators host the real files on mega.nz / Google Drive / MediaFire / Dropbox / -Pixeldrain and link them in the post text. This table records each such link -(so nothing is silently dropped), and doubles as the dedup + dead-letter ledger -the download worker (a later slice) walks. `url` keeps the FULL link including -the `#fragment` — mega.nz's decryption key lives there; truncating it makes the -file undownloadable. - -CHECK whitelists for host + status include the full enum up front (incl. the -download-worker statuses) so the worker slice needs no constraint migration. - -Revision ID: 0049 -Revises: 0048 -Create Date: 2026-06-14 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0049" -down_revision: Union[str, None] = "0048" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "external_link", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column( - "post_id", sa.Integer(), - sa.ForeignKey("post.id", ondelete="CASCADE"), nullable=False, - ), - sa.Column( - "artist_id", sa.Integer(), - sa.ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, - ), - sa.Column("host", sa.String(length=16), nullable=False), - sa.Column("url", sa.Text(), nullable=False), - sa.Column("label", sa.Text(), nullable=True), - sa.Column( - "status", sa.String(length=16), nullable=False, - server_default="pending", - ), - sa.Column("attempts", sa.Integer(), nullable=False, server_default="0"), - sa.Column("last_error", sa.Text(), nullable=True), - sa.Column( - "attachment_id", sa.Integer(), - sa.ForeignKey("post_attachment.id", ondelete="SET NULL"), - nullable=True, - ), - sa.Column( - "created_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column("completed_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("duration_seconds", sa.Float(), nullable=True), - sa.CheckConstraint( - "host IN ('mega','gdrive','mediafire','dropbox','pixeldrain')", - name="ck_external_link_host", - ), - sa.CheckConstraint( - "status IN ('pending','downloading','downloaded','failed'," - "'skipped','dead')", - name="ck_external_link_status", - ), - ) - op.create_index( - "ix_external_link_post_id", "external_link", ["post_id"], - ) - op.create_index( - "ix_external_link_artist_id", "external_link", ["artist_id"], - ) - op.create_index( - "ix_external_link_status", "external_link", ["status"], - ) - op.create_index( - "uq_external_link_post_url", "external_link", ["post_id", "url"], - unique=True, - ) - - -def downgrade() -> None: - op.drop_index("uq_external_link_post_url", table_name="external_link") - op.drop_index("ix_external_link_status", table_name="external_link") - op.drop_index("ix_external_link_artist_id", table_name="external_link") - op.drop_index("ix_external_link_post_id", table_name="external_link") - op.drop_table("external_link") diff --git a/alembic/versions/0050_external_link_host_toggles.py b/alembic/versions/0050_external_link_host_toggles.py deleted file mode 100644 index ac78e75..0000000 --- a/alembic/versions/0050_external_link_host_toggles.py +++ /dev/null @@ -1,38 +0,0 @@ -"""import_settings: per-host enable toggles for external file-host downloads - -Operator levers (#830): disable a single host (e.g. mega.nz when it's -rate-limiting/banning) without touching the others. The worker reads these via -getattr and defaults to enabled, so the toggles default TRUE (works out of the -box, rule #26). - -Revision ID: 0050 -Revises: 0049 -Create Date: 2026-06-14 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0050" -down_revision: Union[str, None] = "0049" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - -_HOSTS = ("mega", "gdrive", "mediafire", "dropbox", "pixeldrain") - - -def upgrade() -> None: - for host in _HOSTS: - op.add_column( - "import_settings", - sa.Column( - f"extdl_{host}_enabled", sa.Boolean(), nullable=False, - server_default=sa.true(), - ), - ) - - -def downgrade() -> None: - for host in _HOSTS: - op.drop_column("import_settings", f"extdl_{host}_enabled") diff --git a/alembic/versions/0051_image_source_provenance.py b/alembic/versions/0051_image_source_provenance.py deleted file mode 100644 index 595077d..0000000 --- a/alembic/versions/0051_image_source_provenance.py +++ /dev/null @@ -1,38 +0,0 @@ -"""image_record: source_url + source_filehash (inline-image localization) - -#830 Phase 2. To render a post body faithfully we serve LOCAL copies of inline -images instead of hotlinking the public CDN. The join key between a body -`` and the local file is the CDN's 32-hex filehash (the same -identity extract_media dedups by). Persist it (indexed) plus the full source -URL for provenance/debugging. Both NULL for filesystem-imported / pre-existing -rows — those fall back to hotlinking until re-downloaded. - -Revision ID: 0051 -Revises: 0050 -Create Date: 2026-06-14 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0051" -down_revision: Union[str, None] = "0050" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column("image_record", sa.Column("source_url", sa.Text(), nullable=True)) - op.add_column( - "image_record", sa.Column("source_filehash", sa.String(length=32), nullable=True) - ) - op.create_index( - "ix_image_record_source_filehash", "image_record", ["source_filehash"] - ) - - -def downgrade() -> None: - op.drop_index("ix_image_record_source_filehash", table_name="image_record") - op.drop_column("image_record", "source_filehash") - op.drop_column("image_record", "source_url") diff --git a/alembic/versions/0052_image_duration_seconds.py b/alembic/versions/0052_image_duration_seconds.py deleted file mode 100644 index ec2a180..0000000 --- a/alembic/versions/0052_image_duration_seconds.py +++ /dev/null @@ -1,32 +0,0 @@ -"""image_record: duration_seconds (Tier-1 video near-dup key) - -#871. Videos previously deduped on sha256 only (pHash is images-only), so a -different encode/remux of the same video imported as a distinct record. Persist -the container duration so the importer can treat same-artist videos with matching -duration (+ aspect ratio) as the same content and dedup/supersede like images. -NULL for images and for video rows imported before this column existed (a -backfill re-probes those so they participate in dedup). - -Revision ID: 0052 -Revises: 0051 -Create Date: 2026-06-16 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0052" -down_revision: Union[str, None] = "0051" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "image_record", sa.Column("duration_seconds", sa.Float(), nullable=True) - ) - - -def downgrade() -> None: - op.drop_column("image_record", "duration_seconds") diff --git a/alembic/versions/0053_ml_settings_video_tagging.py b/alembic/versions/0053_ml_settings_video_tagging.py deleted file mode 100644 index 1f192a4..0000000 --- a/alembic/versions/0053_ml_settings_video_tagging.py +++ /dev/null @@ -1,49 +0,0 @@ -"""ml_settings: video tagging knobs (cadence sampling + noise floor) - -#747. Video tag quality/perf: sample frames at a fixed cadence (interval) so a -tag's frame-presence reflects real screen time, cap total frames so long videos -stay bounded, and keep a tag only if it appears in >= min_tag_frames sampled -frames. Operator-tunable via Settings → ML (replaces the VIDEO_ML_FRAMES env var). - -Revision ID: 0053 -Revises: 0052 -Create Date: 2026-06-16 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0053" -down_revision: Union[str, None] = "0052" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "video_frame_interval_seconds", sa.Float(), nullable=False, - server_default="4.0", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "video_max_frames", sa.Integer(), nullable=False, server_default="64", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "video_min_tag_frames", sa.Integer(), nullable=False, - server_default="3", - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "video_min_tag_frames") - op.drop_column("ml_settings", "video_max_frames") - op.drop_column("ml_settings", "video_frame_interval_seconds") diff --git a/alembic/versions/0054_subscribestar_ledgers.py b/alembic/versions/0054_subscribestar_ledgers.py deleted file mode 100644 index 59972ae..0000000 --- a/alembic/versions/0054_subscribestar_ledgers.py +++ /dev/null @@ -1,82 +0,0 @@ -"""subscribestar_seen_media + subscribestar_failed_media: per-source ledgers - -Revision ID: 0054 -Revises: 0053 -Create Date: 2026-06-17 - -SubscribeStar native ingester (phase 1 of the gallery-dl → native-core -migration). Mirrors the Patreon ledger tables (0037/0038): a seen-ledger so -routine walks skip already-ingested media (recovery bypasses it) and a -dead-letter ledger so persistently-failing media stops re-burning backfill -chunks. `filehash` is a CDN content hash when present, else a synthesized -``:`` key — hence String(128). UNIQUE (source_id, filehash) -is the upsert key on each. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0054" -down_revision: Union[str, None] = "0053" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "subscribestar_seen_media", - sa.Column("id", sa.Integer, primary_key=True), - sa.Column( - "source_id", - sa.Integer, - sa.ForeignKey("source.id", ondelete="CASCADE"), - nullable=False, - index=True, - ), - sa.Column("filehash", sa.String(128), nullable=False), - sa.Column("post_id", sa.String(64), nullable=True), - sa.Column( - "seen_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.UniqueConstraint( - "source_id", "filehash", name="uq_subscribestar_seen_media_source_id" - ), - ) - op.create_table( - "subscribestar_failed_media", - sa.Column("id", sa.Integer, primary_key=True), - sa.Column( - "source_id", - sa.Integer, - sa.ForeignKey("source.id", ondelete="CASCADE"), - nullable=False, - index=True, - ), - sa.Column("filehash", sa.String(128), nullable=False), - sa.Column("attempts", sa.Integer, nullable=False, server_default="1"), - sa.Column("last_error", sa.Text, nullable=True), - sa.Column( - "first_failed_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.Column( - "last_failed_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.UniqueConstraint( - "source_id", "filehash", name="uq_subscribestar_failed_media_source_id" - ), - ) - - -def downgrade() -> None: - op.drop_table("subscribestar_failed_media") - op.drop_table("subscribestar_seen_media") diff --git a/alembic/versions/0055_image_provenance_from_attachment.py b/alembic/versions/0055_image_provenance_from_attachment.py deleted file mode 100644 index 8b2566b..0000000 --- a/alembic/versions/0055_image_provenance_from_attachment.py +++ /dev/null @@ -1,55 +0,0 @@ -"""image_provenance: from_attachment_id (which archive an image was extracted from) - -Milestone #87. When an image is pulled out of a .zip/.rar, record WHICH archive -PostAttachment it came from, so the provenance UI can show the single archive a -file lives inside instead of every attachment on the post. Nullable FK with -ON DELETE SET NULL — a loose (non-archive) download leaves it NULL, and deleting -the archive attachment forgets the linkage without destroying the (image, post) -provenance edge. Existing rows are NULL until the reextract backfill stamps them. - -Revision ID: 0055 -Revises: 0054 -Create Date: 2026-06-22 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0055" -down_revision: Union[str, None] = "0054" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "image_provenance", - sa.Column("from_attachment_id", sa.Integer(), nullable=True), - ) - op.create_index( - "ix_image_provenance_from_attachment_id", - "image_provenance", - ["from_attachment_id"], - ) - op.create_foreign_key( - "fk_image_provenance_from_attachment", - "image_provenance", - "post_attachment", - ["from_attachment_id"], - ["id"], - ondelete="SET NULL", - ) - - -def downgrade() -> None: - op.drop_constraint( - "fk_image_provenance_from_attachment", - "image_provenance", - type_="foreignkey", - ) - op.drop_index( - "ix_image_provenance_from_attachment_id", - table_name="image_provenance", - ) - op.drop_column("image_provenance", "from_attachment_id") diff --git a/alembic/versions/0056_tag_eval_run.py b/alembic/versions/0056_tag_eval_run.py deleted file mode 100644 index 7d8e91f..0000000 --- a/alembic/versions/0056_tag_eval_run.py +++ /dev/null @@ -1,43 +0,0 @@ -"""tag_eval_run: persisted head-vs-centroid tagging eval runs (#1130) - -Milestone #114 slice 1. A long ml-queue eval whose full report must SURVIVE -navigation, so the run + report live in a row the admin card rehydrates from -(mirrors library_audit_run). running -> ready / error. - -Revision ID: 0056 -Revises: 0055 -Create Date: 2026-06-28 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from sqlalchemy.dialects.postgresql import JSONB - -revision: str = "0056" -down_revision: Union[str, None] = "0055" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "tag_eval_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("params", JSONB(), nullable=False), - sa.Column("status", sa.String(length=16), nullable=False, server_default="running"), - sa.Column( - "started_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("report", JSONB(), nullable=True), - sa.Column("error", sa.Text(), nullable=True), - sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True), - ) - op.create_index("ix_tag_eval_run_status", "tag_eval_run", ["status"]) - - -def downgrade() -> None: - op.drop_index("ix_tag_eval_run_status", table_name="tag_eval_run") - op.drop_table("tag_eval_run") diff --git a/alembic/versions/0057_tag_positive_confirmation.py b/alembic/versions/0057_tag_positive_confirmation.py deleted file mode 100644 index 92335c2..0000000 --- a/alembic/versions/0057_tag_positive_confirmation.py +++ /dev/null @@ -1,40 +0,0 @@ -"""tag_positive_confirmation: operator-affirmed correct positives (#1130) - -Mirror of tag_suggestion_rejection. "Keep" on a doubted positive records here so -the eval's doubts list stops resurfacing confirmed-correct images every run. - -Revision ID: 0057 -Revises: 0056 -Create Date: 2026-06-28 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0057" -down_revision: Union[str, None] = "0056" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "tag_positive_confirmation", - sa.Column( - "image_record_id", sa.Integer(), - sa.ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True, - ), - sa.Column( - "tag_id", sa.Integer(), - sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, index=True, - ), - sa.Column( - "confirmed_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - ) - - -def downgrade() -> None: - op.drop_table("tag_positive_confirmation") diff --git a/alembic/versions/0058_tag_head.py b/alembic/versions/0058_tag_head.py deleted file mode 100644 index 7ff45f6..0000000 --- a/alembic/versions/0058_tag_head.py +++ /dev/null @@ -1,95 +0,0 @@ -"""tag_head + head_training_run: production heads that learn from tags (#114) - -The eval (#1130) proved the frozen-embedding + trained-head spine; this lands its -production form. tag_head stores one logistic-regression head per concept (the -new suggestion source, replacing Camie + centroid); head_training_run tracks the -batch that (re)trains them. Adds two head-training tunables to ml_settings. - -Revision ID: 0058 -Revises: 0057 -Create Date: 2026-06-28 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from pgvector.sqlalchemy import Vector -from sqlalchemy.dialects.postgresql import JSONB - -revision: str = "0058" -down_revision: Union[str, None] = "0057" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - -_HEAD_DIM = 1152 - - -def upgrade() -> None: - op.create_table( - "tag_head", - sa.Column( - "tag_id", sa.Integer(), - sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, - ), - sa.Column("embedding_version", sa.String(length=128), nullable=False), - sa.Column("weights", Vector(_HEAD_DIM), nullable=False), - sa.Column("bias", sa.Float(), nullable=False), - sa.Column("suggest_threshold", sa.Float(), nullable=False), - sa.Column("auto_apply_threshold", sa.Float(), nullable=True), - sa.Column("n_pos", sa.Integer(), nullable=False), - sa.Column("n_neg", sa.Integer(), nullable=False), - sa.Column("ap", sa.Float(), nullable=False), - sa.Column("precision_cv", sa.Float(), nullable=False), - sa.Column("recall", sa.Float(), nullable=False), - sa.Column( - "trained_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column("metrics", JSONB(), nullable=True), - ) - - op.create_table( - "head_training_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("params", JSONB(), nullable=False), - sa.Column( - "status", sa.String(length=16), nullable=False, - server_default="running", - ), - sa.Column( - "started_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("n_trained", sa.Integer(), nullable=True), - sa.Column("n_skipped", sa.Integer(), nullable=True), - sa.Column("error", sa.Text(), nullable=True), - sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True), - ) - op.create_index( - "ix_head_training_run_status", "head_training_run", ["status"], - ) - - # Head-training tunables on the ml_settings singleton. - op.add_column( - "ml_settings", - sa.Column( - "head_min_positives", sa.Integer(), nullable=False, - server_default="8", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "head_auto_apply_precision", sa.Float(), nullable=False, - server_default="0.97", - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "head_auto_apply_precision") - op.drop_column("ml_settings", "head_min_positives") - op.drop_index("ix_head_training_run_status", table_name="head_training_run") - op.drop_table("head_training_run") - op.drop_table("tag_head") diff --git a/alembic/versions/0059_head_auto_apply.py b/alembic/versions/0059_head_auto_apply.py deleted file mode 100644 index d0bb9b8..0000000 --- a/alembic/versions/0059_head_auto_apply.py +++ /dev/null @@ -1,70 +0,0 @@ -"""head_auto_apply_run + earned-auto-apply settings (#114) - -A graduated head can apply its tag without a human, gated by a master switch + -a support floor. head_auto_apply_run tracks each sweep / dry-run preview. - -Revision ID: 0059 -Revises: 0058 -Create Date: 2026-06-29 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from sqlalchemy.dialects.postgresql import JSONB - -revision: str = "0059" -down_revision: Union[str, None] = "0058" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "head_auto_apply_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column( - "dry_run", sa.Boolean(), nullable=False, server_default=sa.false() - ), - sa.Column("params", JSONB(), nullable=False), - sa.Column( - "status", sa.String(length=16), nullable=False, - server_default="running", - ), - sa.Column( - "started_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("n_applied", sa.Integer(), nullable=True), - sa.Column("report", JSONB(), nullable=True), - sa.Column("error", sa.Text(), nullable=True), - sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True), - ) - op.create_index( - "ix_head_auto_apply_run_status", "head_auto_apply_run", ["status"], - ) - - op.add_column( - "ml_settings", - sa.Column( - "head_auto_apply_enabled", sa.Boolean(), nullable=False, - server_default=sa.true(), # opt-out: on by default (operator-asked) - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "head_auto_apply_min_positives", sa.Integer(), nullable=False, - server_default="30", - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "head_auto_apply_min_positives") - op.drop_column("ml_settings", "head_auto_apply_enabled") - op.drop_index( - "ix_head_auto_apply_run_status", table_name="head_auto_apply_run" - ) - op.drop_table("head_auto_apply_run") diff --git a/alembic/versions/0060_head_metrics.py b/alembic/versions/0060_head_metrics.py deleted file mode 100644 index e94edb8..0000000 --- a/alembic/versions/0060_head_metrics.py +++ /dev/null @@ -1,74 +0,0 @@ -"""head_metric + head_metrics_snapshot: auto-apply observability (#114) - -Running misfire/under-fire counters per concept (captured at correction time, -since image_tag.source is lost on delete) + a daily per-concept time-series so -the operator can tune the precision target + support floor from real data. - -Revision ID: 0060 -Revises: 0059 -Create Date: 2026-06-29 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0060" -down_revision: Union[str, None] = "0059" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "head_metric", - sa.Column( - "tag_id", sa.Integer(), - sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, - ), - sa.Column("n_misfires", sa.Integer(), nullable=False, server_default="0"), - sa.Column("n_underfires", sa.Integer(), nullable=False, server_default="0"), - sa.Column( - "updated_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - ) - - op.create_table( - "head_metrics_snapshot", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column( - "tag_id", sa.Integer(), - sa.ForeignKey("tag.id", ondelete="CASCADE"), - ), - sa.Column("name", sa.String(length=255), nullable=False), - sa.Column( - "snapshot_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column("n_auto_applied", sa.Integer(), nullable=False, server_default="0"), - sa.Column("n_misfires", sa.Integer(), nullable=False, server_default="0"), - sa.Column("n_underfires", sa.Integer(), nullable=False, server_default="0"), - sa.Column("ap", sa.Float(), nullable=True), - sa.Column("precision_cv", sa.Float(), nullable=True), - sa.Column("recall", sa.Float(), nullable=True), - sa.Column("n_pos", sa.Integer(), nullable=True), - ) - op.create_index( - "ix_head_metrics_snapshot_tag_id", "head_metrics_snapshot", ["tag_id"], - ) - op.create_index( - "ix_head_metrics_snapshot_snapshot_at", "head_metrics_snapshot", - ["snapshot_at"], - ) - - -def downgrade() -> None: - op.drop_index( - "ix_head_metrics_snapshot_snapshot_at", table_name="head_metrics_snapshot" - ) - op.drop_index( - "ix_head_metrics_snapshot_tag_id", table_name="head_metrics_snapshot" - ) - op.drop_table("head_metrics_snapshot") - op.drop_table("head_metric") diff --git a/alembic/versions/0061_image_region.py b/alembic/versions/0061_image_region.py deleted file mode 100644 index b3af8a9..0000000 --- a/alembic/versions/0061_image_region.py +++ /dev/null @@ -1,59 +0,0 @@ -"""image_region: detected/proposed regions + their crop embeddings (#114) - -Storage backbone of the crop pipeline. A region = normalized bbox + the crop's -embedding (CCIP for face/figure → character id; SigLIP for concept regions → -head bag-of-embeddings). Also serves as grounded-tag bbox provenance. - -Revision ID: 0061 -Revises: 0060 -Create Date: 2026-06-29 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from pgvector.sqlalchemy import Vector - -revision: str = "0061" -down_revision: Union[str, None] = "0060" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - -_CCIP_DIM = 768 -_SIGLIP_DIM = 1152 - - -def upgrade() -> None: - op.create_table( - "image_region", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column( - "image_record_id", sa.Integer(), - sa.ForeignKey("image_record.id", ondelete="CASCADE"), nullable=False, - ), - sa.Column("kind", sa.String(length=16), nullable=False), - # Video/animated: source frame timestamp (seconds); NULL for stills. - sa.Column("frame_time", sa.Float(), nullable=True), - sa.Column("rx", sa.Float(), nullable=False), - sa.Column("ry", sa.Float(), nullable=False), - sa.Column("rw", sa.Float(), nullable=False), - sa.Column("rh", sa.Float(), nullable=False), - sa.Column("score", sa.Float(), nullable=True), - sa.Column("detector_version", sa.String(length=64), nullable=True), - sa.Column("crop_version", sa.String(length=64), nullable=True), - sa.Column("embedding_version", sa.String(length=128), nullable=True), - sa.Column("ccip_embedding", Vector(_CCIP_DIM), nullable=True), - sa.Column("siglip_embedding", Vector(_SIGLIP_DIM), nullable=True), - sa.Column( - "created_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - ) - op.create_index( - "ix_image_region_image_record_id", "image_region", ["image_record_id"], - ) - - -def downgrade() -> None: - op.drop_index("ix_image_region_image_record_id", table_name="image_region") - op.drop_table("image_region") diff --git a/alembic/versions/0062_gpu_job.py b/alembic/versions/0062_gpu_job.py deleted file mode 100644 index a044995..0000000 --- a/alembic/versions/0062_gpu_job.py +++ /dev/null @@ -1,55 +0,0 @@ -"""gpu_job: the HTTP-leased GPU work queue for the desktop agent (#114) - -The agent stays HTTP-only — the server enqueues per-(image, task) jobs here and -the agent leases/submits over the web API; Redis/Postgres stay private. - -Revision ID: 0062 -Revises: 0061 -Create Date: 2026-06-29 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0062" -down_revision: Union[str, None] = "0061" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "gpu_job", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column( - "image_record_id", sa.Integer(), - sa.ForeignKey("image_record.id", ondelete="CASCADE"), nullable=False, - ), - sa.Column("task", sa.String(length=32), nullable=False), - sa.Column( - "status", sa.String(length=16), nullable=False, - server_default="pending", - ), - sa.Column("lease_token", sa.String(length=64), nullable=True), - sa.Column("leased_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("lease_expires_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("attempts", sa.Integer(), nullable=False, server_default="0"), - sa.Column("error", sa.Text(), nullable=True), - sa.Column( - "created_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column( - "updated_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - ) - op.create_index("ix_gpu_job_image_record_id", "gpu_job", ["image_record_id"]) - op.create_index("ix_gpu_job_status", "gpu_job", ["status"]) - - -def downgrade() -> None: - op.drop_index("ix_gpu_job_status", table_name="gpu_job") - op.drop_index("ix_gpu_job_image_record_id", table_name="gpu_job") - op.drop_table("gpu_job") diff --git a/alembic/versions/0063_ccip_match_threshold.py b/alembic/versions/0063_ccip_match_threshold.py deleted file mode 100644 index d841398..0000000 --- a/alembic/versions/0063_ccip_match_threshold.py +++ /dev/null @@ -1,33 +0,0 @@ -"""ml_settings.ccip_match_threshold — tunable CCIP character-match cut (#114) - -The v1 matcher used a flat 0.75 cosine; live data showed that over-fires (a -high-reference character matched a scatter of images). 0.85 keeps the confident -single-character matches and drops the noise. Tunable from the GPU agent card. - -Revision ID: 0063 -Revises: 0062 -Create Date: 2026-06-29 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0063" -down_revision: Union[str, None] = "0062" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "ccip_match_threshold", sa.Float(), nullable=False, - server_default="0.85", - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "ccip_match_threshold") diff --git a/alembic/versions/0064_ccip_auto_apply.py b/alembic/versions/0064_ccip_auto_apply.py deleted file mode 100644 index e5323cf..0000000 --- a/alembic/versions/0064_ccip_auto_apply.py +++ /dev/null @@ -1,42 +0,0 @@ -"""ml_settings: CCIP auto-apply switch + threshold (#114) - -Confident CCIP character matches auto-tag (source='ccip_auto') on a daily sweep, -so identity tags keep flowing without pressing a button. ON by default (opt-out, -like head auto-apply); the high threshold (0.92, above the 0.85 suggest cut) + -single-character references keep it safe, and every auto-tag is reversible. - -Revision ID: 0064 -Revises: 0063 -Create Date: 2026-06-30 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0064" -down_revision: Union[str, None] = "0063" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "ccip_auto_apply_enabled", sa.Boolean(), nullable=False, - server_default=sa.true(), - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "ccip_auto_apply_threshold", sa.Float(), nullable=False, - server_default="0.92", - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "ccip_auto_apply_threshold") - op.drop_column("ml_settings", "ccip_auto_apply_enabled") diff --git a/alembic/versions/0065_embedder_model_name.py b/alembic/versions/0065_embedder_model_name.py deleted file mode 100644 index 0a986b3..0000000 --- a/alembic/versions/0065_embedder_model_name.py +++ /dev/null @@ -1,35 +0,0 @@ -"""ml_settings: embedder_model_name (#1190 operator model swap) - -The embedder MODEL VERSION was already a setting (and stamps image_record. -siglip_model_version); the HF model NAME was env-only, so an operator couldn't -actually point the pipeline at a different embedder. Storing the name as a -setting makes the model an operator choice: set name + version → re-embed (the -GPU agent) → retrain heads. Default = the current SigLIP so400m. - -Revision ID: 0065 -Revises: 0064 -Create Date: 2026-06-30 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0065" -down_revision: Union[str, None] = "0064" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "embedder_model_name", sa.String(length=128), nullable=False, - server_default="google/siglip-so400m-patch14-384", - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "embedder_model_name") diff --git a/alembic/versions/0066_drop_centroids.py b/alembic/versions/0066_drop_centroids.py deleted file mode 100644 index d75a334..0000000 --- a/alembic/versions/0066_drop_centroids.py +++ /dev/null @@ -1,57 +0,0 @@ -"""drop the dead per-tag centroid subsystem (#1189 cleanup) - -The v2 pivot replaced per-tag SigLIP centroids with learned heads + CCIP. -Nothing read the centroids anymore — they were recomputed (on merge + a daily -beat) but never consumed for suggestions or auto-apply. Remove the storage + -its two now-unused settings columns. (The recompute tasks, beat, endpoint, -service, and UI card are removed in the same change.) - -Revision ID: 0066 -Revises: 0065 -Create Date: 2026-06-30 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0066" -down_revision: Union[str, None] = "0065" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.drop_table("tag_reference_embedding") - op.drop_column("ml_settings", "centroid_similarity_threshold") - op.drop_column("ml_settings", "min_reference_images") - - -def downgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "min_reference_images", sa.Integer(), nullable=False, - server_default="5", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "centroid_similarity_threshold", sa.Float(), nullable=False, - server_default="0.55", - ), - ) - op.create_table( - "tag_reference_embedding", - sa.Column("tag_id", sa.Integer(), nullable=False), - sa.Column("embedding", sa.LargeBinary(), nullable=False), - sa.Column("reference_count", sa.Integer(), nullable=False), - sa.Column("model_version", sa.String(length=128), nullable=False), - sa.Column( - "updated_at", sa.DateTime(timezone=True), - server_default=sa.func.now(), nullable=False, - ), - sa.ForeignKeyConstraint(["tag_id"], ["tag.id"], ondelete="CASCADE"), - sa.PrimaryKeyConstraint("tag_id"), - ) diff --git a/alembic/versions/0067_retire_camie_allowlist.py b/alembic/versions/0067_retire_camie_allowlist.py deleted file mode 100644 index e3edd02..0000000 --- a/alembic/versions/0067_retire_camie_allowlist.py +++ /dev/null @@ -1,66 +0,0 @@ -"""retire the Camie tagger + allowlist bulk-apply (#1189) - -The v2 pivot made heads + CCIP the tag source and head auto-apply the earned -propagation. The Camie tagger ran only to feed the allowlist bulk-apply (its -predictions had no other consumer), and the allowlist was a second, un-earned -auto-apply path parallel to heads. Both are retired — drop their storage. - -(image_prediction = Camie's per-image predictions; tag_allowlist = the bulk- -apply allowlist. Nothing references INTO these tables, so the drop is clean.) - -Revision ID: 0067 -Revises: 0066 -Create Date: 2026-06-30 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0067" -down_revision: Union[str, None] = "0066" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.drop_table("image_prediction") - op.drop_table("tag_allowlist") - - -def downgrade() -> None: - op.create_table( - "tag_allowlist", - sa.Column("tag_id", sa.Integer(), nullable=False), - sa.Column( - "min_confidence", sa.Float(), nullable=False, server_default="0.9" - ), - sa.Column( - "created_at", sa.DateTime(timezone=True), - server_default=sa.func.now(), nullable=False, - ), - sa.ForeignKeyConstraint(["tag_id"], ["tag.id"], ondelete="CASCADE"), - sa.PrimaryKeyConstraint("tag_id"), - sa.CheckConstraint( - "min_confidence >= 0 AND min_confidence <= 1", - name="ck_tag_allowlist_confidence_range", - ), - ) - op.create_table( - "image_prediction", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("image_record_id", sa.Integer(), nullable=False), - sa.Column("raw_name", sa.String(length=255), nullable=False), - sa.Column("category", sa.String(length=32), nullable=False), - sa.Column("score", sa.Float(), nullable=False), - sa.ForeignKeyConstraint( - ["image_record_id"], ["image_record.id"], ondelete="CASCADE" - ), - ) - op.create_index( - "ix_image_prediction_image", "image_prediction", ["image_record_id"] - ) - op.create_index( - "ix_image_prediction_name_score", "image_prediction", - ["raw_name", "score"], - ) diff --git a/alembic/versions/0068_drop_dead_tagger_settings.py b/alembic/versions/0068_drop_dead_tagger_settings.py deleted file mode 100644 index 770676d..0000000 --- a/alembic/versions/0068_drop_dead_tagger_settings.py +++ /dev/null @@ -1,80 +0,0 @@ -"""drop dead tagger/suggestion settings + columns left after Camie retirement (#1199) - -Hygiene follow-up to #1189. These were left inert to bound that change; nothing -reads them now: -- ml_settings: tagger_store_floor + tagger_model_version (only the deleted Camie - tagger used them), suggestion_threshold_character/general (already dead pre- - retirement — scoring uses per-head thresholds), video_min_tag_frames (only the - deleted video-prediction aggregator used it). -- image_record: tagger_model_version (no writer now), centroid_scores (long-dead - JSON cache, no reader). - -Revision ID: 0068 -Revises: 0067 -Create Date: 2026-06-30 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0068" -down_revision: Union[str, None] = "0067" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.drop_column("ml_settings", "suggestion_threshold_character") - op.drop_column("ml_settings", "suggestion_threshold_general") - op.drop_column("ml_settings", "tagger_store_floor") - op.drop_column("ml_settings", "video_min_tag_frames") - op.drop_column("ml_settings", "tagger_model_version") - op.drop_column("image_record", "tagger_model_version") - op.drop_column("image_record", "centroid_scores") - - -def downgrade() -> None: - op.add_column( - "image_record", - sa.Column("centroid_scores", sa.JSON(), nullable=True), - ) - op.add_column( - "image_record", - sa.Column("tagger_model_version", sa.String(length=128), nullable=True), - ) - op.add_column( - "ml_settings", - sa.Column( - "tagger_model_version", sa.String(length=128), nullable=False, - server_default="camie-tagger-v2", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "video_min_tag_frames", sa.Integer(), nullable=False, - server_default="3", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "tagger_store_floor", sa.Float(), nullable=False, - server_default="0.7", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "suggestion_threshold_general", sa.Float(), nullable=False, - server_default="0.7", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "suggestion_threshold_character", sa.Float(), nullable=False, - server_default="0.7", - ), - ) diff --git a/alembic/versions/0069_default_siglip2.py b/alembic/versions/0069_default_siglip2.py deleted file mode 100644 index 7bef8b1..0000000 --- a/alembic/versions/0069_default_siglip2.py +++ /dev/null @@ -1,51 +0,0 @@ -"""default the embedder to SigLIP 2 — for FRESH installs only (#1203) - -Make SigLIP 2 (so400m, 512px; a 1152-d drop-in) the default embedder. New -installs start on it. An EXISTING library is NOT touched: flipping its stored -embedder version would mark every embedding stale (the scorer is version-gated) -and kill suggestions until a full re-embed+retrain — so an existing instance -switches deliberately via Settings → GPU agent → Embedding model → Re-embed → -Retrain. We detect "fresh" by the absence of any embedded image. - -Revision ID: 0069 -Revises: 0068 -Create Date: 2026-06-30 -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0069" -down_revision: Union[str, None] = "0068" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - -_NEW_NAME = "google/siglip2-so400m-patch16-512" -_NEW_VERSION = "siglip2-so400m-patch16-512" -_OLD_NAME = "google/siglip-so400m-patch14-384" -_OLD_VERSION = "siglip-so400m-patch14-384" - - -def upgrade() -> None: - # Fresh install (nothing embedded yet) → adopt SigLIP 2. - op.execute( - f""" - UPDATE ml_settings SET - embedder_model_name = '{_NEW_NAME}', - embedder_model_version = '{_NEW_VERSION}' - WHERE NOT EXISTS ( - SELECT 1 FROM image_record WHERE siglip_embedding IS NOT NULL - ) - """ - ) - op.alter_column("ml_settings", "embedder_model_name", server_default=_NEW_NAME) - op.alter_column( - "ml_settings", "embedder_model_version", server_default=_NEW_VERSION - ) - - -def downgrade() -> None: - op.alter_column("ml_settings", "embedder_model_name", server_default=_OLD_NAME) - op.alter_column( - "ml_settings", "embedder_model_version", server_default=_OLD_VERSION - ) diff --git a/alembic/versions/0070_gpu_job_lease_indexes.py b/alembic/versions/0070_gpu_job_lease_indexes.py deleted file mode 100644 index 10ec3f9..0000000 --- a/alembic/versions/0070_gpu_job_lease_indexes.py +++ /dev/null @@ -1,44 +0,0 @@ -"""partial indexes so GPU-job leasing stays O(batch), not O(completed) - -The lease claims the lowest-id pending (or expired-leased) jobs. With only a -plain `status` index, `... ORDER BY id LIMIT n` walked the primary-key index from -the start, skipping the entire prefix of already-done/error rows before reaching -pending ones — so leasing slowed to a crawl as `done` piled up (the whole reason -throughput fell off a cliff mid-run and /status stalled). Two partial indexes fix -it: the pending one is id-ordered so the hot path reads just the first n entries, -and the leased-expiry one keeps the crash-recovery reclaim + the orphan sweep -cheap. They cover only the small live slice of the table, so they stay tiny even -as the done/error history grows to millions. - -Revision ID: 0070 -Revises: 0069 -Create Date: 2026-06-30 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0070" -down_revision: Union[str, None] = "0069" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # Hot path: lowest-id pending jobs. Index on id, restricted to pending, so - # `WHERE status='pending' ORDER BY id LIMIT n` is a short index-order scan. - op.create_index( - "ix_gpu_job_pending", "gpu_job", ["id"], - postgresql_where=sa.text("status = 'pending'"), - ) - # Crash-recovery: expired leases, for the lease backstop + recover_orphaned. - op.create_index( - "ix_gpu_job_leased_expires", "gpu_job", ["lease_expires_at"], - postgresql_where=sa.text("status = 'leased'"), - ) - - -def downgrade() -> None: - op.drop_index("ix_gpu_job_leased_expires", table_name="gpu_job") - op.drop_index("ix_gpu_job_pending", table_name="gpu_job") diff --git a/alembic/versions/0071_image_record_earliest_post_date.py b/alembic/versions/0071_image_record_earliest_post_date.py deleted file mode 100644 index b2e8f0c..0000000 --- a/alembic/versions/0071_image_record_earliest_post_date.py +++ /dev/null @@ -1,80 +0,0 @@ -"""image_record.earliest_post_date: original-publish gallery sort key + index - -Revision ID: 0071 -Revises: 0070 -Create Date: 2026-07-01 - -effective_date (0035) keys off the PRIMARY post — which is often the repost / -download the file actually came from — and falls back to created_at, so the -gallery's default order surfaces download dates rather than when content was -first posted (operator-flagged 2026-07-01). Materialize a second sort key, -earliest_post_date = MIN(post_date) across ALL of an image's provenance posts -(every post it appears in), falling back to created_at only when no linked post -carries a date. Indexed (DESC, id DESC) so the "post date" gallery sort is an -index range scan just like effective_date. - -Backfill mirrors 0035: created_at baseline, then override with the MIN over -image_provenance ⋈ post. New rows get the created_at-equivalent server default; -services/importer.py recomputes it whenever a dated post is linked. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0071" -down_revision: Union[str, None] = "0070" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - # Add nullable first so the backfill can populate before NOT NULL. - op.add_column( - "image_record", - sa.Column("earliest_post_date", sa.DateTime(timezone=True), nullable=True), - ) - # Baseline: download date. Set-based (no per-row binds) → immune to the - # 65535 bind-parameter ceiling regardless of library size. - op.execute( - """ - UPDATE image_record - SET earliest_post_date = created_at - """ - ) - # Override with the earliest post_date across EVERY post the image appears - # in (image_provenance is the many-to-many edge; ignore posts with no date). - op.execute( - """ - UPDATE image_record AS ir - SET earliest_post_date = sub.min_date - FROM ( - SELECT ip.image_record_id AS iid, MIN(p.post_date) AS min_date - FROM image_provenance AS ip - JOIN post AS p ON p.id = ip.post_id - WHERE p.post_date IS NOT NULL - GROUP BY ip.image_record_id - ) AS sub - WHERE ir.id = sub.iid - """ - ) - op.alter_column( - "image_record", - "earliest_post_date", - nullable=False, - server_default=sa.text("now()"), - ) - # DESC/DESC matches the gallery's ORDER BY earliest_post_date DESC, id DESC - # so the "post date" scroll is a forward index scan; raw SQL because - # alembic's column list doesn't express per-column DESC cleanly. - op.execute( - "CREATE INDEX ix_image_record_earliest_post_date " - "ON image_record (earliest_post_date DESC, id DESC)" - ) - - -def downgrade() -> None: - op.drop_index( - "ix_image_record_earliest_post_date", table_name="image_record" - ) - op.drop_column("image_record", "earliest_post_date") diff --git a/alembic/versions/0072_gpu_job_triage_status.py b/alembic/versions/0072_gpu_job_triage_status.py deleted file mode 100644 index 1dce875..0000000 --- a/alembic/versions/0072_gpu_job_triage_status.py +++ /dev/null @@ -1,32 +0,0 @@ -"""gpu_job.triage_status — the probe's verdict on an errored job's FILE - -Failure triage (#125): a periodic sweep probes each errored image's file -(sha256 + decode, verify_integrity's machinery) exactly once and stores the -verdict here — 'defect' (the file is bad: recovery material, excluded from -/retry_errors) or 'file_ok' (failure was operational, safe to retry). NULL -means not yet probed; selecting on NULL is what makes the sweep resumable. -No index: the errored slice the sweep scans is tiny by design (tombstones). - -Revision ID: 0072 -Revises: 0071 -Create Date: 2026-07-02 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0072" -down_revision: Union[str, None] = "0071" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "gpu_job", sa.Column("triage_status", sa.String(16), nullable=True) - ) - - -def downgrade() -> None: - op.drop_column("gpu_job", "triage_status") diff --git a/alembic/versions/0073_drop_tag_eval_run.py b/alembic/versions/0073_drop_tag_eval_run.py deleted file mode 100644 index 4aedb38..0000000 --- a/alembic/versions/0073_drop_tag_eval_run.py +++ /dev/null @@ -1,46 +0,0 @@ -"""drop tag_eval_run — the head-vs-centroid eval harness is retired - -The eval (#1130) existed to prove the heads tagging spine on the operator's own -data. It did; the operator accepted the system and retired the harness -(2026-07-02) — card, API, task, model and this table all go. The eval's data -loaders + metric helpers live on in services/ml/training_data.py, where the -production heads trainer uses them nightly. - -Revision ID: 0073 -Revises: 0072 -Create Date: 2026-07-02 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from sqlalchemy.dialects import postgresql - -revision: str = "0073" -down_revision: Union[str, None] = "0072" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.drop_index("ix_tag_eval_run_status", table_name="tag_eval_run") - op.drop_table("tag_eval_run") - - -def downgrade() -> None: - # Recreates the shape from 0056 (data is not restorable). - op.create_table( - "tag_eval_run", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column("params", postgresql.JSONB(), nullable=False), - sa.Column("status", sa.String(length=16), nullable=False, - server_default="running"), - sa.Column("started_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now()), - sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True), - sa.Column("report", postgresql.JSONB(), nullable=True), - sa.Column("error", sa.Text(), nullable=True), - sa.Column("last_progress_at", sa.DateTime(timezone=True), - nullable=True), - ) - op.create_index("ix_tag_eval_run_status", "tag_eval_run", ["status"]) diff --git a/alembic/versions/0074_ml_settings_cpu_embed_enabled.py b/alembic/versions/0074_ml_settings_cpu_embed_enabled.py deleted file mode 100644 index 48ff8ea..0000000 --- a/alembic/versions/0074_ml_settings_cpu_embed_enabled.py +++ /dev/null @@ -1,35 +0,0 @@ -"""ml_settings.cpu_embed_enabled — the CPU embed fallback becomes a switch - -B3 (operator 2026-07-02): the ml-worker's only processing role is the CPU -whole-image embed for stacks without a GPU agent. ON by default (a fresh -install works agent-less); agent-equipped stacks that drop the ml-worker -container turn it off so import hooks stop queueing embed work into a queue -nothing consumes — the daily GPU 'embed' backfill covers those images. - -Revision ID: 0074 -Revises: 0073 -Create Date: 2026-07-02 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0074" -down_revision: Union[str, None] = "0073" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "cpu_embed_enabled", sa.Boolean(), nullable=False, - server_default=sa.true(), - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "cpu_embed_enabled") diff --git a/alembic/versions/0075_tag_is_system.py b/alembic/versions/0075_tag_is_system.py deleted file mode 100644 index a6b7e7a..0000000 --- a/alembic/versions/0075_tag_is_system.py +++ /dev/null @@ -1,60 +0,0 @@ -"""tag.is_system + seed the three hygiene system tags - -Training hygiene (operator 2026-07-03, milestone #128): rough WIPs tagged as a -character poison that character's head and CCIP references; banners/editor -screenshots pollute whole-image similarity. The fix keys on SYSTEM tags the -product ships — not operator configuration — so the seed lives here. - -Seeding ADOPTS an existing same-(name, kind=general) tag (case-insensitive, -matching TagService.rename's collision stance) instead of inserting a -duplicate, so an operator who already tagged `wip` keeps their applications. - -Revision ID: 0075 -Revises: 0074 -Create Date: 2026-07-03 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0075" -down_revision: Union[str, None] = "0074" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - -SYSTEM_TAG_NAMES = ("wip", "banner", "editor screenshot") - - -def upgrade() -> None: - op.add_column( - "tag", - sa.Column( - "is_system", sa.Boolean(), nullable=False, - server_default=sa.false(), - ), - ) - conn = op.get_bind() - for name in SYSTEM_TAG_NAMES: - adopted = conn.execute( - sa.text( - "UPDATE tag SET is_system = true " - "WHERE lower(name) = lower(:name) AND kind = 'general'" - ), - {"name": name}, - ) - if adopted.rowcount == 0: - conn.execute( - sa.text( - "INSERT INTO tag (name, kind, is_system) " - "VALUES (:name, 'general', true)" - ), - {"name": name}, - ) - - -def downgrade() -> None: - # The seeded rows survive as ordinary general tags — dropping the flag is - # enough to disarm the mechanism, and deleting rows would orphan any - # operator applications made while the flag existed. - op.drop_column("tag", "is_system") diff --git a/alembic/versions/0076_pixiv_ledgers.py b/alembic/versions/0076_pixiv_ledgers.py deleted file mode 100644 index 2655130..0000000 --- a/alembic/versions/0076_pixiv_ledgers.py +++ /dev/null @@ -1,82 +0,0 @@ -"""pixiv_seen_media + pixiv_failed_media: per-source ledgers - -Revision ID: 0076 -Revises: 0075 -Create Date: 2026-07-03 - -Pixiv native ingester (milestone #129, gallery-dl → native-core migration). -Mirrors the Patreon (0037/0038) and SubscribeStar (0054) ledger tables: a -seen-ledger so routine walks skip already-ingested media (recovery bypasses -it) and a dead-letter ledger so persistently-failing media stops re-burning -backfill chunks. Pixiv URLs carry no content hash, so `filehash` is always the -synthesized ``:p`` / ``:ugoira`` key — String(128) -matches the siblings. UNIQUE (source_id, filehash) is the upsert key on each. -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0076" -down_revision: Union[str, None] = "0075" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.create_table( - "pixiv_seen_media", - sa.Column("id", sa.Integer, primary_key=True), - sa.Column( - "source_id", - sa.Integer, - sa.ForeignKey("source.id", ondelete="CASCADE"), - nullable=False, - index=True, - ), - sa.Column("filehash", sa.String(128), nullable=False), - sa.Column("post_id", sa.String(64), nullable=True), - sa.Column( - "seen_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.UniqueConstraint( - "source_id", "filehash", name="uq_pixiv_seen_media_source_id" - ), - ) - op.create_table( - "pixiv_failed_media", - sa.Column("id", sa.Integer, primary_key=True), - sa.Column( - "source_id", - sa.Integer, - sa.ForeignKey("source.id", ondelete="CASCADE"), - nullable=False, - index=True, - ), - sa.Column("filehash", sa.String(128), nullable=False), - sa.Column("attempts", sa.Integer, nullable=False, server_default="1"), - sa.Column("last_error", sa.Text, nullable=True), - sa.Column( - "first_failed_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.Column( - "last_failed_at", - sa.DateTime(timezone=True), - nullable=False, - server_default=sa.text("NOW()"), - ), - sa.UniqueConstraint( - "source_id", "filehash", name="uq_pixiv_failed_media_source_id" - ), - ) - - -def downgrade() -> None: - op.drop_table("pixiv_failed_media") - op.drop_table("pixiv_seen_media") diff --git a/alembic/versions/0077_artist_name_not_unique.py b/alembic/versions/0077_artist_name_not_unique.py deleted file mode 100644 index 6a09288..0000000 --- a/alembic/versions/0077_artist_name_not_unique.py +++ /dev/null @@ -1,32 +0,0 @@ -"""drop uq_artist_name — decouple display name from identity/storage - -Revision ID: 0077 -Revises: 0076 -Create Date: 2026-07-04 - -Artist model fragility fix (milestone #130). One `slug` column was doing -identity + storage-path + display, and BOTH `name` and `slug` were UNIQUE, so -the display name couldn't be edited freely and two genuinely different creators -collided. Decouple: `slug` stays the immutable, unique storage/identity key (the -on-disk path component — untouched here); `name` becomes freely editable, NON- -unique display text. This migration only drops the `uq_artist_name` constraint; -no data moves and no path changes. -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0077" -down_revision: Union[str, None] = "0076" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.drop_constraint("uq_artist_name", "artist", type_="unique") - - -def downgrade() -> None: - # Re-adding the UNIQUE would fail if duplicate names now exist; callers that - # need to reverse this must dedupe names first. - op.create_unique_constraint("uq_artist_name", "artist", ["name"]) diff --git a/alembic/versions/0078_ml_settings_detectors.py b/alembic/versions/0078_ml_settings_detectors.py deleted file mode 100644 index 6d04601..0000000 --- a/alembic/versions/0078_ml_settings_detectors.py +++ /dev/null @@ -1,83 +0,0 @@ -"""ml_settings crop-proposer / detector config (#134) - -Move the WHERE-to-crop detector config (per-proposer enable + weights + conf, -plus caps + dedupe IoU) into the DB so it's UI-tunable and announced to the GPU -agent in the lease (like the embedder model) — no restart, agent env is now -bootstrap-only. All server_defaults are the working values so existing rows + -fresh installs crop out-of-the-box with all three proposers ON. - -Revision ID: 0078 -Revises: 0077 -Create Date: 2026-07-05 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0078" -down_revision: Union[str, None] = "0077" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -_ANATOMY_DEFAULT = ( - "https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt" -) -_PANEL_DEFAULT = "mosesb/best-comic-panel-detection::best.pt" - - -def upgrade() -> None: - op.add_column("ml_settings", sa.Column( - "detector_person_enabled", sa.Boolean(), nullable=False, - server_default=sa.true())) - op.add_column("ml_settings", sa.Column( - "detector_person_weights", sa.String(512), nullable=False, - server_default="yolo11n.pt")) - op.add_column("ml_settings", sa.Column( - "detector_person_conf", sa.Float(), nullable=False, - server_default=sa.text("0.35"))) - op.add_column("ml_settings", sa.Column( - "detector_anatomy_enabled", sa.Boolean(), nullable=False, - server_default=sa.true())) - op.add_column("ml_settings", sa.Column( - "detector_anatomy_weights", sa.String(512), nullable=False, - server_default=_ANATOMY_DEFAULT)) - op.add_column("ml_settings", sa.Column( - "detector_anatomy_conf", sa.Float(), nullable=False, - server_default=sa.text("0.30"))) - op.add_column("ml_settings", sa.Column( - "detector_panel_enabled", sa.Boolean(), nullable=False, - server_default=sa.true())) - op.add_column("ml_settings", sa.Column( - "detector_panel_weights", sa.String(512), nullable=False, - server_default=_PANEL_DEFAULT)) - op.add_column("ml_settings", sa.Column( - "detector_panel_conf", sa.Float(), nullable=False, - server_default=sa.text("0.30"))) - op.add_column("ml_settings", sa.Column( - "detector_max_figures", sa.Integer(), nullable=False, - server_default=sa.text("8"))) - op.add_column("ml_settings", sa.Column( - "detector_max_components", sa.Integer(), nullable=False, - server_default=sa.text("8"))) - op.add_column("ml_settings", sa.Column( - "detector_max_panels", sa.Integer(), nullable=False, - server_default=sa.text("8"))) - op.add_column("ml_settings", sa.Column( - "detector_max_regions", sa.Integer(), nullable=False, - server_default=sa.text("128"))) - op.add_column("ml_settings", sa.Column( - "detector_dedupe_iou", sa.Float(), nullable=False, - server_default=sa.text("0.85"))) - - -def downgrade() -> None: - for col in ( - "detector_person_enabled", "detector_person_weights", "detector_person_conf", - "detector_anatomy_enabled", "detector_anatomy_weights", "detector_anatomy_conf", - "detector_panel_enabled", "detector_panel_weights", "detector_panel_conf", - "detector_max_figures", "detector_max_components", "detector_max_panels", - "detector_max_regions", "detector_dedupe_iou", - ): - op.drop_column("ml_settings", col) diff --git a/alembic/versions/0079_character_prototypes.py b/alembic/versions/0079_character_prototypes.py deleted file mode 100644 index 8ada2f4..0000000 --- a/alembic/versions/0079_character_prototypes.py +++ /dev/null @@ -1,77 +0,0 @@ -"""character prototype store (#1317) — precomputed, incremental CCIP references - -New tables character_prototype + ccip_prototype_state, plus MLSettings columns -ccip_ref_signature (cheap global change gate) + ccip_prototype_cap (per-character -reference cap). The reference set the CCIP matcher uses becomes a precomputed -artifact refreshed incrementally off the request path. See milestone 138 / -backend.app.services.ml.character_prototypes. - -Revision ID: 0079 -Revises: 0078 -Create Date: 2026-07-06 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op -from pgvector.sqlalchemy import Vector - -revision: str = "0079" -down_revision: Union[str, None] = "0078" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - -# Matches models.image_region.CCIP_DIM (the CCIP figure-embedding width). -_CCIP_DIM = 768 - - -def upgrade() -> None: - op.create_table( - "character_prototype", - sa.Column("id", sa.Integer(), primary_key=True), - sa.Column( - "tag_id", sa.Integer(), - sa.ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, - ), - sa.Column("ccip_embedding", Vector(_CCIP_DIM), nullable=False), - sa.Column( - "region_id", sa.Integer(), - sa.ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True, - ), - ) - op.create_index( - "ix_character_prototype_tag_id", "character_prototype", ["tag_id"] - ) - op.create_table( - "ccip_prototype_state", - sa.Column( - "tag_id", sa.Integer(), - sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, - ), - sa.Column("fingerprint", sa.String(64), nullable=False), - sa.Column( - "updated_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - ) - op.add_column( - "ml_settings", - sa.Column("ccip_ref_signature", sa.String(128), nullable=True), - ) - op.add_column( - "ml_settings", - sa.Column( - "ccip_prototype_cap", sa.Integer(), nullable=False, - server_default=sa.text("64"), - ), - ) - - -def downgrade() -> None: - op.drop_column("ml_settings", "ccip_prototype_cap") - op.drop_column("ml_settings", "ccip_ref_signature") - op.drop_table("ccip_prototype_state") - op.drop_index( - "ix_character_prototype_tag_id", table_name="character_prototype" - ) - op.drop_table("character_prototype") diff --git a/alembic/versions/0080_tag_head_train_fingerprint.py b/alembic/versions/0080_tag_head_train_fingerprint.py deleted file mode 100644 index b4bd224..0000000 --- a/alembic/versions/0080_tag_head_train_fingerprint.py +++ /dev/null @@ -1,31 +0,0 @@ -"""tag_head.train_fingerprint (#1317 phase 2) — incremental head retraining - -A per-head training-data fingerprint (positive + rejection count/latest-timestamp) -so a manual Retrain refits only the tags whose data changed; the nightly run -ignores it (full reconcile). Nullable — a NULL fingerprint (existing heads) forces -a refit on the first incremental run, then it's stamped. - -Revision ID: 0080 -Revises: 0079 -Create Date: 2026-07-06 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0080" -down_revision: Union[str, None] = "0079" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "tag_head", - sa.Column("train_fingerprint", sa.String(128), nullable=True), - ) - - -def downgrade() -> None: - op.drop_column("tag_head", "train_fingerprint") diff --git a/alembic/versions/0081_stricter_auto_apply_defaults.py b/alembic/versions/0081_stricter_auto_apply_defaults.py deleted file mode 100644 index 8030eec..0000000 --- a/alembic/versions/0081_stricter_auto_apply_defaults.py +++ /dev/null @@ -1,43 +0,0 @@ -"""stricter auto-apply defaults (milestone 139) — cut auto-apply misfires - -head_auto_apply_min_positives 30→50 and ccip_auto_apply_threshold 0.92→0.95 -(operator-asked 2026-07-06). The head graduation precision bar stays 0.97 — the -operator confirmed the general-tag confidence was already well tuned; only the -support floor + the CCIP match confidence are raised. The model defaults change -for fresh installs; here we bump the existing singleton row IFF it is still at -the previous default, so a deliberate operator change is NOT clobbered. - -Revision ID: 0081 -Revises: 0080 -Create Date: 2026-07-06 -""" -from typing import Sequence, Union - -from alembic import op - -revision: str = "0081" -down_revision: Union[str, None] = "0080" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.execute( - "UPDATE ml_settings SET head_auto_apply_min_positives = 50 " - "WHERE head_auto_apply_min_positives = 30" - ) - op.execute( - "UPDATE ml_settings SET ccip_auto_apply_threshold = 0.95 " - "WHERE ccip_auto_apply_threshold = 0.92" - ) - - -def downgrade() -> None: - op.execute( - "UPDATE ml_settings SET head_auto_apply_min_positives = 30 " - "WHERE head_auto_apply_min_positives = 50" - ) - op.execute( - "UPDATE ml_settings SET ccip_auto_apply_threshold = 0.92 " - "WHERE ccip_auto_apply_threshold = 0.95" - ) diff --git a/alembic/versions/0082_presentation_auto_hide.py b/alembic/versions/0082_presentation_auto_hide.py deleted file mode 100644 index 8dc1f3c..0000000 --- a/alembic/versions/0082_presentation_auto_hide.py +++ /dev/null @@ -1,85 +0,0 @@ -"""presentation-chrome auto-hide (#141) — settings knobs + review table - -MLSettings gains presentation_auto_apply_enabled / _threshold and -presentation_conflict_threshold: banner + editor-screenshot auto-hide on the -sweep with a FLAT threshold (decoupled from content-head graduation), and a -conflict threshold that flags an auto-hide that "also looks like content". - -New table presentation_review records an auto-hidden chrome image that also -scored high on a content head, surfaced in the Hidden view for a keep-hidden / -un-hide decision. Resolved rows are pruned by retention. - -Revision ID: 0082 -Revises: 0081 -Create Date: 2026-07-07 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0082" -down_revision: Union[str, None] = "0081" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "presentation_auto_apply_enabled", sa.Boolean(), nullable=False, - server_default=sa.text("true"), - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "presentation_auto_apply_threshold", sa.Float(), nullable=False, - server_default=sa.text("0.90"), - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "presentation_conflict_threshold", sa.Float(), nullable=False, - server_default=sa.text("0.50"), - ), - ) - op.create_table( - "presentation_review", - sa.Column( - "image_record_id", sa.Integer(), - sa.ForeignKey("image_record.id", ondelete="CASCADE"), - primary_key=True, - ), - sa.Column( - "tag_id", sa.Integer(), - sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, - ), - sa.Column( - "conflict_tag_id", sa.Integer(), - sa.ForeignKey("tag.id", ondelete="SET NULL"), nullable=True, - ), - sa.Column("conflict_score", sa.Float(), nullable=False), - sa.Column( - "created_at", sa.DateTime(timezone=True), nullable=False, - server_default=sa.func.now(), - ), - sa.Column("resolved_at", sa.DateTime(timezone=True), nullable=True), - ) - # The review list queries the unresolved flags (resolved_at IS NULL). - op.create_index( - "ix_presentation_review_resolved_at", "presentation_review", - ["resolved_at"], - ) - - -def downgrade() -> None: - op.drop_index( - "ix_presentation_review_resolved_at", table_name="presentation_review" - ) - op.drop_table("presentation_review") - op.drop_column("ml_settings", "presentation_conflict_threshold") - op.drop_column("ml_settings", "presentation_auto_apply_threshold") - op.drop_column("ml_settings", "presentation_auto_apply_enabled") diff --git a/alembic/versions/0083_post_translation.py b/alembic/versions/0083_post_translation.py deleted file mode 100644 index d491d03..0000000 --- a/alembic/versions/0083_post_translation.py +++ /dev/null @@ -1,73 +0,0 @@ -"""post-text translation via Interpreter (milestone 143) — Post columns + settings - -Post gains the translated title/description + the detected source language, -Interpreter engine_version (cache key), and translated_at — filled by the -translate sweep. ImportSettings gains translation_enabled (OFF by default), -interpreter_base_url (EMPTY — the operator sets their own, behind a reverse -proxy), and translation_target_lang (en). - -Revision ID: 0083 -Revises: 0082 -Create Date: 2026-07-07 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0083" -down_revision: Union[str, None] = "0082" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "post", sa.Column("post_title_translated", sa.Text(), nullable=True) - ) - op.add_column( - "post", sa.Column("description_translated", sa.Text(), nullable=True) - ) - op.add_column( - "post", - sa.Column("translated_source_lang", sa.String(8), nullable=True), - ) - op.add_column( - "post", - sa.Column("translation_engine_version", sa.String(128), nullable=True), - ) - op.add_column( - "post", - sa.Column("translated_at", sa.DateTime(timezone=True), nullable=True), - ) - op.add_column( - "import_settings", - sa.Column( - "translation_enabled", sa.Boolean(), nullable=False, - server_default=sa.text("false"), - ), - ) - op.add_column( - "import_settings", - sa.Column( - "interpreter_base_url", sa.Text(), nullable=False, server_default="", - ), - ) - op.add_column( - "import_settings", - sa.Column( - "translation_target_lang", sa.Text(), nullable=False, - server_default="en", - ), - ) - - -def downgrade() -> None: - op.drop_column("import_settings", "translation_target_lang") - op.drop_column("import_settings", "interpreter_base_url") - op.drop_column("import_settings", "translation_enabled") - op.drop_column("post", "translated_at") - op.drop_column("post", "translation_engine_version") - op.drop_column("post", "translated_source_lang") - op.drop_column("post", "description_translated") - op.drop_column("post", "post_title_translated") diff --git a/alembic/versions/0084_translation_strictness_override.py b/alembic/versions/0084_translation_strictness_override.py deleted file mode 100644 index cd514b4..0000000 --- a/alembic/versions/0084_translation_strictness_override.py +++ /dev/null @@ -1,51 +0,0 @@ -"""translation strictness setting + per-post translation override (milestone 155) - -ImportSettings gains ``translation_min_confidence`` (the latin-script acceptance -floor, now operator-tunable in the UI; default 0.9 — stricter than the old -hardcoded 0.8, since Interpreter confidently mis-detects short ASCII English at -~0.86). Post gains ``translation_override`` — a sticky per-post choice of -auto / force / original so the operator can force a skipped translation on, or -knock a wrongly-translated one back to the original, and have it survive a -Re-translate-all. - -Revision ID: 0084 -Revises: 0083 -Create Date: 2026-07-10 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0084" -down_revision: Union[str, None] = "0083" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "import_settings", - sa.Column( - "translation_min_confidence", sa.Float(), nullable=False, - server_default=sa.text("0.9"), - ), - ) - op.add_column( - "post", - sa.Column( - "translation_override", sa.String(16), nullable=False, - server_default="auto", - ), - ) - op.create_check_constraint( - "ck_post_translation_override", - "post", - "translation_override IN ('auto', 'force', 'original')", - ) - - -def downgrade() -> None: - op.drop_constraint("ck_post_translation_override", "post", type_="check") - op.drop_column("post", "translation_override") - op.drop_column("import_settings", "translation_min_confidence") diff --git a/alembic/versions/0085_wip_title_tagging.py b/alembic/versions/0085_wip_title_tagging.py deleted file mode 100644 index 4d260b1..0000000 --- a/alembic/versions/0085_wip_title_tagging.py +++ /dev/null @@ -1,35 +0,0 @@ -"""title-based WIP auto-tagging (task #1458) — ImportSettings toggle - -ImportSettings gains wip_title_tagging_enabled (ON by default): when a freshly -imported post's title explicitly declares work-in-progress ("WIP" / "work in -progress"), the importer applies the `wip` system tag to its images. No new -table — the tag itself is the seeded `wip` system tag (migration 0075) and the -application reuses image_tag with source='wip_title'. - -Revision ID: 0085 -Revises: 0084 -Create Date: 2026-07-12 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0085" -down_revision: Union[str, None] = "0084" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "import_settings", - sa.Column( - "wip_title_tagging_enabled", sa.Boolean(), nullable=False, - server_default=sa.text("true"), - ), - ) - - -def downgrade() -> None: - op.drop_column("import_settings", "wip_title_tagging_enabled") diff --git a/alembic/versions/0086_process_auto_apply_settings.py b/alembic/versions/0086_process_auto_apply_settings.py deleted file mode 100644 index 16f03c7..0000000 --- a/alembic/versions/0086_process_auto_apply_settings.py +++ /dev/null @@ -1,61 +0,0 @@ -"""process auto-apply settings + review mode (#1464) — system-tag refactor - -The system-tag behavior refactor gives `wip` / `editor screenshot` (the PROCESS -group) their own provisional auto-apply, parallel to the presentation (chrome) -sweep. MLSettings gains three knobs: enabled (OFF by default — a new whole-library -auto-tagger is opt-in), the flat apply threshold, and the ring-loud conflict -threshold. presentation_review gains a `mode` column so one review surface serves -both chrome and process flags (existing rows backfill 'chrome'). server_defaults -so the existing rows fill cleanly. - -Revision ID: 0086 -Revises: 0085 -Create Date: 2026-07-13 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0086" -down_revision: Union[str, None] = "0085" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "ml_settings", - sa.Column( - "process_auto_apply_enabled", sa.Boolean(), nullable=False, - server_default=sa.text("false"), - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "process_auto_apply_threshold", sa.Float(), nullable=False, - server_default="0.90", - ), - ) - op.add_column( - "ml_settings", - sa.Column( - "process_conflict_threshold", sa.Float(), nullable=False, - server_default="0.50", - ), - ) - op.add_column( - "presentation_review", - sa.Column( - "mode", sa.String(16), nullable=False, - server_default="chrome", - ), - ) - - -def downgrade() -> None: - op.drop_column("presentation_review", "mode") - op.drop_column("ml_settings", "process_conflict_threshold") - op.drop_column("ml_settings", "process_auto_apply_threshold") - op.drop_column("ml_settings", "process_auto_apply_enabled") diff --git a/alembic/versions/0087_baseline.py b/alembic/versions/0087_baseline.py new file mode 100644 index 0000000..bf803ae --- /dev/null +++ b/alembic/versions/0087_baseline.py @@ -0,0 +1,872 @@ +"""Collapsed baseline — the whole schema in one revision. + +Replaces revisions 0001..0087, which narrated the build-out of this project +and were deleted in milestone 328 step 1. A new install creates the schema in +one step instead of replaying that history. + +WHY THE REVISION ID IS "0087" AND NOT "0001" +-------------------------------------------- +It is deliberately the id of the LAST revision this baseline collapses, so an +existing database needs no intervention at all: + + * a fresh install finds current=none, head=0087, runs this file once, and + ends stamped at 0087. + * an existing install is ALREADY at 0087, so `alembic upgrade head` finds + current == head and does nothing. + +The alternative — numbering this 0001 and stamping every existing database — +means running `alembic stamp` against live data, and stamp VALIDATES NOTHING. +It writes a version string whether or not the schema actually matches, so a +wrong baseline would be discovered later, by the next real migration, with no +clean way back. Keeping the id removes that operation instead of making it +safe. Future revisions continue at 0088. + +The one case this makes worse, and it fails LOUDLY rather than silently: a +database still sitting between 0001 and 0086 (i.e. never upgraded to head) +cannot be located in this chain and errors out. Upgrade to 0087 on a +pre-squash build first, then take this one. + +WHAT IS HAND-WRITTEN HERE +------------------------- +Most of this file is `alembic revision --autogenerate` output, but four +things are NOT in SQLAlchemy metadata and the generator cannot produce them. +Each fails differently, and none of them fail at generation time: + + 1. CREATE EXTENSION vector (was 0001) — without it the VECTOR + columns below cannot be created at all. + 2. CREATE EXTENSION tsm_system_rows (was 0004) — used by the random-sample + query path; its absence surfaces only when that query runs. + 3. The HNSW index on image_record.siglip_embedding (was 0036). Raw SQL + because alembic's create_index cannot express `USING hnsw (... + vector_cosine_ops)`. Its absence is the quietest failure of the four: + everything works, similarity search just stops using an index. + 4. `import pgvector.sqlalchemy.vector`. Autogenerate EMITS references to + pgvector.sqlalchemy.vector.VECTOR but does not add the import, so the + generated file dies with NameError on first run. + +The acceptance test for this file is not that it reads correctly — it is +`.forgejo/workflows/baseline.yml`, which builds a database from the old +0001..0087 chain (read out of git) and one from this file, and diffs +pg_dump --schema-only output. That is what proves nothing was missed. + +Revision ID: 0087 +Revises: +Create Date: 2026-08-30 + +""" +from typing import Sequence, Union + +from alembic import op +import sqlalchemy as sa +from sqlalchemy.dialects import postgresql + +# Autogenerate references pgvector.sqlalchemy.vector.VECTOR without importing +# it. Item 4 above. +import pgvector.sqlalchemy.vector + +revision: str = "0087" +down_revision: Union[str, None] = None +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + # Extensions FIRST: the VECTOR columns below cannot be created without + # `vector`, so ordering here is load-bearing, not tidiness. + op.execute("CREATE EXTENSION IF NOT EXISTS vector") + op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows") + + op.create_table('app_setting', + sa.Column('key', sa.String(length=64), nullable=False), + sa.Column('value', sa.Text(), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.PrimaryKeyConstraint('key', name=op.f('pk_app_setting')) + ) + op.create_table('artist', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('name', sa.String(length=255), nullable=False), + sa.Column('slug', sa.String(length=255), nullable=False), + sa.Column('notes', sa.Text(), nullable=True), + sa.Column('is_subscription', sa.Boolean(), nullable=False), + sa.Column('auto_check', sa.Boolean(), nullable=False), + sa.Column('check_interval_seconds', sa.Integer(), nullable=True), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.PrimaryKeyConstraint('id', name=op.f('pk_artist')), + sa.UniqueConstraint('slug', name=op.f('uq_artist_slug')) + ) + op.create_table('backup_run', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('kind', sa.String(length=16), nullable=False), + sa.Column('status', sa.String(length=16), nullable=False), + sa.Column('tag', sa.String(length=64), nullable=True), + sa.Column('triggered_by', sa.String(length=32), nullable=False), + sa.Column('started_at', sa.DateTime(timezone=True), nullable=False), + sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('sql_path', sa.Text(), nullable=True), + sa.Column('tar_path', sa.Text(), nullable=True), + sa.Column('size_bytes', sa.BigInteger(), nullable=True), + sa.Column('error', sa.Text(), nullable=True), + sa.Column('manifest', sa.JSON(), server_default='{}', nullable=False), + sa.Column('restored_from_id', sa.Integer(), nullable=True), + sa.ForeignKeyConstraint(['restored_from_id'], ['backup_run.id'], name=op.f('fk_backup_run_restored_from_id_backup_run'), ondelete='SET NULL'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_backup_run')) + ) + op.create_index(op.f('ix_backup_run_finished_at'), 'backup_run', ['finished_at'], unique=False) + op.create_index(op.f('ix_backup_run_kind'), 'backup_run', ['kind'], unique=False) + op.create_index(op.f('ix_backup_run_started_at'), 'backup_run', ['started_at'], unique=False) + op.create_index(op.f('ix_backup_run_status'), 'backup_run', ['status'], unique=False) + op.create_index(op.f('ix_backup_run_tag'), 'backup_run', ['tag'], unique=False) + op.create_table('credential', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('platform', sa.String(length=64), nullable=False), + sa.Column('credential_type', sa.String(length=32), nullable=False), + sa.Column('encrypted_blob', sa.LargeBinary(), nullable=False), + sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('expires_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('last_verified', sa.DateTime(timezone=True), nullable=True), + sa.PrimaryKeyConstraint('id', name=op.f('pk_credential')), + sa.UniqueConstraint('platform', name=op.f('uq_credential_platform')) + ) + op.create_table('head_auto_apply_run', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('dry_run', sa.Boolean(), nullable=False), + sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False), + sa.Column('status', sa.String(length=16), nullable=False), + sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('n_applied', sa.Integer(), nullable=True), + sa.Column('report', postgresql.JSONB(astext_type=sa.Text()), nullable=True), + sa.Column('error', sa.Text(), nullable=True), + sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True), + sa.PrimaryKeyConstraint('id', name=op.f('pk_head_auto_apply_run')) + ) + op.create_index(op.f('ix_head_auto_apply_run_status'), 'head_auto_apply_run', ['status'], unique=False) + op.create_table('head_training_run', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False), + sa.Column('status', sa.String(length=16), nullable=False), + sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('n_trained', sa.Integer(), nullable=True), + sa.Column('n_skipped', sa.Integer(), nullable=True), + sa.Column('error', sa.Text(), nullable=True), + sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True), + sa.PrimaryKeyConstraint('id', name=op.f('pk_head_training_run')) + ) + op.create_index(op.f('ix_head_training_run_status'), 'head_training_run', ['status'], unique=False) + op.create_table('import_batch', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('triggered_by', sa.String(length=32), nullable=False), + sa.Column('source_path', sa.Text(), nullable=False), + sa.Column('scan_mode', sa.String(length=16), nullable=False), + sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('total_files', sa.Integer(), nullable=False), + sa.Column('imported', sa.Integer(), nullable=False), + sa.Column('skipped', sa.Integer(), nullable=False), + sa.Column('failed', sa.Integer(), nullable=False), + sa.Column('attachments', sa.Integer(), nullable=False), + sa.Column('refreshed', sa.Integer(), nullable=False), + sa.Column('status', sa.String(length=16), nullable=False), + sa.PrimaryKeyConstraint('id', name=op.f('pk_import_batch')) + ) + op.create_index(op.f('ix_import_batch_status'), 'import_batch', ['status'], unique=False) + op.create_table('import_settings', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('import_scan_path', sa.Text(), nullable=False), + sa.Column('min_width', sa.Integer(), nullable=False), + sa.Column('min_height', sa.Integer(), nullable=False), + sa.Column('skip_transparent', sa.Boolean(), nullable=False), + sa.Column('transparency_threshold', sa.Float(), nullable=False), + sa.Column('skip_single_color', sa.Boolean(), nullable=False), + sa.Column('single_color_threshold', sa.Float(), nullable=False), + sa.Column('single_color_tolerance', sa.Integer(), nullable=False), + sa.Column('phash_threshold', sa.Integer(), nullable=False), + sa.Column('download_rate_limit_seconds', sa.Float(), nullable=False), + sa.Column('download_validate_files', sa.Boolean(), nullable=False), + sa.Column('download_schedule_default_seconds', sa.Integer(), nullable=False), + sa.Column('download_event_retention_days', sa.Integer(), nullable=False), + sa.Column('download_failure_warning_threshold', sa.Integer(), nullable=False), + sa.Column('backup_db_nightly_enabled', sa.Boolean(), nullable=False), + sa.Column('backup_db_nightly_hour_utc', sa.Integer(), nullable=False), + sa.Column('backup_db_keep_last_n', sa.Integer(), nullable=False), + sa.Column('backup_images_keep_last_n', sa.Integer(), nullable=False), + sa.Column('series_suggest_enabled', sa.Boolean(), nullable=False), + sa.Column('series_suggest_threshold', sa.Float(), nullable=False), + sa.Column('extdl_mega_enabled', sa.Boolean(), server_default='true', nullable=False), + sa.Column('extdl_gdrive_enabled', sa.Boolean(), server_default='true', nullable=False), + sa.Column('extdl_mediafire_enabled', sa.Boolean(), server_default='true', nullable=False), + sa.Column('extdl_dropbox_enabled', sa.Boolean(), server_default='true', nullable=False), + sa.Column('extdl_pixeldrain_enabled', sa.Boolean(), server_default='true', nullable=False), + sa.Column('translation_enabled', sa.Boolean(), server_default='false', nullable=False), + sa.Column('interpreter_base_url', sa.Text(), server_default='', nullable=False), + sa.Column('translation_target_lang', sa.Text(), server_default='en', nullable=False), + sa.Column('translation_min_confidence', sa.Float(), server_default='0.9', nullable=False), + sa.Column('wip_title_tagging_enabled', sa.Boolean(), server_default='true', nullable=False), + sa.Column('wip_soft_title_tagging_enabled', sa.Boolean(), server_default='false', nullable=False), + sa.CheckConstraint('id = 1', name=op.f('ck_import_settings_singleton')), + sa.PrimaryKeyConstraint('id', name=op.f('pk_import_settings')) + ) + op.create_table('library_audit_run', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('rule', sa.String(length=32), nullable=False), + sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False), + sa.Column('status', sa.String(length=16), nullable=False), + sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('scanned_count', sa.Integer(), nullable=False), + sa.Column('matched_count', sa.Integer(), nullable=False), + sa.Column('matched_ids', postgresql.JSONB(astext_type=sa.Text()), nullable=False), + sa.Column('error', sa.Text(), nullable=True), + sa.Column('resume_after_id', sa.Integer(), nullable=False), + sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True), + sa.PrimaryKeyConstraint('id', name=op.f('pk_library_audit_run')) + ) + op.create_index(op.f('ix_library_audit_run_rule'), 'library_audit_run', ['rule'], unique=False) + op.create_index(op.f('ix_library_audit_run_status'), 'library_audit_run', ['status'], unique=False) + op.create_table('ml_settings', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('cpu_embed_enabled', sa.Boolean(), nullable=False), + sa.Column('video_frame_interval_seconds', sa.Float(), nullable=False), + sa.Column('video_max_frames', sa.Integer(), nullable=False), + sa.Column('head_min_positives', sa.Integer(), nullable=False), + sa.Column('head_auto_apply_precision', sa.Float(), nullable=False), + sa.Column('head_auto_apply_enabled', sa.Boolean(), nullable=False), + sa.Column('head_auto_apply_min_positives', sa.Integer(), nullable=False), + sa.Column('ccip_match_threshold', sa.Float(), nullable=False), + sa.Column('ccip_auto_apply_enabled', sa.Boolean(), nullable=False), + sa.Column('ccip_auto_apply_threshold', sa.Float(), nullable=False), + sa.Column('presentation_auto_apply_enabled', sa.Boolean(), nullable=False), + sa.Column('presentation_auto_apply_threshold', sa.Float(), nullable=False), + sa.Column('presentation_conflict_threshold', sa.Float(), nullable=False), + sa.Column('process_auto_apply_enabled', sa.Boolean(), nullable=False), + sa.Column('process_auto_apply_threshold', sa.Float(), nullable=False), + sa.Column('process_conflict_threshold', sa.Float(), nullable=False), + sa.Column('embedder_model_version', sa.String(length=128), nullable=False), + sa.Column('embedder_model_name', sa.String(length=128), nullable=False), + sa.Column('detector_person_enabled', sa.Boolean(), nullable=False), + sa.Column('detector_person_weights', sa.String(length=512), nullable=False), + sa.Column('detector_person_conf', sa.Float(), nullable=False), + sa.Column('detector_anatomy_enabled', sa.Boolean(), nullable=False), + sa.Column('detector_anatomy_weights', sa.String(length=512), nullable=False), + sa.Column('detector_anatomy_conf', sa.Float(), nullable=False), + sa.Column('detector_panel_enabled', sa.Boolean(), nullable=False), + sa.Column('detector_panel_weights', sa.String(length=512), nullable=False), + sa.Column('detector_panel_conf', sa.Float(), nullable=False), + sa.Column('detector_max_figures', sa.Integer(), nullable=False), + sa.Column('detector_max_components', sa.Integer(), nullable=False), + sa.Column('detector_max_panels', sa.Integer(), nullable=False), + sa.Column('detector_max_regions', sa.Integer(), nullable=False), + sa.Column('detector_dedupe_iou', sa.Float(), nullable=False), + sa.Column('ccip_ref_signature', sa.String(length=128), nullable=True), + sa.Column('ccip_prototype_cap', sa.Integer(), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.CheckConstraint('id = 1', name=op.f('ck_ml_settings_singleton')), + sa.PrimaryKeyConstraint('id', name=op.f('pk_ml_settings')) + ) + op.create_table('tag', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('name', sa.String(length=255), nullable=False), + sa.Column('kind', sa.Enum('artist', 'character', 'fandom', 'general', 'series', 'archive', 'post', name='tag_kind'), nullable=False), + sa.Column('fandom_id', sa.Integer(), nullable=True), + sa.Column('is_system', sa.Boolean(), server_default=sa.text('false'), nullable=False), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.CheckConstraint("(fandom_id IS NULL) OR (kind = 'character')", name=op.f('ck_tag_ck_tag_fandom_requires_character')), + sa.ForeignKeyConstraint(['fandom_id'], ['tag.id'], name=op.f('fk_tag_fandom_id_tag'), ondelete='SET NULL'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_tag')) + ) + op.create_index(op.f('ix_tag_fandom_id'), 'tag', ['fandom_id'], unique=False) + op.create_table('task_run', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('celery_task_id', sa.String(length=64), nullable=False), + sa.Column('queue', sa.String(length=32), nullable=False), + sa.Column('task_name', sa.String(length=128), nullable=False), + sa.Column('target_id', sa.Integer(), nullable=True), + sa.Column('started_at', sa.DateTime(timezone=True), nullable=False), + sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('duration_ms', sa.Integer(), nullable=True), + sa.Column('status', sa.String(length=16), nullable=False), + sa.Column('error_type', sa.String(length=128), nullable=True), + sa.Column('error_message', sa.Text(), nullable=True), + sa.Column('retry_count', sa.Integer(), nullable=True), + sa.Column('worker_hostname', sa.String(length=128), nullable=True), + sa.Column('args_summary', sa.String(length=255), nullable=True), + sa.PrimaryKeyConstraint('id', name=op.f('pk_task_run')) + ) + op.create_index(op.f('ix_task_run_celery_task_id'), 'task_run', ['celery_task_id'], unique=False) + op.create_index(op.f('ix_task_run_finished_at'), 'task_run', ['finished_at'], unique=False) + op.create_index(op.f('ix_task_run_queue'), 'task_run', ['queue'], unique=False) + op.create_index(op.f('ix_task_run_started_at'), 'task_run', ['started_at'], unique=False) + op.create_index(op.f('ix_task_run_status'), 'task_run', ['status'], unique=False) + op.create_index(op.f('ix_task_run_task_name'), 'task_run', ['task_name'], unique=False) + op.create_table('artist_visit', + sa.Column('artist_id', sa.Integer(), nullable=False), + sa.Column('last_viewed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_artist_visit_artist_id_artist'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('artist_id', name=op.f('pk_artist_visit')) + ) + op.create_table('ccip_prototype_state', + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('fingerprint', sa.String(length=64), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_ccip_prototype_state_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_ccip_prototype_state')) + ) + op.create_table('head_metric', + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('n_misfires', sa.Integer(), nullable=False), + sa.Column('n_underfires', sa.Integer(), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metric_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_head_metric')) + ) + op.create_table('head_metrics_snapshot', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('name', sa.String(length=255), nullable=False), + sa.Column('snapshot_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('n_auto_applied', sa.Integer(), nullable=False), + sa.Column('n_misfires', sa.Integer(), nullable=False), + sa.Column('n_underfires', sa.Integer(), nullable=False), + sa.Column('ap', sa.Float(), nullable=True), + sa.Column('precision_cv', sa.Float(), nullable=True), + sa.Column('recall', sa.Float(), nullable=True), + sa.Column('n_pos', sa.Integer(), nullable=True), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metrics_snapshot_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_head_metrics_snapshot')) + ) + op.create_index(op.f('ix_head_metrics_snapshot_snapshot_at'), 'head_metrics_snapshot', ['snapshot_at'], unique=False) + op.create_index(op.f('ix_head_metrics_snapshot_tag_id'), 'head_metrics_snapshot', ['tag_id'], unique=False) + op.create_table('source', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('artist_id', sa.Integer(), nullable=False), + sa.Column('platform', sa.String(length=64), nullable=False), + sa.Column('url', sa.Text(), nullable=False), + sa.Column('enabled', sa.Boolean(), nullable=False), + sa.Column('config_overrides', sa.JSON(), nullable=True), + sa.Column('last_checked_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('last_error', sa.Text(), nullable=True), + sa.Column('error_type', sa.String(length=32), nullable=True), + sa.Column('check_interval_override', sa.Integer(), nullable=True), + sa.Column('consecutive_failures', sa.Integer(), nullable=False), + sa.Column('backfill_runs_remaining', sa.Integer(), server_default='0', nullable=False), + sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_source_artist_id_artist'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_source')) + ) + op.create_index(op.f('ix_source_artist_id'), 'source', ['artist_id'], unique=False) + op.create_index(op.f('ix_source_error_type'), 'source', ['error_type'], unique=False) + op.create_table('tag_alias', + sa.Column('alias_string', sa.String(length=255), nullable=False), + sa.Column('alias_category', sa.String(length=32), nullable=False), + sa.Column('canonical_tag_id', sa.Integer(), nullable=False), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['canonical_tag_id'], ['tag.id'], name=op.f('fk_tag_alias_canonical_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('alias_string', 'alias_category', name=op.f('pk_tag_alias')) + ) + op.create_index(op.f('ix_tag_alias_canonical_tag_id'), 'tag_alias', ['canonical_tag_id'], unique=False) + op.create_table('tag_head', + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('embedding_version', sa.String(length=128), nullable=False), + sa.Column('weights', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=False), + sa.Column('bias', sa.Float(), nullable=False), + sa.Column('suggest_threshold', sa.Float(), nullable=False), + sa.Column('auto_apply_threshold', sa.Float(), nullable=True), + sa.Column('n_pos', sa.Integer(), nullable=False), + sa.Column('n_neg', sa.Integer(), nullable=False), + sa.Column('ap', sa.Float(), nullable=False), + sa.Column('precision_cv', sa.Float(), nullable=False), + sa.Column('recall', sa.Float(), nullable=False), + sa.Column('trained_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('train_fingerprint', sa.String(length=128), nullable=True), + sa.Column('metrics', postgresql.JSONB(astext_type=sa.Text()), nullable=True), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_head_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_tag_head')) + ) + op.create_table('patreon_failed_media', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=False), + sa.Column('filehash', sa.String(length=128), nullable=False), + sa.Column('attempts', sa.Integer(), nullable=False), + sa.Column('last_error', sa.Text(), nullable=True), + sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_patreon_failed_media_source_id_source'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_patreon_failed_media')), + sa.UniqueConstraint('source_id', 'filehash', name='uq_patreon_failed_media_source_id') + ) + op.create_index(op.f('ix_patreon_failed_media_source_id'), 'patreon_failed_media', ['source_id'], unique=False) + op.create_table('patreon_seen_media', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=False), + sa.Column('filehash', sa.String(length=128), nullable=False), + sa.Column('post_id', sa.String(length=64), nullable=True), + sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_patreon_seen_media_source_id_source'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_patreon_seen_media')), + sa.UniqueConstraint('source_id', 'filehash', name='uq_patreon_seen_media_source_id') + ) + op.create_index(op.f('ix_patreon_seen_media_source_id'), 'patreon_seen_media', ['source_id'], unique=False) + op.create_table('pixiv_failed_media', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=False), + sa.Column('filehash', sa.String(length=128), nullable=False), + sa.Column('attempts', sa.Integer(), nullable=False), + sa.Column('last_error', sa.Text(), nullable=True), + sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_pixiv_failed_media_source_id_source'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_pixiv_failed_media')), + sa.UniqueConstraint('source_id', 'filehash', name='uq_pixiv_failed_media_source_id') + ) + op.create_index(op.f('ix_pixiv_failed_media_source_id'), 'pixiv_failed_media', ['source_id'], unique=False) + op.create_table('pixiv_seen_media', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=False), + sa.Column('filehash', sa.String(length=128), nullable=False), + sa.Column('post_id', sa.String(length=64), nullable=True), + sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_pixiv_seen_media_source_id_source'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_pixiv_seen_media')), + sa.UniqueConstraint('source_id', 'filehash', name='uq_pixiv_seen_media_source_id') + ) + op.create_index(op.f('ix_pixiv_seen_media_source_id'), 'pixiv_seen_media', ['source_id'], unique=False) + op.create_table('post', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=True), + sa.Column('artist_id', sa.Integer(), nullable=False), + sa.Column('external_post_id', sa.String(length=128), nullable=False), + sa.Column('post_url', sa.Text(), nullable=True), + sa.Column('post_title', sa.Text(), nullable=True), + sa.Column('post_date', sa.DateTime(timezone=True), nullable=True), + sa.Column('raw_metadata', sa.JSON(), nullable=True), + sa.Column('description', sa.Text(), nullable=True), + sa.Column('attachment_count', sa.Integer(), nullable=True), + sa.Column('post_title_translated', sa.Text(), nullable=True), + sa.Column('description_translated', sa.Text(), nullable=True), + sa.Column('translated_source_lang', sa.String(length=8), nullable=True), + sa.Column('translation_engine_version', sa.String(length=128), nullable=True), + sa.Column('translated_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('translation_override', sa.String(length=16), server_default='auto', nullable=False), + sa.Column('downloaded_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.CheckConstraint("translation_override IN ('auto', 'force', 'original')", name=op.f('ck_post_ck_post_translation_override')), + sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_artist_id_artist'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_post_source_id_source'), ondelete='SET NULL'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_post')), + sa.UniqueConstraint('source_id', 'external_post_id', name='uq_post_source_external_id') + ) + op.create_index(op.f('ix_post_artist_id'), 'post', ['artist_id'], unique=False) + op.create_index(op.f('ix_post_source_id'), 'post', ['source_id'], unique=False) + op.create_table('subscribestar_failed_media', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=False), + sa.Column('filehash', sa.String(length=128), nullable=False), + sa.Column('attempts', sa.Integer(), nullable=False), + sa.Column('last_error', sa.Text(), nullable=True), + sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_subscribestar_failed_media_source_id_source'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_subscribestar_failed_media')), + sa.UniqueConstraint('source_id', 'filehash', name='uq_subscribestar_failed_media_source_id') + ) + op.create_index(op.f('ix_subscribestar_failed_media_source_id'), 'subscribestar_failed_media', ['source_id'], unique=False) + op.create_table('subscribestar_seen_media', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=False), + sa.Column('filehash', sa.String(length=128), nullable=False), + sa.Column('post_id', sa.String(length=64), nullable=True), + sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_subscribestar_seen_media_source_id_source'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_subscribestar_seen_media')), + sa.UniqueConstraint('source_id', 'filehash', name='uq_subscribestar_seen_media_source_id') + ) + op.create_index(op.f('ix_subscribestar_seen_media_source_id'), 'subscribestar_seen_media', ['source_id'], unique=False) + op.create_table('download_event', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=False), + sa.Column('post_id', sa.Integer(), nullable=True), + sa.Column('status', sa.String(length=32), nullable=False), + sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('bytes_downloaded', sa.BigInteger(), nullable=False), + sa.Column('files_count', sa.Integer(), nullable=False), + sa.Column('error', sa.Text(), nullable=True), + sa.Column('metadata', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'{}'::jsonb"), nullable=False), + sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_download_event_post_id_post'), ondelete='SET NULL'), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_download_event_source_id_source'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_download_event')) + ) + op.create_index(op.f('ix_download_event_post_id'), 'download_event', ['post_id'], unique=False) + op.create_index(op.f('ix_download_event_source_id'), 'download_event', ['source_id'], unique=False) + op.create_table('image_record', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('path', sa.Text(), nullable=False), + sa.Column('sha256', sa.String(length=64), nullable=False), + sa.Column('phash', sa.String(length=32), nullable=True), + sa.Column('size_bytes', sa.BigInteger(), nullable=False), + sa.Column('mime', sa.String(length=64), nullable=False), + sa.Column('width', sa.Integer(), nullable=True), + sa.Column('height', sa.Integer(), nullable=True), + sa.Column('duration_seconds', sa.Float(), nullable=True), + sa.Column('integrity_status', sa.String(length=24), nullable=False), + sa.Column('thumbnail_path', sa.Text(), nullable=True), + sa.Column('source_url', sa.Text(), nullable=True), + sa.Column('source_filehash', sa.String(length=32), nullable=True), + sa.Column('origin', sa.Enum('downloaded', 'imported_filesystem', 'uploaded', name='origin_enum'), nullable=False), + sa.Column('primary_post_id', sa.Integer(), nullable=True), + sa.Column('artist_id', sa.Integer(), nullable=True), + sa.Column('siglip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=True), + sa.Column('siglip_model_version', sa.String(length=128), nullable=True), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('effective_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('earliest_post_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_image_record_artist_id_artist'), ondelete='SET NULL'), + sa.ForeignKeyConstraint(['primary_post_id'], ['post.id'], name=op.f('fk_image_record_primary_post_id_post'), ondelete='SET NULL'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_image_record')), + sa.UniqueConstraint('path', name=op.f('uq_image_record_path')) + ) + op.create_index(op.f('ix_image_record_artist_id'), 'image_record', ['artist_id'], unique=False) + op.create_index(op.f('ix_image_record_integrity_status'), 'image_record', ['integrity_status'], unique=False) + op.create_index(op.f('ix_image_record_phash'), 'image_record', ['phash'], unique=False) + op.create_index(op.f('ix_image_record_primary_post_id'), 'image_record', ['primary_post_id'], unique=False) + op.create_index(op.f('ix_image_record_sha256'), 'image_record', ['sha256'], unique=True) + op.create_index(op.f('ix_image_record_source_filehash'), 'image_record', ['source_filehash'], unique=False) + op.create_table('post_attachment', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('post_id', sa.Integer(), nullable=True), + sa.Column('artist_id', sa.Integer(), nullable=True), + sa.Column('sha256', sa.String(length=64), nullable=False), + sa.Column('path', sa.Text(), nullable=False), + sa.Column('original_filename', sa.Text(), nullable=False), + sa.Column('ext', sa.String(length=32), nullable=False), + sa.Column('mime', sa.String(length=128), nullable=True), + sa.Column('size_bytes', sa.BigInteger(), nullable=False), + sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_attachment_artist_id_artist'), ondelete='SET NULL'), + sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_post_attachment_post_id_post'), ondelete='SET NULL'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_post_attachment')) + ) + op.create_index(op.f('ix_post_attachment_artist_id'), 'post_attachment', ['artist_id'], unique=False) + op.create_index(op.f('ix_post_attachment_post_id'), 'post_attachment', ['post_id'], unique=False) + op.create_index(op.f('ix_post_attachment_sha256'), 'post_attachment', ['sha256'], unique=False) + op.create_index('uq_post_attachment_null_post_sha', 'post_attachment', ['sha256'], unique=True, postgresql_where=sa.text('post_id IS NULL')) + op.create_index('uq_post_attachment_post_sha', 'post_attachment', ['post_id', 'sha256'], unique=True, postgresql_where=sa.text('post_id IS NOT NULL')) + op.create_table('series_suggestion', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('post_id', sa.Integer(), nullable=False), + sa.Column('series_tag_id', sa.Integer(), nullable=False), + sa.Column('score', sa.Float(), nullable=False), + sa.Column('signals', sa.JSON(), nullable=True), + sa.Column('status', sa.String(length=16), server_default='pending', nullable=False), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_series_suggestion_post_id_post'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_suggestion_series_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_series_suggestion')), + sa.UniqueConstraint('post_id', 'series_tag_id', name='uq_series_suggestion_post_series') + ) + op.create_index(op.f('ix_series_suggestion_post_id'), 'series_suggestion', ['post_id'], unique=False) + op.create_index(op.f('ix_series_suggestion_series_tag_id'), 'series_suggestion', ['series_tag_id'], unique=False) + op.create_index(op.f('ix_series_suggestion_status'), 'series_suggestion', ['status'], unique=False) + op.create_table('external_link', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('post_id', sa.Integer(), nullable=False), + sa.Column('artist_id', sa.Integer(), nullable=True), + sa.Column('host', sa.String(length=16), nullable=False), + sa.Column('url', sa.Text(), nullable=False), + sa.Column('label', sa.Text(), nullable=True), + sa.Column('status', sa.String(length=16), server_default='pending', nullable=False), + sa.Column('attempts', sa.Integer(), server_default=sa.text('0'), nullable=False), + sa.Column('last_error', sa.Text(), nullable=True), + sa.Column('attachment_id', sa.Integer(), nullable=True), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('completed_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('duration_seconds', sa.Float(), nullable=True), + sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_external_link_artist_id_artist'), ondelete='SET NULL'), + sa.ForeignKeyConstraint(['attachment_id'], ['post_attachment.id'], name=op.f('fk_external_link_attachment_id_post_attachment'), ondelete='SET NULL'), + sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_external_link_post_id_post'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_external_link')) + ) + op.create_index(op.f('ix_external_link_artist_id'), 'external_link', ['artist_id'], unique=False) + op.create_index(op.f('ix_external_link_post_id'), 'external_link', ['post_id'], unique=False) + op.create_index('ix_external_link_status', 'external_link', ['status'], unique=False) + op.create_index('uq_external_link_post_url', 'external_link', ['post_id', 'url'], unique=True) + op.create_table('gpu_job', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('image_record_id', sa.Integer(), nullable=False), + sa.Column('task', sa.String(length=32), nullable=False), + sa.Column('status', sa.String(length=16), nullable=False), + sa.Column('lease_token', sa.String(length=64), nullable=True), + sa.Column('leased_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('lease_expires_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('attempts', sa.Integer(), nullable=False), + sa.Column('error', sa.Text(), nullable=True), + sa.Column('triage_status', sa.String(length=16), nullable=True), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_gpu_job_image_record_id_image_record'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_gpu_job')) + ) + op.create_index(op.f('ix_gpu_job_image_record_id'), 'gpu_job', ['image_record_id'], unique=False) + op.create_index('ix_gpu_job_leased_expires', 'gpu_job', ['lease_expires_at'], unique=False, postgresql_where=sa.text("status = 'leased'")) + op.create_index('ix_gpu_job_pending', 'gpu_job', ['id'], unique=False, postgresql_where=sa.text("status = 'pending'")) + op.create_index(op.f('ix_gpu_job_status'), 'gpu_job', ['status'], unique=False) + op.create_table('image_provenance', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('image_record_id', sa.Integer(), nullable=False), + sa.Column('post_id', sa.Integer(), nullable=False), + sa.Column('source_id', sa.Integer(), nullable=True), + sa.Column('from_attachment_id', sa.Integer(), nullable=True), + sa.Column('captured_metadata', sa.JSON(), nullable=True), + sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['from_attachment_id'], ['post_attachment.id'], name=op.f('fk_image_provenance_from_attachment_id_post_attachment'), ondelete='SET NULL'), + sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_provenance_image_record_id_image_record'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_image_provenance_post_id_post'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_image_provenance_source_id_source'), ondelete='SET NULL'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_image_provenance')), + sa.UniqueConstraint('image_record_id', 'post_id', name='uq_image_provenance_image_post') + ) + op.create_index(op.f('ix_image_provenance_from_attachment_id'), 'image_provenance', ['from_attachment_id'], unique=False) + op.create_index(op.f('ix_image_provenance_image_record_id'), 'image_provenance', ['image_record_id'], unique=False) + op.create_index(op.f('ix_image_provenance_post_id'), 'image_provenance', ['post_id'], unique=False) + op.create_index(op.f('ix_image_provenance_source_id'), 'image_provenance', ['source_id'], unique=False) + op.create_table('image_region', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('image_record_id', sa.Integer(), nullable=False), + sa.Column('kind', sa.String(length=16), nullable=False), + sa.Column('frame_time', sa.Float(), nullable=True), + sa.Column('rx', sa.Float(), nullable=False), + sa.Column('ry', sa.Float(), nullable=False), + sa.Column('rw', sa.Float(), nullable=False), + sa.Column('rh', sa.Float(), nullable=False), + sa.Column('score', sa.Float(), nullable=True), + sa.Column('detector_version', sa.String(length=64), nullable=True), + sa.Column('crop_version', sa.String(length=64), nullable=True), + sa.Column('embedding_version', sa.String(length=128), nullable=True), + sa.Column('ccip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=768), nullable=True), + sa.Column('siglip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=True), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_region_image_record_id_image_record'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_image_region')) + ) + op.create_index(op.f('ix_image_region_image_record_id'), 'image_region', ['image_record_id'], unique=False) + op.create_table('image_tag', + sa.Column('image_record_id', sa.Integer(), nullable=False), + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('source', sa.String(length=32), nullable=False), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_tag_image_record_id_image_record'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_image_tag_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_image_tag')) + ) + op.create_table('import_task', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('batch_id', sa.Integer(), nullable=False), + sa.Column('source_path', sa.Text(), nullable=False), + sa.Column('task_type', sa.String(length=16), nullable=False), + sa.Column('status', sa.String(length=16), nullable=False), + sa.Column('recovery_count', sa.Integer(), nullable=False), + sa.Column('refetched', sa.Boolean(), nullable=False), + sa.Column('result_image_id', sa.Integer(), nullable=True), + sa.Column('error', sa.Text(), nullable=True), + sa.Column('size_bytes', sa.BigInteger(), nullable=True), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('started_at', sa.DateTime(timezone=True), nullable=True), + sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True), + sa.ForeignKeyConstraint(['batch_id'], ['import_batch.id'], name=op.f('fk_import_task_batch_id_import_batch'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['result_image_id'], ['image_record.id'], name=op.f('fk_import_task_result_image_id_image_record'), ondelete='SET NULL'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_import_task')) + ) + op.create_index(op.f('ix_import_task_batch_id'), 'import_task', ['batch_id'], unique=False) + op.create_index(op.f('ix_import_task_status'), 'import_task', ['status'], unique=False) + op.create_table('presentation_review', + sa.Column('image_record_id', sa.Integer(), nullable=False), + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('conflict_tag_id', sa.Integer(), nullable=True), + sa.Column('conflict_score', sa.Float(), nullable=False), + sa.Column('mode', sa.String(length=16), server_default='chrome', nullable=False), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('resolved_at', sa.DateTime(timezone=True), nullable=True), + sa.ForeignKeyConstraint(['conflict_tag_id'], ['tag.id'], name=op.f('fk_presentation_review_conflict_tag_id_tag'), ondelete='SET NULL'), + sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_presentation_review_image_record_id_image_record'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_presentation_review_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_presentation_review')) + ) + op.create_table('series_page', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('series_tag_id', sa.Integer(), nullable=False), + sa.Column('image_id', sa.Integer(), nullable=False), + sa.Column('status', sa.String(length=16), server_default='placed', nullable=False), + sa.Column('page_number', sa.Integer(), nullable=True), + sa.Column('stated_page', sa.Integer(), nullable=True), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['image_id'], ['image_record.id'], name=op.f('fk_series_page_image_id_image_record'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_page_series_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_series_page')), + sa.UniqueConstraint('image_id', name=op.f('uq_series_page_image_id')) + ) + op.create_index(op.f('ix_series_page_series_tag_id'), 'series_page', ['series_tag_id'], unique=False) + op.create_table('tag_positive_confirmation', + sa.Column('image_record_id', sa.Integer(), nullable=False), + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('confirmed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_tag_positive_confirmation_image_record_id_image_record'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_positive_confirmation_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_positive_confirmation')) + ) + op.create_index(op.f('ix_tag_positive_confirmation_tag_id'), 'tag_positive_confirmation', ['tag_id'], unique=False) + op.create_table('tag_suggestion_rejection', + sa.Column('image_record_id', sa.Integer(), nullable=False), + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('rejected_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_tag_suggestion_rejection_image_record_id_image_record'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_suggestion_rejection_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_suggestion_rejection')) + ) + op.create_index(op.f('ix_tag_suggestion_rejection_tag_id'), 'tag_suggestion_rejection', ['tag_id'], unique=False) + op.create_table('character_prototype', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('tag_id', sa.Integer(), nullable=False), + sa.Column('ccip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=768), nullable=False), + sa.Column('region_id', sa.Integer(), nullable=True), + sa.ForeignKeyConstraint(['region_id'], ['image_region.id'], name=op.f('fk_character_prototype_region_id_image_region'), ondelete='SET NULL'), + sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_character_prototype_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_character_prototype')) + ) + op.create_index(op.f('ix_character_prototype_tag_id'), 'character_prototype', ['tag_id'], unique=False) + op.create_table('series_chapter', + sa.Column('id', sa.Integer(), nullable=False), + sa.Column('series_tag_id', sa.Integer(), nullable=False), + sa.Column('anchor_page_id', sa.Integer(), nullable=False), + sa.Column('title', sa.Text(), nullable=True), + sa.Column('stated_part', sa.Integer(), nullable=True), + sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False), + sa.ForeignKeyConstraint(['anchor_page_id'], ['series_page.id'], name=op.f('fk_series_chapter_anchor_page_id_series_page'), ondelete='CASCADE'), + sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_chapter_series_tag_id_tag'), ondelete='CASCADE'), + sa.PrimaryKeyConstraint('id', name=op.f('pk_series_chapter')), + sa.UniqueConstraint('anchor_page_id', name=op.f('uq_series_chapter_anchor_page_id')) + ) + op.create_index(op.f('ix_series_chapter_series_tag_id'), 'series_chapter', ['series_tag_id'], unique=False) + + # The HNSW index, item 3 above. Must match the query's cosine-distance + # operator class or the planner will not use it. + op.execute( + "CREATE INDEX ix_image_record_siglip_hnsw " + "ON image_record USING hnsw (siglip_embedding vector_cosine_ops)" + ) + + +def downgrade() -> None: + # Dropping image_record takes its indexes with it, so the HNSW index needs + # no separate drop. The extensions are deliberately left in place: they are + # database-scoped and something else may be using them. + op.drop_index(op.f('ix_series_chapter_series_tag_id'), table_name='series_chapter') + op.drop_table('series_chapter') + op.drop_index(op.f('ix_character_prototype_tag_id'), table_name='character_prototype') + op.drop_table('character_prototype') + op.drop_index(op.f('ix_tag_suggestion_rejection_tag_id'), table_name='tag_suggestion_rejection') + op.drop_table('tag_suggestion_rejection') + op.drop_index(op.f('ix_tag_positive_confirmation_tag_id'), table_name='tag_positive_confirmation') + op.drop_table('tag_positive_confirmation') + op.drop_index(op.f('ix_series_page_series_tag_id'), table_name='series_page') + op.drop_table('series_page') + op.drop_table('presentation_review') + op.drop_index(op.f('ix_import_task_status'), table_name='import_task') + op.drop_index(op.f('ix_import_task_batch_id'), table_name='import_task') + op.drop_table('import_task') + op.drop_table('image_tag') + op.drop_index(op.f('ix_image_region_image_record_id'), table_name='image_region') + op.drop_table('image_region') + op.drop_index(op.f('ix_image_provenance_source_id'), table_name='image_provenance') + op.drop_index(op.f('ix_image_provenance_post_id'), table_name='image_provenance') + op.drop_index(op.f('ix_image_provenance_image_record_id'), table_name='image_provenance') + op.drop_index(op.f('ix_image_provenance_from_attachment_id'), table_name='image_provenance') + op.drop_table('image_provenance') + op.drop_index(op.f('ix_gpu_job_status'), table_name='gpu_job') + op.drop_index('ix_gpu_job_pending', table_name='gpu_job', postgresql_where=sa.text("status = 'pending'")) + op.drop_index('ix_gpu_job_leased_expires', table_name='gpu_job', postgresql_where=sa.text("status = 'leased'")) + op.drop_index(op.f('ix_gpu_job_image_record_id'), table_name='gpu_job') + op.drop_table('gpu_job') + op.drop_index('uq_external_link_post_url', table_name='external_link') + op.drop_index('ix_external_link_status', table_name='external_link') + op.drop_index(op.f('ix_external_link_post_id'), table_name='external_link') + op.drop_index(op.f('ix_external_link_artist_id'), table_name='external_link') + op.drop_table('external_link') + op.drop_index(op.f('ix_series_suggestion_status'), table_name='series_suggestion') + op.drop_index(op.f('ix_series_suggestion_series_tag_id'), table_name='series_suggestion') + op.drop_index(op.f('ix_series_suggestion_post_id'), table_name='series_suggestion') + op.drop_table('series_suggestion') + op.drop_index('uq_post_attachment_post_sha', table_name='post_attachment', postgresql_where=sa.text('post_id IS NOT NULL')) + op.drop_index('uq_post_attachment_null_post_sha', table_name='post_attachment', postgresql_where=sa.text('post_id IS NULL')) + op.drop_index(op.f('ix_post_attachment_sha256'), table_name='post_attachment') + op.drop_index(op.f('ix_post_attachment_post_id'), table_name='post_attachment') + op.drop_index(op.f('ix_post_attachment_artist_id'), table_name='post_attachment') + op.drop_table('post_attachment') + op.drop_index(op.f('ix_image_record_source_filehash'), table_name='image_record') + op.drop_index(op.f('ix_image_record_sha256'), table_name='image_record') + op.drop_index(op.f('ix_image_record_primary_post_id'), table_name='image_record') + op.drop_index(op.f('ix_image_record_phash'), table_name='image_record') + op.drop_index(op.f('ix_image_record_integrity_status'), table_name='image_record') + op.drop_index(op.f('ix_image_record_artist_id'), table_name='image_record') + op.drop_table('image_record') + op.drop_index(op.f('ix_download_event_source_id'), table_name='download_event') + op.drop_index(op.f('ix_download_event_post_id'), table_name='download_event') + op.drop_table('download_event') + op.drop_index(op.f('ix_subscribestar_seen_media_source_id'), table_name='subscribestar_seen_media') + op.drop_table('subscribestar_seen_media') + op.drop_index(op.f('ix_subscribestar_failed_media_source_id'), table_name='subscribestar_failed_media') + op.drop_table('subscribestar_failed_media') + op.drop_index(op.f('ix_post_source_id'), table_name='post') + op.drop_index(op.f('ix_post_artist_id'), table_name='post') + op.drop_table('post') + op.drop_index(op.f('ix_pixiv_seen_media_source_id'), table_name='pixiv_seen_media') + op.drop_table('pixiv_seen_media') + op.drop_index(op.f('ix_pixiv_failed_media_source_id'), table_name='pixiv_failed_media') + op.drop_table('pixiv_failed_media') + op.drop_index(op.f('ix_patreon_seen_media_source_id'), table_name='patreon_seen_media') + op.drop_table('patreon_seen_media') + op.drop_index(op.f('ix_patreon_failed_media_source_id'), table_name='patreon_failed_media') + op.drop_table('patreon_failed_media') + op.drop_table('tag_head') + op.drop_index(op.f('ix_tag_alias_canonical_tag_id'), table_name='tag_alias') + op.drop_table('tag_alias') + op.drop_index(op.f('ix_source_error_type'), table_name='source') + op.drop_index(op.f('ix_source_artist_id'), table_name='source') + op.drop_table('source') + op.drop_index(op.f('ix_head_metrics_snapshot_tag_id'), table_name='head_metrics_snapshot') + op.drop_index(op.f('ix_head_metrics_snapshot_snapshot_at'), table_name='head_metrics_snapshot') + op.drop_table('head_metrics_snapshot') + op.drop_table('head_metric') + op.drop_table('ccip_prototype_state') + op.drop_table('artist_visit') + op.drop_index(op.f('ix_task_run_task_name'), table_name='task_run') + op.drop_index(op.f('ix_task_run_status'), table_name='task_run') + op.drop_index(op.f('ix_task_run_started_at'), table_name='task_run') + op.drop_index(op.f('ix_task_run_queue'), table_name='task_run') + op.drop_index(op.f('ix_task_run_finished_at'), table_name='task_run') + op.drop_index(op.f('ix_task_run_celery_task_id'), table_name='task_run') + op.drop_table('task_run') + op.drop_index(op.f('ix_tag_fandom_id'), table_name='tag') + op.drop_table('tag') + op.drop_table('ml_settings') + op.drop_index(op.f('ix_library_audit_run_status'), table_name='library_audit_run') + op.drop_index(op.f('ix_library_audit_run_rule'), table_name='library_audit_run') + op.drop_table('library_audit_run') + op.drop_table('import_settings') + op.drop_index(op.f('ix_import_batch_status'), table_name='import_batch') + op.drop_table('import_batch') + op.drop_index(op.f('ix_head_training_run_status'), table_name='head_training_run') + op.drop_table('head_training_run') + op.drop_index(op.f('ix_head_auto_apply_run_status'), table_name='head_auto_apply_run') + op.drop_table('head_auto_apply_run') + op.drop_table('credential') + op.drop_index(op.f('ix_backup_run_tag'), table_name='backup_run') + op.drop_index(op.f('ix_backup_run_status'), table_name='backup_run') + op.drop_index(op.f('ix_backup_run_started_at'), table_name='backup_run') + op.drop_index(op.f('ix_backup_run_kind'), table_name='backup_run') + op.drop_index(op.f('ix_backup_run_finished_at'), table_name='backup_run') + op.drop_table('backup_run') + op.drop_table('artist') + op.drop_table('app_setting') diff --git a/alembic/versions/0087_wip_soft_title_tagging.py b/alembic/versions/0087_wip_soft_title_tagging.py deleted file mode 100644 index 58478e1..0000000 --- a/alembic/versions/0087_wip_soft_title_tagging.py +++ /dev/null @@ -1,33 +0,0 @@ -"""soft WIP title tier toggle (#1474) — ImportSettings.wip_soft_title_tagging_enabled - -The soft tier also tags sketch/doodle/scribble titles, but with a provisional source -that never trains the head. OFF by default (a lower-precision tier is opt-in). -server_default so the existing singleton row (id=1) fills cleanly. - -Revision ID: 0087 -Revises: 0086 -Create Date: 2026-07-13 -""" -from typing import Sequence, Union - -import sqlalchemy as sa -from alembic import op - -revision: str = "0087" -down_revision: Union[str, None] = "0086" -branch_labels: Union[str, Sequence[str], None] = None -depends_on: Union[str, Sequence[str], None] = None - - -def upgrade() -> None: - op.add_column( - "import_settings", - sa.Column( - "wip_soft_title_tagging_enabled", sa.Boolean(), nullable=False, - server_default=sa.text("false"), - ), - ) - - -def downgrade() -> None: - op.drop_column("import_settings", "wip_soft_title_tagging_enabled") diff --git a/backend/app/utils/artist_backfill.py b/backend/app/utils/artist_backfill.py deleted file mode 100644 index 3979b24..0000000 --- a/backend/app/utils/artist_backfill.py +++ /dev/null @@ -1,44 +0,0 @@ -"""Literal SQL for the FC-2d-vii-c artist backfill / artist-tag delete. - -Intentionally pure string constants — NO model/slug imports, NO logic — -so migration 0008 and its test share one drift-proof source of truth. -Backfill steps are ordered primary -> provenance -> artist-tag and each -only touches rows still NULL (idempotent, first match wins). The -artist-tag step matches Artist.name = Tag.name: the importer always -created both from the same artist_name string. -""" - -BACKFILL_PRIMARY_SQL = """ -UPDATE image_record AS ir -SET artist_id = s.artist_id -FROM post p -JOIN source s ON s.id = p.source_id -WHERE ir.primary_post_id = p.id - AND ir.artist_id IS NULL -""" - -BACKFILL_PROVENANCE_SQL = """ -UPDATE image_record AS ir -SET artist_id = s.artist_id -FROM ( - SELECT DISTINCT ON (ip.image_record_id) - ip.image_record_id, src.artist_id - FROM image_provenance ip - JOIN source src ON src.id = ip.source_id - ORDER BY ip.image_record_id, ip.id -) AS s -WHERE ir.id = s.image_record_id - AND ir.artist_id IS NULL -""" - -BACKFILL_TAG_SQL = """ -UPDATE image_record AS ir -SET artist_id = a.id -FROM image_tag it -JOIN tag t ON t.id = it.tag_id AND t.kind = 'artist' -JOIN artist a ON a.name = t.name -WHERE it.image_record_id = ir.id - AND ir.artist_id IS NULL -""" - -DELETE_ARTIST_TAGS_SQL = "DELETE FROM tag WHERE kind = 'artist'" diff --git a/tests/test_migration_0002.py b/tests/test_migration_0002.py deleted file mode 100644 index 1786731..0000000 --- a/tests/test_migration_0002.py +++ /dev/null @@ -1,58 +0,0 @@ -"""Smoke test for migration 0002: confirms model classes import and the -tag-kind uniqueness rule shape is correct. -""" - -from backend.app.models import ( - Base, - ImportBatch, - ImportSettings, - ImportTask, - Tag, - TagKind, -) - - -def test_new_tables_registered(): - expected = {"import_batch", "import_task", "import_settings"} - assert expected.issubset(Base.metadata.tables.keys()) - - -def test_tag_has_kind_and_fandom_id(): - cols = {c.name for c in Tag.__table__.columns} - assert "kind" in cols - assert "fandom_id" in cols - assert "namespace" not in cols - - -def test_tag_kind_enum_values(): - # Current TagKind enum after alembic 0023 dropped meta + rating - # (operator-retired 2026-05-26). `artist` is still in the enum - # for backward-compat with historical rows, though new artist - # tags don't get created (Artist row is canonical per FC-2d-vii-c). - expected = { - "artist", - "character", - "fandom", - "general", - "series", - "archive", - "post", - } - assert {k.value for k in TagKind} == expected - - -def test_image_record_has_integrity_status(): - from backend.app.models import ImageRecord - cols = {c.name for c in ImageRecord.__table__.columns} - assert "integrity_status" in cols - - -def test_import_task_has_state_columns(): - cols = {c.name for c in ImportTask.__table__.columns} - for required in ("batch_id", "source_path", "task_type", "status", "result_image_id"): - assert required in cols - - -def test_import_settings_singleton_constraint(): - constraints = {c.name for c in ImportSettings.__table__.constraints} - assert "ck_import_settings_singleton" in constraints diff --git a/tests/test_migration_0003.py b/tests/test_migration_0003.py deleted file mode 100644 index 1752120..0000000 --- a/tests/test_migration_0003.py +++ /dev/null @@ -1,46 +0,0 @@ -"""Smoke test for migration 0003: model classes import, schema shape correct.""" - -from backend.app.models import ( - Base, - ImageRecord, - MLSettings, - TagAlias, - TagSuggestionRejection, -) - - -def test_new_tables_registered(): - expected = { - "tag_suggestion_rejection", - "tag_alias", - "ml_settings", - } - assert expected.issubset(Base.metadata.tables.keys()) - - -def test_image_record_columns_renamed(): - cols = {c.name for c in ImageRecord.__table__.columns} - # Legacy tagger columns are all gone: tagger_predictions/wd14_* dropped in - # 0046, tagger_model_version + centroid_scores dropped in 0068 (#1199, Camie - # retirement). The SigLIP embedding columns are the live ML fields. - assert "siglip_embedding" in cols - assert "siglip_model_version" in cols - assert "tagger_model_version" not in cols - assert "centroid_scores" not in cols - assert "tagger_predictions" not in cols - assert "wd14_predictions" not in cols - - -def test_tag_alias_composite_pk(): - pk_cols = {c.name for c in TagAlias.__table__.primary_key.columns} - assert pk_cols == {"alias_string", "alias_category"} - - -def test_ml_settings_singleton_constraint(): - names = {c.name for c in MLSettings.__table__.constraints} - assert "ck_ml_settings_singleton" in names - - -def test_tag_suggestion_rejection_pk(): - pk_cols = {c.name for c in TagSuggestionRejection.__table__.primary_key.columns} - assert pk_cols == {"image_record_id", "tag_id"} diff --git a/tests/test_migration_0004.py b/tests/test_migration_0004.py deleted file mode 100644 index 7975d0c..0000000 --- a/tests/test_migration_0004.py +++ /dev/null @@ -1,27 +0,0 @@ -"""Integration: the tsm_system_rows extension is installed by migration 0004. - -Needs a real Postgres (CI does not provision one), so integration-marked. -""" - -import pytest -from sqlalchemy import text - -pytestmark = pytest.mark.integration - - -@pytest.mark.asyncio -async def test_tsm_system_rows_extension_present(db): - row = ( - await db.execute( - text("SELECT 1 FROM pg_extension WHERE extname = 'tsm_system_rows'") - ) - ).first() - assert row is not None - - -@pytest.mark.asyncio -async def test_system_rows_sampling_is_usable(db): - # Should parse and execute even on an empty table. - await db.execute( - text("SELECT * FROM image_record TABLESAMPLE SYSTEM_ROWS(1)") - ) diff --git a/tests/test_migration_0007.py b/tests/test_migration_0007.py deleted file mode 100644 index 6f716f4..0000000 --- a/tests/test_migration_0007.py +++ /dev/null @@ -1,48 +0,0 @@ -"""FC-2d-iv: post.description + post.attachment_count round-trip.""" - -from datetime import UTC, datetime - -import pytest - -from backend.app.models import Artist, Post, Source - -pytestmark = pytest.mark.integration - - -async def _post(db, **post_kwargs): - artist = Artist(name="Nadia", slug="nadia") - db.add(artist) - await db.flush() - src = Source(artist_id=artist.id, platform="web", url="http://x") - db.add(src) - await db.flush() - post = Post( - source_id=src.id, artist_id=artist.id, external_post_id="p1", - post_date=datetime(2026, 3, 1, tzinfo=UTC), - **post_kwargs, - ) - db.add(post) - await db.flush() - return post.id - - -def test_post_has_new_columns(): - cols = {c.name for c in Post.__table__.columns} - assert "description" in cols - assert "attachment_count" in cols - - -@pytest.mark.asyncio -async def test_description_and_attachment_count_round_trip(db): - pid = await _post(db, description="

hi

", attachment_count=3) - row = await db.get(Post, pid) - assert row.description == "

hi

" - assert row.attachment_count == 3 - - -@pytest.mark.asyncio -async def test_new_fields_default_null(db): - pid = await _post(db) - row = await db.get(Post, pid) - assert row.description is None - assert row.attachment_count is None diff --git a/tests/test_migration_0008.py b/tests/test_migration_0008.py deleted file mode 100644 index 0d7552a..0000000 --- a/tests/test_migration_0008.py +++ /dev/null @@ -1,137 +0,0 @@ -"""FC-2d-vii-c: image_record.artist_id + backfill + artist-tag delete.""" - -from datetime import UTC, datetime, timedelta - -import pytest -from sqlalchemy import func, select, text - -from backend.app.models import ( - Artist, - ImageProvenance, - ImageRecord, - Post, - Source, - Tag, - TagKind, -) -from backend.app.models.tag import image_tag -from backend.app.utils.artist_backfill import ( - BACKFILL_PRIMARY_SQL, - BACKFILL_PROVENANCE_SQL, - BACKFILL_TAG_SQL, - DELETE_ARTIST_TAGS_SQL, -) - -pytestmark = pytest.mark.integration - - -def test_image_record_has_artist_id_column(): - assert "artist_id" in {c.name for c in ImageRecord.__table__.columns} - - -async def _img(db, n): - rec = ImageRecord( - path=f"/images/bf/{n}.jpg", sha256=f"bf{n:062d}", - size_bytes=1, mime="image/jpeg", width=1, height=1, - origin="imported_filesystem", integrity_status="unknown", - ) - rec.created_at = datetime.now(UTC) - timedelta(minutes=n) - db.add(rec) - await db.flush() - return rec - - -async def _artist_source(db, name, slug): - a = Artist(name=name, slug=slug) - db.add(a) - await db.flush() - s = Source(artist_id=a.id, platform="patreon", - url=f"https://p.test/{slug}") - db.add(s) - await db.flush() - return a, s - - -async def _run_backfill(db): - await db.execute(text(BACKFILL_PRIMARY_SQL)) - await db.execute(text(BACKFILL_PROVENANCE_SQL)) - await db.execute(text(BACKFILL_TAG_SQL)) - - -@pytest.mark.asyncio -async def test_backfill_primary_post(db): - rec = await _img(db, 1) - a, s = await _artist_source(db, "Alice", "alice") - post = Post(source_id=s.id, artist_id=a.id, external_post_id="1") - db.add(post) - await db.flush() - rec.primary_post_id = post.id - await db.flush() - await _run_backfill(db) - got = await db.scalar( - select(ImageRecord.artist_id).where(ImageRecord.id == rec.id) - ) - assert got == a.id - - -@pytest.mark.asyncio -async def test_backfill_provenance_fallback(db): - rec = await _img(db, 1) - a, s = await _artist_source(db, "Bob", "bob") - post = Post(source_id=s.id, artist_id=a.id, external_post_id="2") - db.add(post) - await db.flush() - db.add(ImageProvenance(image_record_id=rec.id, post_id=post.id, - source_id=s.id)) - await db.flush() - await _run_backfill(db) - got = await db.scalar( - select(ImageRecord.artist_id).where(ImageRecord.id == rec.id) - ) - assert got == a.id - - -@pytest.mark.asyncio -async def test_backfill_artist_tag_by_name(db): - rec = await _img(db, 1) - a = Artist(name="Carol", slug="carol") - db.add(a) - await db.flush() - tag = Tag(name="Carol", kind=TagKind.artist) - db.add(tag) - await db.flush() - await db.execute(image_tag.insert().values( - image_record_id=rec.id, tag_id=tag.id, source="auto")) - await db.flush() - await _run_backfill(db) - got = await db.scalar( - select(ImageRecord.artist_id).where(ImageRecord.id == rec.id) - ) - assert got == a.id - - -@pytest.mark.asyncio -async def test_no_signal_stays_null(db): - rec = await _img(db, 1) - await _run_backfill(db) - got = await db.scalar( - select(ImageRecord.artist_id).where(ImageRecord.id == rec.id) - ) - assert got is None - - -@pytest.mark.asyncio -async def test_delete_removes_only_artist_tags(db): - artist_tag = Tag(name="Dave", kind=TagKind.artist) - general_tag = Tag(name="forest", kind=TagKind.general) - db.add_all([artist_tag, general_tag]) - await db.flush() - await db.execute(text(DELETE_ARTIST_TAGS_SQL)) - remaining = await db.scalar( - select(func.count()).select_from(Tag).where(Tag.kind == TagKind.artist) - ) - assert remaining == 0 - survived = await db.scalar( - select(func.count()).select_from(Tag).where(Tag.kind == TagKind.general) - ) - assert survived >= 1 diff --git a/tests/test_migration_0009.py b/tests/test_migration_0009.py deleted file mode 100644 index f2a0d1f..0000000 --- a/tests/test_migration_0009.py +++ /dev/null @@ -1,37 +0,0 @@ -"""FC-2d-iii: post_attachment table + import_batch.attachments column.""" - -import pytest - -from backend.app.models import ImportBatch, PostAttachment - -pytestmark = pytest.mark.integration - - -def test_post_attachment_columns(): - cols = {c.name for c in PostAttachment.__table__.columns} - assert { - "id", "post_id", "artist_id", "sha256", "path", - "original_filename", "ext", "mime", "size_bytes", "captured_at", - } <= cols - - -def test_import_batch_has_attachments_counter(): - assert "attachments" in {c.name for c in ImportBatch.__table__.columns} - - -@pytest.mark.asyncio -async def test_post_attachment_roundtrip(db): - from backend.app.models import Artist - - a = Artist(name="Zed", slug="zed") - db.add(a) - await db.flush() - att = PostAttachment( - post_id=None, artist_id=a.id, sha256="z" + "0" * 63, - path="/images/attachments/z00/z.zip", original_filename="pack.zip", - ext=".zip", mime="application/zip", size_bytes=123, - ) - db.add(att) - await db.flush() - got = await db.get(PostAttachment, att.id) - assert got.original_filename == "pack.zip" and got.post_id is None diff --git a/tests/test_migration_0010.py b/tests/test_migration_0010.py deleted file mode 100644 index a261e6c..0000000 --- a/tests/test_migration_0010.py +++ /dev/null @@ -1,35 +0,0 @@ -import pytest -from sqlalchemy.exc import IntegrityError - -from backend.app.models import Artist, Source - -pytestmark = pytest.mark.integration - - -@pytest.mark.asyncio -async def test_duplicate_artist_platform_url_rejected(db): - artist = Artist(name="Alice", slug="alice") - db.add(artist) - await db.flush() - db.add(Source( - artist_id=artist.id, platform="patreon", - url="https://patreon.com/alice", enabled=True, - )) - await db.flush() - db.add(Source( - artist_id=artist.id, platform="patreon", - url="https://patreon.com/alice", enabled=True, - )) - with pytest.raises(IntegrityError): - await db.flush() - - -@pytest.mark.asyncio -async def test_same_url_under_different_artist_ok(db): - a = Artist(name="A", slug="a") - b = Artist(name="B", slug="b") - db.add_all([a, b]) - await db.flush() - db.add(Source(artist_id=a.id, platform="patreon", url="https://x/y", enabled=True)) - db.add(Source(artist_id=b.id, platform="patreon", url="https://x/y", enabled=True)) - await db.flush() # must NOT raise diff --git a/tests/test_migration_0011.py b/tests/test_migration_0011.py deleted file mode 100644 index 17e9702..0000000 --- a/tests/test_migration_0011.py +++ /dev/null @@ -1,32 +0,0 @@ -import pytest -from sqlalchemy import inspect, text - -pytestmark = pytest.mark.integration - - -@pytest.mark.asyncio -async def test_credential_has_credential_type_not_kind(db): - cols = (await db.run_sync( - lambda sync_session: [c["name"] for c in inspect(sync_session.bind).get_columns("credential")] - )) - assert "credential_type" in cols - assert "kind" not in cols - assert "status" not in cols - assert "last_verified" in cols - - -@pytest.mark.asyncio -async def test_credential_round_trip(db): - from backend.app.models import Credential - - db.add(Credential( - platform="patreon", - credential_type="cookies", - encrypted_blob=b"\x00\x01\x02", - )) - await db.flush() - row = (await db.execute( - text("SELECT credential_type, last_verified FROM credential WHERE platform='patreon'") - )).one() - assert row.credential_type == "cookies" - assert row.last_verified is None diff --git a/tests/test_migration_0012.py b/tests/test_migration_0012.py deleted file mode 100644 index a993ab4..0000000 --- a/tests/test_migration_0012.py +++ /dev/null @@ -1,32 +0,0 @@ -import pytest -from sqlalchemy import select - -from backend.app.models import AppSetting - -pytestmark = pytest.mark.integration - - -@pytest.mark.asyncio -async def test_app_setting_table_round_trip(db): - db.add(AppSetting(key="extension_api_key", value="abc123")) - await db.flush() - row = (await db.execute( - select(AppSetting).where(AppSetting.key == "extension_api_key") - )).scalar_one() - assert row.value == "abc123" - assert row.updated_at is not None - - -@pytest.mark.asyncio -async def test_app_setting_upsert(db): - db.add(AppSetting(key="k", value="v1")) - await db.flush() - row = (await db.execute( - select(AppSetting).where(AppSetting.key == "k") - )).scalar_one() - row.value = "v2" - await db.flush() - again = (await db.execute( - select(AppSetting.value).where(AppSetting.key == "k") - )).scalar_one() - assert again == "v2" diff --git a/tests/test_migration_0013.py b/tests/test_migration_0013.py deleted file mode 100644 index 1a18384..0000000 --- a/tests/test_migration_0013.py +++ /dev/null @@ -1,31 +0,0 @@ -import pytest -from sqlalchemy import inspect, select - -from backend.app.models import ImportSettings - -pytestmark = pytest.mark.integration - - -@pytest.mark.asyncio -async def test_download_event_has_metadata(db): - cols = await db.run_sync( - lambda s: {c["name"]: c for c in inspect(s.bind).get_columns("download_event")} - ) - assert "metadata" in cols - assert cols["metadata"]["nullable"] is False - - -@pytest.mark.asyncio -async def test_import_settings_has_downloader_fields(db): - cols = await db.run_sync( - lambda s: {c["name"]: c for c in inspect(s.bind).get_columns("import_settings")} - ) - assert "download_rate_limit_seconds" in cols - assert "download_validate_files" in cols - - -@pytest.mark.asyncio -async def test_import_settings_defaults(db): - row = (await db.execute(select(ImportSettings).where(ImportSettings.id == 1))).scalar_one() - assert row.download_rate_limit_seconds == 3.0 - assert row.download_validate_files is True