This reverts 2529b51. Not a retreat — a reordering, on the operator's
call, and the better sequence.
The squash's acceptance test (run 4971) found ~130 places where the ORM
models do not describe the deployed schema (#3275), including a
unique=True the database never had and two UNIQUE indexes that exist
only in migrations. Collapsing now would have baked all of that into the
one file a public installer starts from.
So: fix the drift first as ordinary migrations on the intact chain, let
the operator deploy so their database moves to the corrected head, and
only then collapse. The baseline is then generated from reconciled
models and reproduces a schema worth reproducing.
Nothing is lost by reverting. The baseline was never deployed, and
regenerating it after the fixes is strictly better than patching this
copy — it will come out of autogenerate correct rather than needing the
same hand-finishing twice.
289 lines
11 KiB
Python
289 lines
11 KiB
Python
"""sidecar-audit followup: correct external_post_id + post_url across all platforms
|
|
|
|
Revision ID: 0025
|
|
Revises: 0024
|
|
Create Date: 2026-05-27
|
|
|
|
Closes the operator-flagged 2026-05-27 sidecar audit findings. Three
|
|
data-correctness bugs across non-Patreon platforms had been silently
|
|
corrupting Posts since FC-3 shipped; the parser fix (sidecar.py, same
|
|
commit) addresses new imports. This migration cleans up existing rows.
|
|
|
|
Per-platform actions:
|
|
|
|
subscribestar — gallery-dl wrote the per-attachment id in `id` and
|
|
the actual post id in `post_id`. FC's parser picked `id`, so every
|
|
multi-image SubscribeStar post was fragmented into N Post rows.
|
|
1. For each SubscribeStar Post, read its sidecar (via the related
|
|
ImageRecord's on-disk path), pull `post_id`, overwrite
|
|
external_post_id and post_url.
|
|
2. Merge groups of Posts under one source that now share an
|
|
external_post_id (fragments of the same actual post). Same
|
|
ImageProvenance pre-delete + repoint dance as alembic 0022.
|
|
|
|
hentaifoundry — sidecars have NO `url` field; `src` is the image
|
|
URL. FC's parser stored post_url=NULL. Read each HF Post's sidecar
|
|
for `user` + `index`, derive the canonical /pictures/user/<u>/<i>
|
|
permalink. external_post_id (= `index`) was already correct.
|
|
|
|
discord — gallery-dl wrote the CDN attachment URL in `url`. FC's
|
|
parser stored that as post_url. Read each Discord Post's sidecar
|
|
for the server/channel/message triple, derive the proper
|
|
discord.com/channels/.../<message> permalink. external_post_id (=
|
|
`message_id`) was already correct.
|
|
|
|
pixiv — pure-SQL backfill: replace any `i.pximg.net`-style URL on
|
|
Post.post_url with the derived `/artworks/<id>` permalink. Pixiv
|
|
external_post_id (= `id`) was already correct; no sidecar IO
|
|
needed.
|
|
|
|
Idempotent: re-running on already-corrected data is a no-op (skips
|
|
rows whose derived value matches what's already stored).
|
|
|
|
Posts whose related ImageRecord paths don't resolve on disk (orphaned
|
|
filesystem state) are skipped with a count in the migration output —
|
|
those will be picked up by a future deep-scan.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from pathlib import Path
|
|
from typing import Sequence, Union
|
|
|
|
from alembic import op
|
|
from sqlalchemy import text
|
|
|
|
revision: str = "0025"
|
|
down_revision: Union[str, None] = "0024"
|
|
branch_labels: Union[str, Sequence[str], None] = None
|
|
depends_on: Union[str, Sequence[str], None] = None
|
|
|
|
|
|
# Mirror of sidecar._NUMBERING_PREFIX. Kept inline so the migration is
|
|
# self-contained (the operator's banked rule:
|
|
# reference_postgres_enum_swap_drop_checks.md says migrations shouldn't
|
|
# import from runtime app code).
|
|
_NUMBERING_PREFIX = re.compile(r"^\d+_(.+)$")
|
|
|
|
|
|
def _find_sidecar(media_path: Path) -> Path | None:
|
|
"""gallery-dl writes the sidecar under the unprefixed stem
|
|
(`HOLLOW-ICHIGO.json`) while the media file gets a NN_ ordering
|
|
prefix (`01_HOLLOW-ICHIGO.png`). Try in order:
|
|
1. <stem>.json next to the media
|
|
2. <media>.json next to the media (full-name variant)
|
|
3. strip the NN_ prefix from the stem, then <stripped>.json
|
|
"""
|
|
if not media_path:
|
|
return None
|
|
cand = media_path.with_suffix(".json")
|
|
if cand.is_file():
|
|
return cand
|
|
cand = media_path.parent / f"{media_path.name}.json"
|
|
if cand.is_file():
|
|
return cand
|
|
m = _NUMBERING_PREFIX.match(media_path.stem)
|
|
if m:
|
|
cand = media_path.parent / f"{m.group(1)}.json"
|
|
if cand.is_file():
|
|
return cand
|
|
return None
|
|
|
|
|
|
def _str_id(v) -> str | None:
|
|
"""str() a JSON scalar id; reject bool (JSON booleans are ints in
|
|
Python's eyes but they aren't valid sidecar ids)."""
|
|
if isinstance(v, bool):
|
|
return None
|
|
if isinstance(v, (str, int)) and str(v).strip():
|
|
return str(v).strip()
|
|
return None
|
|
|
|
|
|
def _str_field(v) -> str | None:
|
|
if isinstance(v, str) and v.strip():
|
|
return v.strip()
|
|
return None
|
|
|
|
|
|
def upgrade() -> None:
|
|
conn = op.get_bind()
|
|
|
|
# ── PART 1: Per-platform corrections requiring filesystem IO ─────
|
|
# SubscribeStar, HentaiFoundry, Discord all need fields from the
|
|
# sidecar to construct the right post_url. We walk each Post's
|
|
# related ImageRecord.path to find the sidecar, read it, derive,
|
|
# and update.
|
|
targets = conn.execute(text("""
|
|
SELECT p.id, p.external_post_id, p.post_url, s.platform
|
|
FROM post p
|
|
JOIN source s ON s.id = p.source_id
|
|
WHERE s.platform IN ('subscribestar', 'hentaifoundry', 'discord')
|
|
""")).fetchall()
|
|
|
|
stats: dict[str, dict[str, int]] = {
|
|
plat: {"read": 0, "updated": 0, "no_sidecar": 0}
|
|
for plat in ("subscribestar", "hentaifoundry", "discord")
|
|
}
|
|
for post_row in targets:
|
|
plat = post_row.platform
|
|
path = _first_attachment_path(conn, post_row.id)
|
|
if not path:
|
|
stats[plat]["no_sidecar"] += 1
|
|
continue
|
|
sidecar = _find_sidecar(Path(path))
|
|
if sidecar is None:
|
|
stats[plat]["no_sidecar"] += 1
|
|
continue
|
|
try:
|
|
data = json.loads(sidecar.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
stats[plat]["no_sidecar"] += 1
|
|
continue
|
|
stats[plat]["read"] += 1
|
|
|
|
new_epid = post_row.external_post_id
|
|
new_url = None
|
|
if plat == "subscribestar":
|
|
pid = _str_id(data.get("post_id"))
|
|
if pid:
|
|
new_epid = pid
|
|
new_url = f"https://www.subscribestar.com/posts/{pid}"
|
|
elif plat == "hentaifoundry":
|
|
user = _str_field(data.get("user")) or _str_field(data.get("artist"))
|
|
idx = _str_id(data.get("index"))
|
|
if user and idx:
|
|
new_url = f"https://www.hentai-foundry.com/pictures/user/{user}/{idx}"
|
|
elif plat == "discord":
|
|
sid = _str_id(data.get("server_id"))
|
|
cid = _str_id(data.get("channel_id"))
|
|
mid = _str_id(data.get("message_id"))
|
|
if sid and cid and mid:
|
|
new_url = f"https://discord.com/channels/{sid}/{cid}/{mid}"
|
|
|
|
# Idempotent: skip if nothing changed.
|
|
if new_epid == post_row.external_post_id and new_url == post_row.post_url:
|
|
continue
|
|
conn.execute(
|
|
text("""
|
|
UPDATE post
|
|
SET external_post_id = :epid, post_url = :url
|
|
WHERE id = :id
|
|
"""),
|
|
{"epid": new_epid, "url": new_url, "id": post_row.id},
|
|
)
|
|
stats[plat]["updated"] += 1
|
|
|
|
for plat, s in stats.items():
|
|
print(
|
|
f"0025: {plat} — read {s['read']} sidecars, "
|
|
f"updated {s['updated']} Posts, "
|
|
f"{s['no_sidecar']} Posts had no resolvable sidecar"
|
|
)
|
|
|
|
# ── PART 2: Merge SubscribeStar fragments now sharing epid ───────
|
|
# After Part 1, each group of Posts under one source with the SAME
|
|
# new external_post_id is a fragment-set of the same actual post.
|
|
# Merge to one canonical row. Pre-handle the same ImageProvenance
|
|
# collision pattern as alembic 0022 (uq_image_provenance_image_post).
|
|
fragment_groups = conn.execute(text("""
|
|
SELECT p.source_id, p.external_post_id,
|
|
ARRAY_AGG(p.id ORDER BY p.id ASC) AS post_ids
|
|
FROM post p
|
|
JOIN source s ON s.id = p.source_id
|
|
WHERE s.platform = 'subscribestar'
|
|
AND p.external_post_id IS NOT NULL
|
|
GROUP BY p.source_id, p.external_post_id
|
|
HAVING COUNT(*) > 1
|
|
""")).fetchall()
|
|
|
|
merged = 0
|
|
for grp in fragment_groups:
|
|
post_ids = list(grp.post_ids)
|
|
keep_id, *drop_ids = post_ids
|
|
for drop_id in drop_ids:
|
|
# Pre-DELETE colliding ImageProvenance under drop_ that
|
|
# already exist under keep (alembic 0022 banked the pattern).
|
|
conn.execute(
|
|
text("""
|
|
DELETE FROM image_provenance
|
|
WHERE post_id = :drop_
|
|
AND image_record_id IN (
|
|
SELECT image_record_id FROM image_provenance
|
|
WHERE post_id = :keep
|
|
)
|
|
"""),
|
|
{"keep": keep_id, "drop_": drop_id},
|
|
)
|
|
conn.execute(
|
|
text("""
|
|
UPDATE image_provenance SET post_id = :keep
|
|
WHERE post_id = :drop_
|
|
"""),
|
|
{"keep": keep_id, "drop_": drop_id},
|
|
)
|
|
conn.execute(
|
|
text("""
|
|
UPDATE image_record SET primary_post_id = :keep
|
|
WHERE primary_post_id = :drop_
|
|
"""),
|
|
{"keep": keep_id, "drop_": drop_id},
|
|
)
|
|
conn.execute(
|
|
text("""
|
|
UPDATE post_attachment SET post_id = :keep
|
|
WHERE post_id = :drop_
|
|
"""),
|
|
{"keep": keep_id, "drop_": drop_id},
|
|
)
|
|
conn.execute(
|
|
text("DELETE FROM post WHERE id = :drop_"),
|
|
{"drop_": drop_id},
|
|
)
|
|
merged += 1
|
|
print(f"0025: subscribestar — merged {merged} duplicate Post fragments")
|
|
|
|
# ── PART 3: Pixiv post_url backfill (pure SQL) ───────────────────
|
|
# Pixiv's external_post_id is already correct (gallery-dl's `id` is
|
|
# the post id). Only post_url needs derivation: replace anything
|
|
# under i.pximg.net (the file URL) with the /artworks/<id> permalink.
|
|
pixiv_updated = conn.execute(text("""
|
|
UPDATE post p
|
|
SET post_url = 'https://www.pixiv.net/artworks/' || p.external_post_id
|
|
FROM source s
|
|
WHERE p.source_id = s.id
|
|
AND s.platform = 'pixiv'
|
|
AND p.external_post_id IS NOT NULL
|
|
AND (p.post_url IS NULL
|
|
OR p.post_url LIKE 'https://i.pximg.net/%'
|
|
OR p.post_url LIKE 'http://i.pximg.net/%')
|
|
""")).rowcount
|
|
print(f"0025: pixiv — backfilled post_url on {pixiv_updated} Posts")
|
|
|
|
|
|
def _first_attachment_path(conn, post_id: int) -> str | None:
|
|
"""Return any ImageRecord.path attached to this post (via
|
|
ImageProvenance). Lowest-id row keeps the migration deterministic
|
|
so re-running on the same DB picks the same sidecar."""
|
|
row = conn.execute(
|
|
text("""
|
|
SELECT ir.path
|
|
FROM image_provenance ip
|
|
JOIN image_record ir ON ir.id = ip.image_record_id
|
|
WHERE ip.post_id = :pid
|
|
ORDER BY ip.id ASC
|
|
LIMIT 1
|
|
"""),
|
|
{"pid": post_id},
|
|
).first()
|
|
return row[0] if row else None
|
|
|
|
|
|
def downgrade() -> None:
|
|
# Lossy: external_post_id values were overwritten with the correct
|
|
# post_id; original per-attachment ids weren't preserved. Post-merge
|
|
# also deleted drop rows. No safe restore. To roll back the schema
|
|
# invariant, fork from 0024 and re-run sidecar imports.
|
|
pass
|