Compare commits
356 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| dfc3922d24 | |||
| 96c29c370b | |||
| 5e1655384f | |||
| 8dbf29f803 | |||
| 05f226a8f6 | |||
| bd2807cdd1 | |||
| 82b26b8aaa | |||
| 96e984cded | |||
| 13253b18d1 | |||
| 896e4f248c | |||
| d96918d777 | |||
| c342c73a25 | |||
| ca25f688c3 | |||
| 796e92540a | |||
| 2c67c27044 | |||
| 7fcef53d5b | |||
| 3eb08e926b | |||
| 5c3f8ebd70 | |||
| 9e81ced359 | |||
| 7c4b24c80d | |||
| 11e9f5af60 | |||
| 3e1303ea3c | |||
| 909fa37b15 | |||
| 2c544ad5af | |||
| 90c68f8b2a | |||
| 3e22e78aa4 | |||
| 013b9d7f06 | |||
| dfab8f65ff | |||
| 7bb765b6ed | |||
| 59746d213d | |||
| 618f7cdc36 | |||
| 3610ba495f | |||
| 028ea33a7c | |||
| 65211a3f2f | |||
| 444c1fb075 | |||
| e6d5f67f11 | |||
| 26c68b0a75 | |||
| a712cef92d | |||
| e75427b19a | |||
| 75eab188c8 | |||
| 0319812b45 | |||
| 22cdf0f334 | |||
| 79089b50b0 | |||
| 5447fab987 | |||
| 7a40a50fe9 | |||
| d55e52ae9b | |||
| c8b815afe6 | |||
| 3f92669f12 | |||
| 9ba3db75fd | |||
| 70d4017cf6 | |||
| bad37e07b2 | |||
| 14c244bd3d | |||
| 074c5868fb | |||
| f1a664e5a7 | |||
| 7b2a2051e9 | |||
| 9deebfa133 | |||
| 4854d74c5a | |||
| 409bbd43db | |||
| 4e83b4225a | |||
| c774042a85 | |||
| d5d23a92f2 | |||
| 2bfc9936a1 | |||
| a50902071a | |||
| c999c64cbe | |||
| 978f49adcc | |||
| e4cebf70d1 | |||
| 4958e8f7d4 | |||
| 4c6406ee18 | |||
| a8f624a0f1 | |||
| bb47e80b3e | |||
| df76bc0f58 | |||
| de4ef6ae74 | |||
| dc1083b5e0 | |||
| 408fcd488a | |||
| f2fbe2ae6e | |||
| b1778ca9f2 | |||
| fe0ed52595 | |||
| fb05c5eef7 | |||
| e46893fefd | |||
| e90e6b2c34 | |||
| 8e98e79968 | |||
| 0666e15211 | |||
| a00a2786e3 | |||
| 9770dd3474 | |||
| 7daf90f41e | |||
| aaa375654b | |||
| 5bc8ef65ad | |||
| 978959bdc4 | |||
| 7309d1d6d4 | |||
| 747390631d | |||
| daaa7543a8 | |||
| 19a91a1641 | |||
| c0fd80e694 | |||
| 9e262cc5f0 | |||
| db490e92df | |||
| 8ad40da145 | |||
| 1804a2c622 | |||
| 43b02d79a4 | |||
| 677317244e | |||
| a92817677d | |||
| e0d2a20588 | |||
| 394c7dcd67 | |||
| a73d9327d8 | |||
| 5201fab088 | |||
| b79708524e | |||
| 1d84f67418 | |||
| 1226d3b23a | |||
| 6c5dbfe4a0 | |||
| 91265df3d6 | |||
| 68cda6114d | |||
| c217009425 | |||
| f4f49d407e | |||
| a183be7e6e | |||
| f2e9ae07dc | |||
| d9d502a60d | |||
| 1819caaf5b | |||
| 22dc516dc7 | |||
| 11acdb0322 | |||
| 4c42a15fa1 | |||
| 2eb9fd5dd0 | |||
| 14c4dd1ea0 | |||
| 619e7712c2 | |||
| 01e5ce1410 | |||
| 416d8d71cd | |||
| 3bb94674cf | |||
| a3c9499e93 | |||
| 87c7318125 | |||
| a75c602175 | |||
| e4e35163ab | |||
| b7d07324ee | |||
| c65da42593 | |||
| ef8f4f7193 | |||
| 19eb4e9388 | |||
| a559fabdd5 | |||
| 711fd2bb75 | |||
| 3556a54260 | |||
| df2310bc70 | |||
| 9374f63953 | |||
| ec3d27b219 | |||
| 6332ae13fd | |||
| 3c89223dcb | |||
| 23f452021f | |||
| 62cca64dce | |||
| 89dfa42e18 | |||
| 2b69540ecc | |||
| 4fe53cdf6b | |||
| cb9b286c53 | |||
| a497104661 | |||
| 5bb25245a5 | |||
| 911d535f56 | |||
| 03bd3b2eda | |||
| e82c2ee57b | |||
| cd43439401 | |||
| bde19944db | |||
| 402086c34c | |||
| fb7383eea7 | |||
| e47fa0cf4b | |||
| b211900390 | |||
| 697a86d31c | |||
| 9a2cd569c3 | |||
| d592e0ca02 | |||
| 7a872a3619 | |||
| 4bb11ce7dc | |||
| e42a86d995 | |||
| b2e59e7e17 | |||
| e53f8959af | |||
| 5b615b7ded | |||
| d6c15f4ea0 | |||
| 3b2f7a41c3 | |||
| 218bfebb92 | |||
| ec43e823e1 | |||
| 682beafbc5 | |||
| 96c30eba13 | |||
| 2ec7d86a3b | |||
| 6222928746 | |||
| 1bdaa04aa2 | |||
| 1c2dc7659a | |||
| 7395e77d75 | |||
| 618dafde85 | |||
| 96fffaff64 | |||
| 575d817919 | |||
| add1c1ad14 | |||
| 593f65c9cc | |||
| 2a8f7cd8b6 | |||
| 86efbf7f2c | |||
| 3a0cca5aca | |||
| 83f8af8090 | |||
| a5b3702863 | |||
| 9a2617c1a2 | |||
| 509a7958cf | |||
| 81688815a0 | |||
| 5a6a95682d | |||
| 91b0145bc8 | |||
| 26e47a86cb | |||
| 773128c3bf | |||
| 928e3037f0 | |||
| ce7b154ae9 | |||
| b08b12eb8f | |||
| 9430a9d9c3 | |||
| 4fd6d4cc29 | |||
| 304e8aa878 | |||
| 0497394710 | |||
| 21a73cd1dc | |||
| 79cd1234e2 | |||
| 23aee56ce3 | |||
| 3f6ea601f8 | |||
| 6a25db4b8b | |||
| c802b26406 | |||
| 3a4270e6be | |||
| ae569c0f9a | |||
| 1adc47f59c | |||
| 9fe534139a | |||
| 16e0268da7 | |||
| 1f4ce8513b | |||
| 5a116ca9d0 | |||
| ef3ee5aceb | |||
| 914033db29 | |||
| d495605c12 | |||
| 76d8ad42a8 | |||
| 711abea567 | |||
| 6d630d13d6 | |||
| 3f30327fa5 | |||
| 4f9464d215 | |||
| e678d1dfdf | |||
| d9ab6e15c6 | |||
| e05e0b9f37 | |||
| 56cc253009 | |||
| 844bb86802 | |||
| 576e16d14d | |||
| a8f6a464aa | |||
| 9cb24c9e1b | |||
| 6590dcdb39 | |||
| ab9922ad2e | |||
| 3162cff96b | |||
| b65e956ad2 | |||
| 0533807669 | |||
| d3245f0c22 | |||
| 279dff3fb6 | |||
| e450145304 | |||
| a6e8d4b52e | |||
| 37e66cddc4 | |||
| f1860866de | |||
| 9cf6b2d363 | |||
| b181d779fe | |||
| 0fbb19dc24 | |||
| 8326e5447a | |||
| 1fd594baaf | |||
| ecac6c4bda | |||
| 6ef0fed41f | |||
| 9f7261b9c0 | |||
| f05aaa707b | |||
| 4df98171ab | |||
| 8d75ade1d5 | |||
| 75c63e1511 | |||
| 98673d4dca | |||
| 89b48f8f35 | |||
| 4bff1d8558 | |||
| d60e0b9494 | |||
| e30f50e6fe | |||
| 9c27a2d3c7 | |||
| e66987f092 | |||
| 93e37681b7 | |||
| 80ef9bce48 | |||
| 3898ce7be4 | |||
| 64ca858574 | |||
| 91be9df671 | |||
| 412edec028 | |||
| 937421485d | |||
| 9d0c0b7da8 | |||
| 43b778aa04 | |||
| 9cbdb70e13 | |||
| 8e4d252ae4 | |||
| bd06794647 | |||
| fdd3e01f56 | |||
| f575cfb93b | |||
| c82fb308b6 | |||
| 717b601c81 | |||
| cfa4fb4084 | |||
| 2aa2002f22 | |||
| 66ff671f09 | |||
| 19aece1fc4 | |||
| c9089b1d03 | |||
| 644d538bab | |||
| ff35da4743 | |||
| 2f66de2928 | |||
| 8cf8d2ca4d | |||
| 94e7d20792 | |||
| fb605af959 | |||
| 4c56cf121f | |||
| b1d58bc3b8 | |||
| 9564d073b9 | |||
| 65386f02a0 | |||
| f87a06a6bd | |||
| 5d284aae9f | |||
| af7b5c95e9 | |||
| 667b05f14e | |||
| 8de7ccd07d | |||
| d65f0b2091 | |||
| 856e9104b4 | |||
| 66f19d67f5 | |||
| 6fc8ae3106 | |||
| 0397642b21 | |||
| a5101494b6 | |||
| e3a7aff7a3 | |||
| 9cd6d09e60 | |||
| 237575447d | |||
| 810baf63ac | |||
| 44bb12a93d | |||
| 1eefed9ab3 | |||
| adeee64a2d | |||
| ed358757dc | |||
| 99b66aa85f | |||
| 77f7a23410 | |||
| d181f4afb8 | |||
| ff9e96e0e2 | |||
| 61ce1ce13c | |||
| d28db32012 | |||
| 77e9859da3 | |||
| 2886fa4997 | |||
| 36f8ec80fd | |||
| f256f587ee | |||
| e35fb1edf7 | |||
| 08420cd619 | |||
| 8649a13118 | |||
| 8979e0e377 | |||
| 00e2608ba1 | |||
| 9ab5d709c8 | |||
| 44410db492 | |||
| e76aa36a29 | |||
| c95b760294 | |||
| 35fe420701 | |||
| 9e74c80e2f | |||
| c87e8e0932 | |||
| 8c3900b998 | |||
| 972d9014ce | |||
| 75b6b8056e | |||
| 42ddac9996 | |||
| 384d8d5e50 | |||
| 1322056b22 | |||
| 32bdde049f | |||
| 9d18dacbe8 | |||
| 171c486939 | |||
| 21c1b0a81c | |||
| 597c6d48d3 | |||
| a37dad33c7 | |||
| cabd73287a | |||
| def967a1a8 | |||
| eebc8e2413 | |||
| 2358cedf3e | |||
| 215a8993a1 | |||
| a459d21a65 | |||
| 73520b7cc3 | |||
| 56970fb66d | |||
| bf8eb4468f | |||
| e1fc65bd1b | |||
| 104cac5dca |
@@ -242,20 +242,32 @@ jobs:
|
|||||||
id: tag
|
id: tag
|
||||||
run: |
|
run: |
|
||||||
# Three trigger shapes:
|
# Three trigger shapes:
|
||||||
# refs/tags/v… → tag-push: publish ONLY the immutable version
|
# refs/tags/v… → tag-push: opt-in milestone label (vYY.MM.DD,
|
||||||
# tag (e.g. :v26.05.26.5). Don't touch :latest;
|
# no `.N` per family release-posture rule).
|
||||||
# that already got published by the main-push
|
# Publish ONLY the immutable version tag;
|
||||||
# build for the merge commit.
|
# don't touch :latest (the main-push build
|
||||||
# refs/heads/main → push to main (incl. PR merge commits):
|
# for the merge commit already did that).
|
||||||
# publish :main + :latest (floating).
|
# refs/heads/main → push to main: publish :main + :latest
|
||||||
|
# (floating) AND :c-<short_sha> (immutable
|
||||||
|
# per-commit rollback substrate, per family
|
||||||
|
# release-posture rule "Tags are milestones,
|
||||||
|
# not gates — commit-SHA images are the
|
||||||
|
# rollback unit"). Rollback to any commit
|
||||||
|
# becomes `docker pull …:c-<sha>` without a
|
||||||
|
# release ceremony.
|
||||||
# anything else → safety net; shouldn't fire given the `on:`
|
# anything else → safety net; shouldn't fire given the `on:`
|
||||||
# config above (dev was dropped). Tag :dev to
|
# config above. Tag :dev to surface the
|
||||||
# surface the unexpected run in the registry.
|
# unexpected run in the registry.
|
||||||
|
# POSIX-safe substring (the runner shell is dash/BusyBox sh, not
|
||||||
|
# bash — `${var:0:7}` errors with "Bad substitution"; cut works
|
||||||
|
# everywhere). Operator-flagged 2026-06-01 after first :c-<sha>
|
||||||
|
# main-push build failed at this step.
|
||||||
|
SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7)
|
||||||
if [ "${GITHUB_REF#refs/tags/}" != "${GITHUB_REF}" ]; then
|
if [ "${GITHUB_REF#refs/tags/}" != "${GITHUB_REF}" ]; then
|
||||||
TAG_NAME="${GITHUB_REF#refs/tags/}"
|
TAG_NAME="${GITHUB_REF#refs/tags/}"
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:${TAG_NAME}" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:${TAG_NAME}" >> "$GITHUB_OUTPUT"
|
||||||
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:main,git.fabledsword.com/bvandeusen/fabledcurator:latest" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:main,git.fabledsword.com/bvandeusen/fabledcurator:latest,git.fabledsword.com/bvandeusen/fabledcurator:c-${SHORT_SHA}" >> "$GITHUB_OUTPUT"
|
||||||
else
|
else
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:dev" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:dev" >> "$GITHUB_OUTPUT"
|
||||||
fi
|
fi
|
||||||
@@ -286,13 +298,19 @@ jobs:
|
|||||||
id: tag
|
id: tag
|
||||||
run: |
|
run: |
|
||||||
# Mirrors build-web's three-shape logic (tag-push / main-push /
|
# Mirrors build-web's three-shape logic (tag-push / main-push /
|
||||||
# safety-net dev). The -ml image follows the same release cadence
|
# safety-net dev) including the per-commit :c-<short_sha> tag
|
||||||
# as the web image.
|
# on main-push per the family release-posture rule. The -ml
|
||||||
|
# image follows the same release cadence as the web image.
|
||||||
|
# POSIX-safe substring (the runner shell is dash/BusyBox sh, not
|
||||||
|
# bash — `${var:0:7}` errors with "Bad substitution"; cut works
|
||||||
|
# everywhere). Operator-flagged 2026-06-01 after first :c-<sha>
|
||||||
|
# main-push build failed at this step.
|
||||||
|
SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7)
|
||||||
if [ "${GITHUB_REF#refs/tags/}" != "${GITHUB_REF}" ]; then
|
if [ "${GITHUB_REF#refs/tags/}" != "${GITHUB_REF}" ]; then
|
||||||
TAG_NAME="${GITHUB_REF#refs/tags/}"
|
TAG_NAME="${GITHUB_REF#refs/tags/}"
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:${TAG_NAME}" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:${TAG_NAME}" >> "$GITHUB_OUTPUT"
|
||||||
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:main,git.fabledsword.com/bvandeusen/fabledcurator-ml:latest" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:main,git.fabledsword.com/bvandeusen/fabledcurator-ml:latest,git.fabledsword.com/bvandeusen/fabledcurator-ml:c-${SHORT_SHA}" >> "$GITHUB_OUTPUT"
|
||||||
else
|
else
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:dev" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:dev" >> "$GITHUB_OUTPUT"
|
||||||
fi
|
fi
|
||||||
|
|||||||
+48
-153
@@ -1,7 +1,8 @@
|
|||||||
name: CI
|
name: CI
|
||||||
|
|
||||||
# CI lanes per FabledRulebook/forgejo.md "CI philosophy":
|
# CI lanes per FabledRulebook/forgejo.md "CI philosophy":
|
||||||
# - backend-lint-and-test: ruff + `pytest -m "not integration"`, no service containers.
|
# - lint: ruff only, no dep install — fast-fail for the common lint bounce.
|
||||||
|
# - backend-lint-and-test: `pytest -m "not integration"`, no service containers.
|
||||||
# - frontend-build: vitest unit + vite build.
|
# - frontend-build: vitest unit + vite build.
|
||||||
# - integration: pgvector + redis service containers; alembic + `pytest -m integration`.
|
# - integration: pgvector + redis service containers; alembic + `pytest -m integration`.
|
||||||
|
|
||||||
@@ -14,6 +15,20 @@ on:
|
|||||||
# (single-operator Forgejo repo) so push coverage is complete.
|
# (single-operator Forgejo repo) so push coverage is complete.
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
|
# Fast-fail lint lane. ruff is pre-installed in the ci-python image, so
|
||||||
|
# this runs with NO dependency install and surfaces the most common bounce
|
||||||
|
# class (lint: I001 / UP037 / ASYNC109 / W293 …) in seconds — instead of
|
||||||
|
# after the backend job's ~30-60s wheel install. ruff is static analysis,
|
||||||
|
# so no DB/secret env is needed.
|
||||||
|
lint:
|
||||||
|
runs-on: python-ci
|
||||||
|
container:
|
||||||
|
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- name: Ruff lint
|
||||||
|
run: ruff check backend/ tests/ alembic/
|
||||||
|
|
||||||
backend-lint-and-test:
|
backend-lint-and-test:
|
||||||
runs-on: python-ci
|
runs-on: python-ci
|
||||||
container:
|
container:
|
||||||
@@ -51,9 +66,8 @@ jobs:
|
|||||||
pip install -r requirements.txt pytest pytest-asyncio
|
pip install -r requirements.txt pytest pytest-asyncio
|
||||||
fi
|
fi
|
||||||
|
|
||||||
- name: Ruff lint
|
# Ruff moved to the dedicated fast `lint` job above (fails in seconds,
|
||||||
run: ruff check backend/ tests/ alembic/
|
# no dep install). This job is now unit tests only.
|
||||||
|
|
||||||
- name: Pytest (unit only — integration runs in the integration job)
|
- name: Pytest (unit only — integration runs in the integration job)
|
||||||
run: pytest tests/ -v -m "not integration"
|
run: pytest tests/ -v -m "not integration"
|
||||||
|
|
||||||
@@ -78,28 +92,25 @@ jobs:
|
|||||||
- run: npm run test:unit
|
- run: npm run test:unit
|
||||||
- run: npm run build
|
- run: npm run build
|
||||||
|
|
||||||
# Integration suite split into THREE parallel shards (2026-05-25, runner
|
# Single integration job — collapsed from a 3-way shard split on 2026-06-04.
|
||||||
# capacity bumped 2→6). Each shard gets its own Postgres + Redis service
|
# The shards existed to parallelize ~8.5min of integration tests; once the
|
||||||
# set and runs alembic + a disjoint subset of integration tests. Shards
|
# throwaway Postgres runs with fsync OFF (the durability step below) the whole
|
||||||
# share no DB state, so the autouse TRUNCATE fixture in tests/conftest.py
|
# suite runs in ~45s, so the split only triplicated the ~2min fixed overhead
|
||||||
# stays single-threaded per shard but multiple shards run in parallel
|
# (container + `uv pip install` + `alembic upgrade head`) and burned 3 of 6
|
||||||
# wall-clock. Approximate split — rebalance once --durations=15 output
|
# runner slots for no wall-clock gain. One job now: spin up once, install
|
||||||
# reveals which shard is the long pole.
|
# once, migrate once, run every integration test.
|
||||||
#
|
#
|
||||||
# Each shard's docker-ps filter uses its own unique job name to scope
|
# The docker-ps filter scopes to THIS job's own Postgres/Redis service
|
||||||
# service-container resolution. act_runner appears to strip underscores
|
# containers by job name. act_runner strips underscores from job names when
|
||||||
# from job names when building container labels — `int_api` yielded
|
# labelling containers (`int_api` matched nothing on 2026-05-25), so the name
|
||||||
# zero matches on 2026-05-25 — so shards use no-separator names
|
# stays separator-free (`integration`). The step prints `docker ps -a` first
|
||||||
# (`intapi`, `intimp`, `intcore`) instead. Each step prints
|
# so a future naming-convention shift surfaces in the log without a
|
||||||
# `docker ps -a` first so a future naming-convention shift surfaces in
|
# guess-and-push cycle.
|
||||||
# the log without another guess-and-push cycle.
|
|
||||||
#
|
#
|
||||||
# Pre-baking requirements.txt into ci-python:3.14 is intentionally NOT
|
# Pre-baking requirements.txt into ci-python:3.14 is intentionally NOT done —
|
||||||
# done — per ci-requirements.md, FC is the only Python consumer of that
|
# per ci-requirements.md, FC is the only Python consumer of that image and the
|
||||||
# image and the CI-Runner project's "add deps to image when used by >1
|
# CI-Runner "add deps to image when used by >1 project" rule keeps it per-job.
|
||||||
# project" rule keeps the install per-job.
|
integration:
|
||||||
|
|
||||||
intapi:
|
|
||||||
runs-on: python-ci
|
runs-on: python-ci
|
||||||
container:
|
container:
|
||||||
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
||||||
@@ -130,14 +141,14 @@ jobs:
|
|||||||
--health-retries 10
|
--health-retries 10
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
- name: API integration shard (resolve service IPs, migrate, test)
|
- name: Integration suite (resolve service IPs, migrate, test)
|
||||||
run: |
|
run: |
|
||||||
set -eux
|
set -eux
|
||||||
echo "=== container landscape (diagnostic for filter scoping) ==="
|
echo "=== container landscape (diagnostic for filter scoping) ==="
|
||||||
docker ps -a --format '{{.ID}} {{.Image}} -> {{.Names}}'
|
docker ps -a --format '{{.ID}} {{.Image}} -> {{.Names}}'
|
||||||
echo "=== end landscape ==="
|
echo "=== end landscape ==="
|
||||||
PG=$(docker ps --filter "name=intapi" --filter "ancestor=pgvector/pgvector:pg16" -q | head -n1)
|
PG=$(docker ps --filter "name=integration" --filter "ancestor=pgvector/pgvector:pg16" -q | head -n1)
|
||||||
RD=$(docker ps --filter "name=intapi" --filter "ancestor=redis:7-alpine" -q | head -n1)
|
RD=$(docker ps --filter "name=integration" --filter "ancestor=redis:7-alpine" -q | head -n1)
|
||||||
test -n "$PG" && test -n "$RD"
|
test -n "$PG" && test -n "$RD"
|
||||||
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
|
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
|
||||||
RD_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$RD")
|
RD_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$RD")
|
||||||
@@ -154,130 +165,14 @@ jobs:
|
|||||||
else
|
else
|
||||||
pip install -r requirements.txt pytest pytest-asyncio
|
pip install -r requirements.txt pytest pytest-asyncio
|
||||||
fi
|
fi
|
||||||
|
# Relax durability on the throwaway CI Postgres so the per-test
|
||||||
|
# TRUNCATE's commit-fsync — the integration teardown's dominant cost
|
||||||
|
# (~1.5-2s/test, which collapsed the suite from ~13min to ~45s) — is
|
||||||
|
# skipped. fsync/full_page_writes are sighup GUCs and synchronous_commit
|
||||||
|
# is user-context, so ALTER SYSTEM + pg_reload_conf() applies them with
|
||||||
|
# NO restart. Ephemeral DB ⇒ fsync-off is safe. Non-fatal so a perms
|
||||||
|
# surprise can't red the job; fabledcurator is the postgres image's
|
||||||
|
# bootstrap superuser.
|
||||||
|
python -c "import os,psycopg; c=psycopg.connect(host=os.environ['DB_HOST'],port=5432,user=os.environ['DB_USER'],password=os.environ['DB_PASSWORD'],dbname=os.environ['DB_NAME'],autocommit=True); [c.execute(q) for q in ('ALTER SYSTEM SET fsync=off','ALTER SYSTEM SET synchronous_commit=off','ALTER SYSTEM SET full_page_writes=off','SELECT pg_reload_conf()')]; c.close()" || echo 'WARN: durability GUC relax failed (continuing)'
|
||||||
alembic upgrade head
|
alembic upgrade head
|
||||||
pytest tests/test_api_*.py -v -m integration --durations=15
|
pytest tests/ -v -m integration --durations=15
|
||||||
|
|
||||||
intimp:
|
|
||||||
runs-on: python-ci
|
|
||||||
container:
|
|
||||||
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
|
||||||
env:
|
|
||||||
DB_USER: fabledcurator
|
|
||||||
DB_PASSWORD: ci_integration
|
|
||||||
DB_PORT: "5432"
|
|
||||||
DB_NAME: fabledcurator_test
|
|
||||||
SECRET_KEY: ci_integration_placeholder
|
|
||||||
services:
|
|
||||||
postgres:
|
|
||||||
image: pgvector/pgvector:pg16
|
|
||||||
env:
|
|
||||||
POSTGRES_USER: fabledcurator
|
|
||||||
POSTGRES_PASSWORD: ci_integration
|
|
||||||
POSTGRES_DB: fabledcurator_test
|
|
||||||
options: >-
|
|
||||||
--health-cmd "pg_isready -U fabledcurator"
|
|
||||||
--health-interval 10s
|
|
||||||
--health-timeout 5s
|
|
||||||
--health-retries 10
|
|
||||||
redis:
|
|
||||||
image: redis:7-alpine
|
|
||||||
options: >-
|
|
||||||
--health-cmd "redis-cli ping"
|
|
||||||
--health-interval 10s
|
|
||||||
--health-timeout 5s
|
|
||||||
--health-retries 10
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- name: Importer integration shard (resolve service IPs, migrate, test)
|
|
||||||
run: |
|
|
||||||
set -eux
|
|
||||||
echo "=== container landscape (diagnostic for filter scoping) ==="
|
|
||||||
docker ps -a --format '{{.ID}} {{.Image}} -> {{.Names}}'
|
|
||||||
echo "=== end landscape ==="
|
|
||||||
PG=$(docker ps --filter "name=intimp" --filter "ancestor=pgvector/pgvector:pg16" -q | head -n1)
|
|
||||||
RD=$(docker ps --filter "name=intimp" --filter "ancestor=redis:7-alpine" -q | head -n1)
|
|
||||||
test -n "$PG" && test -n "$RD"
|
|
||||||
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
|
|
||||||
RD_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$RD")
|
|
||||||
test -n "$PG_IP" && test -n "$RD_IP"
|
|
||||||
export DB_HOST="$PG_IP"
|
|
||||||
export CELERY_BROKER_URL="redis://$RD_IP:6379/0"
|
|
||||||
export CELERY_RESULT_BACKEND="redis://$RD_IP:6379/0"
|
|
||||||
for i in $(seq 1 60); do
|
|
||||||
(echo > "/dev/tcp/$PG_IP/5432") >/dev/null 2>&1 && break
|
|
||||||
sleep 2
|
|
||||||
done
|
|
||||||
if command -v uv >/dev/null 2>&1; then
|
|
||||||
uv pip install --system -r requirements.txt pytest pytest-asyncio
|
|
||||||
else
|
|
||||||
pip install -r requirements.txt pytest pytest-asyncio
|
|
||||||
fi
|
|
||||||
alembic upgrade head
|
|
||||||
pytest tests/test_importer*.py tests/test_import_*.py tests/test_migration_*.py tests/test_phash_*.py tests/test_sidecar_*.py tests/test_scan_*.py tests/test_archive_extractor.py tests/test_backfill_phash.py -v -m integration --durations=15
|
|
||||||
|
|
||||||
intcore:
|
|
||||||
runs-on: python-ci
|
|
||||||
container:
|
|
||||||
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
|
||||||
env:
|
|
||||||
DB_USER: fabledcurator
|
|
||||||
DB_PASSWORD: ci_integration
|
|
||||||
DB_PORT: "5432"
|
|
||||||
DB_NAME: fabledcurator_test
|
|
||||||
SECRET_KEY: ci_integration_placeholder
|
|
||||||
services:
|
|
||||||
postgres:
|
|
||||||
image: pgvector/pgvector:pg16
|
|
||||||
env:
|
|
||||||
POSTGRES_USER: fabledcurator
|
|
||||||
POSTGRES_PASSWORD: ci_integration
|
|
||||||
POSTGRES_DB: fabledcurator_test
|
|
||||||
options: >-
|
|
||||||
--health-cmd "pg_isready -U fabledcurator"
|
|
||||||
--health-interval 10s
|
|
||||||
--health-timeout 5s
|
|
||||||
--health-retries 10
|
|
||||||
redis:
|
|
||||||
image: redis:7-alpine
|
|
||||||
options: >-
|
|
||||||
--health-cmd "redis-cli ping"
|
|
||||||
--health-interval 10s
|
|
||||||
--health-timeout 5s
|
|
||||||
--health-retries 10
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- name: Core integration shard (everything not api / importer / migration / phash / sidecar / scan / archive / backfill)
|
|
||||||
run: |
|
|
||||||
set -eux
|
|
||||||
echo "=== container landscape (diagnostic for filter scoping) ==="
|
|
||||||
docker ps -a --format '{{.ID}} {{.Image}} -> {{.Names}}'
|
|
||||||
echo "=== end landscape ==="
|
|
||||||
PG=$(docker ps --filter "name=intcore" --filter "ancestor=pgvector/pgvector:pg16" -q | head -n1)
|
|
||||||
RD=$(docker ps --filter "name=intcore" --filter "ancestor=redis:7-alpine" -q | head -n1)
|
|
||||||
test -n "$PG" && test -n "$RD"
|
|
||||||
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
|
|
||||||
RD_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$RD")
|
|
||||||
test -n "$PG_IP" && test -n "$RD_IP"
|
|
||||||
export DB_HOST="$PG_IP"
|
|
||||||
export CELERY_BROKER_URL="redis://$RD_IP:6379/0"
|
|
||||||
export CELERY_RESULT_BACKEND="redis://$RD_IP:6379/0"
|
|
||||||
for i in $(seq 1 60); do
|
|
||||||
(echo > "/dev/tcp/$PG_IP/5432") >/dev/null 2>&1 && break
|
|
||||||
sleep 2
|
|
||||||
done
|
|
||||||
if command -v uv >/dev/null 2>&1; then
|
|
||||||
uv pip install --system -r requirements.txt pytest pytest-asyncio
|
|
||||||
else
|
|
||||||
pip install -r requirements.txt pytest pytest-asyncio
|
|
||||||
fi
|
|
||||||
alembic upgrade head
|
|
||||||
pytest tests/ -v -m integration --durations=15 \
|
|
||||||
--ignore-glob='tests/test_api_*.py' \
|
|
||||||
--ignore-glob='tests/test_importer*.py' \
|
|
||||||
--ignore-glob='tests/test_import_*.py' \
|
|
||||||
--ignore-glob='tests/test_migration_*.py' \
|
|
||||||
--ignore-glob='tests/test_phash_*.py' \
|
|
||||||
--ignore-glob='tests/test_sidecar_*.py' \
|
|
||||||
--ignore-glob='tests/test_scan_*.py' \
|
|
||||||
--ignore-glob='tests/test_archive_extractor.py' \
|
|
||||||
--ignore-glob='tests/test_backfill_phash.py'
|
|
||||||
|
|||||||
@@ -61,8 +61,12 @@ Thumbs.db
|
|||||||
|
|
||||||
# Claude Code per-user local overrides (shared .claude/settings.json is OK to commit)
|
# Claude Code per-user local overrides (shared .claude/settings.json is OK to commit)
|
||||||
.claude/settings.local.json
|
.claude/settings.local.json
|
||||||
|
# Transient scheduler lock/state (committed by accident in 3f30327)
|
||||||
|
.claude/scheduled_tasks.lock
|
||||||
|
.claude/scheduled_tasks*.json
|
||||||
|
|
||||||
# Alembic / DB scratch
|
# Alembic / DB scratch
|
||||||
alembic/versions/__pycache__/
|
alembic/versions/__pycache__/
|
||||||
*.sqlite
|
*.sqlite
|
||||||
*.sqlite-journal
|
*.sqlite-journal
|
||||||
|
.superpowers/
|
||||||
|
|||||||
+4
-1
@@ -18,13 +18,16 @@ ENV PYTHONUNBUFFERED=1 \
|
|||||||
|
|
||||||
# System deps: ffmpeg (transcode + thumbnails, FC-2), unar (archives, FC-2),
|
# System deps: ffmpeg (transcode + thumbnails, FC-2), unar (archives, FC-2),
|
||||||
# libpq for psycopg, postgresql-client + zstd for FC-5 backup/restore
|
# libpq for psycopg, postgresql-client + zstd for FC-5 backup/restore
|
||||||
# (pg_dump + tar --zstd), image libs.
|
# (pg_dump + tar --zstd), image libs, megatools (mega.nz public-link downloads
|
||||||
|
# for off-platform file-host links, #830 — `megatools dl`; Debian-native, no
|
||||||
|
# external MEGA apt repo needed).
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
ffmpeg \
|
ffmpeg \
|
||||||
unar \
|
unar \
|
||||||
libpq5 \
|
libpq5 \
|
||||||
postgresql-client \
|
postgresql-client \
|
||||||
zstd \
|
zstd \
|
||||||
|
megatools \
|
||||||
libjpeg62-turbo \
|
libjpeg62-turbo \
|
||||||
libwebp7 \
|
libwebp7 \
|
||||||
libpng16-16 \
|
libpng16-16 \
|
||||||
|
|||||||
+25
-1
@@ -1,13 +1,28 @@
|
|||||||
"""Alembic environment — reads DATABASE_URL from app config."""
|
"""Alembic environment — reads DATABASE_URL from app config."""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import re
|
||||||
from logging.config import fileConfig
|
from logging.config import fileConfig
|
||||||
|
|
||||||
from sqlalchemy import engine_from_config, pool
|
from sqlalchemy import engine_from_config, pool, text
|
||||||
|
|
||||||
from alembic import context
|
from alembic import context
|
||||||
from backend.app.config import get_config
|
from backend.app.config import get_config
|
||||||
from backend.app.models import Base
|
from backend.app.models import Base
|
||||||
|
|
||||||
|
# Fail a blocked migration FAST instead of hanging forever. Migrations run
|
||||||
|
# against the live DB while workers hold locks; 0040's `ALTER series_page` queued
|
||||||
|
# behind a tag-merge that held a series_page lock for minutes (the merge runs an
|
||||||
|
# unindexed full scan over image_record while repointing series_page) and hung
|
||||||
|
# with no timeout — silent, indefinite (operator-flagged 2026-06-07). With a
|
||||||
|
# lock_timeout a blocked DDL errors ("canceling statement due to lock timeout")
|
||||||
|
# and the entrypoint's `alembic upgrade head` exits non-zero, so the deploy
|
||||||
|
# retries / surfaces loudly rather than wedging. Override via env when a known
|
||||||
|
# slow-lock window is expected.
|
||||||
|
_MIGRATION_LOCK_TIMEOUT = os.environ.get("MIGRATION_LOCK_TIMEOUT", "30s")
|
||||||
|
if not re.fullmatch(r"\d+\s*(ms|s|min)?", _MIGRATION_LOCK_TIMEOUT.strip()):
|
||||||
|
_MIGRATION_LOCK_TIMEOUT = "30s" # ignore a malformed override
|
||||||
|
|
||||||
config = context.config
|
config = context.config
|
||||||
|
|
||||||
if config.config_file_name is not None:
|
if config.config_file_name is not None:
|
||||||
@@ -38,6 +53,15 @@ def run_migrations_online() -> None:
|
|||||||
poolclass=pool.NullPool,
|
poolclass=pool.NullPool,
|
||||||
)
|
)
|
||||||
with connectable.connect() as connection:
|
with connectable.connect() as connection:
|
||||||
|
# Session-level lock_timeout for every DDL statement in this run. Set
|
||||||
|
# (and commit) before alembic opens its own transaction so the GUC
|
||||||
|
# persists on this connection regardless of how alembic structures its
|
||||||
|
# transactions. Value is from our own env, so f-string interpolation is
|
||||||
|
# safe (and it's been pattern-validated above); SET takes no bind params.
|
||||||
|
connection.execute(
|
||||||
|
text(f"SET lock_timeout = '{_MIGRATION_LOCK_TIMEOUT}'")
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
context.configure(
|
context.configure(
|
||||||
connection=connection,
|
connection=connection,
|
||||||
target_metadata=target_metadata,
|
target_metadata=target_metadata,
|
||||||
|
|||||||
@@ -0,0 +1,50 @@
|
|||||||
|
"""drop migration_run — one-and-done GS/IR migration tooling removed
|
||||||
|
|
||||||
|
Revision ID: 0027
|
||||||
|
Revises: 0026
|
||||||
|
Create Date: 2026-05-29
|
||||||
|
|
||||||
|
The GS/IR migration tooling (services/migrators, /api/migrate, the
|
||||||
|
run_migration task, LegacyMigrationCard, and the MigrationRun model) was
|
||||||
|
removed after the migration cutover completed. This drops its now-orphaned
|
||||||
|
run-log table. Downgrade recreates the table (mirrors the old model) so the
|
||||||
|
migration is reversible.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from sqlalchemy.dialects.postgresql import JSONB
|
||||||
|
|
||||||
|
revision: str = "0027"
|
||||||
|
down_revision: Union[str, None] = "0026"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.drop_table("migration_run")
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"migration_run",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column("kind", sa.String(length=32), nullable=False),
|
||||||
|
sa.Column("status", sa.String(length=32), nullable=False),
|
||||||
|
sa.Column("dry_run", sa.Boolean(), nullable=False, server_default=sa.false()),
|
||||||
|
sa.Column(
|
||||||
|
"started_at", sa.DateTime(timezone=True), nullable=False,
|
||||||
|
server_default=sa.func.now(),
|
||||||
|
),
|
||||||
|
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"counts", JSONB(), nullable=False, server_default=sa.text("'{}'::jsonb"),
|
||||||
|
),
|
||||||
|
sa.Column("error", sa.Text(), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"metadata", JSONB(), nullable=False, server_default=sa.text("'{}'::jsonb"),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.create_index("ix_migration_run_kind", "migration_run", ["kind"])
|
||||||
|
op.create_index("ix_migration_run_status", "migration_run", ["status"])
|
||||||
@@ -0,0 +1,190 @@
|
|||||||
|
"""collapse-sidecar-synthetic: repoint Posts/ImageProvenance/DownloadEvents
|
||||||
|
from `sidecar:<platform>:<slug>` synthetic Source anchors onto the real
|
||||||
|
Source for the same (artist, platform) when one exists, then delete the
|
||||||
|
synthetic.
|
||||||
|
|
||||||
|
Revision ID: 0028
|
||||||
|
Revises: 0027
|
||||||
|
Create Date: 2026-05-31
|
||||||
|
|
||||||
|
Background: alembic 0022 (2026-05-26) consolidated the old per-post-URL
|
||||||
|
Source rows into one canonical Source per (artist, platform). When NO
|
||||||
|
real campaign URL was salvageable among the candidates, it rewrote the
|
||||||
|
canonical row to url='sidecar:<platform>:<slug>' enabled=false as a
|
||||||
|
disabled anchor for any Posts already attached.
|
||||||
|
|
||||||
|
That was fine while it was the only Source for that artist+platform.
|
||||||
|
But: the unique constraint on Source is (artist_id, platform, url), not
|
||||||
|
(artist_id, platform). When the operator later added the real
|
||||||
|
subscription via the UI / extension / etc., a SECOND row landed —
|
||||||
|
the real one — with id > the synthetic. Both coexisted.
|
||||||
|
|
||||||
|
Two follow-on problems surfaced 2026-05-31:
|
||||||
|
|
||||||
|
1. The Subscriptions UI listed both rows. The synthetic was disabled
|
||||||
|
so the scheduler never polled it, but it looked like a phantom
|
||||||
|
subscription. (Fixed in same commit by SourceService.list filter.)
|
||||||
|
2. importer._source_for_sidecar picked Source by `ORDER BY id ASC
|
||||||
|
LIMIT 1`, so EVERY gallery-dl download since the real Source was
|
||||||
|
added attached its Post to the SYNTHETIC anchor, not the real
|
||||||
|
Source. (Fixed in same commit by preferring non-sidecar URLs.)
|
||||||
|
|
||||||
|
This migration is the data half of the cleanup: for every (artist,
|
||||||
|
platform) with both a synthetic AND a real Source, repoint the
|
||||||
|
synthetic's children (Posts, ImageProvenance, DownloadEvents) onto the
|
||||||
|
real Source and delete the synthetic. Reuses the same epid/provenance
|
||||||
|
collision dance from alembic 0022 because the same uniqueness
|
||||||
|
constraints fire row-by-row during bulk UPDATEs.
|
||||||
|
|
||||||
|
Lone synthetic anchors — those where no real Source for the same
|
||||||
|
(artist, platform) exists (e.g., filesystem-imported artist with no
|
||||||
|
subscription added) — are LEFT INTACT. They anchor real imported
|
||||||
|
content; deleting them would CASCADE-delete the Posts the operator
|
||||||
|
imported. The SourceService.list filter hides them from the UI; the
|
||||||
|
operator can delete them by hand if they want the underlying imports
|
||||||
|
gone.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
from sqlalchemy import text
|
||||||
|
|
||||||
|
revision: str = "0028"
|
||||||
|
down_revision: Union[str, None] = "0027"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
conn = op.get_bind()
|
||||||
|
|
||||||
|
# Find (artist_id, platform) groups where BOTH a sidecar synthetic
|
||||||
|
# and at least one real Source exist.
|
||||||
|
groups = conn.execute(text("""
|
||||||
|
SELECT artist_id, platform
|
||||||
|
FROM source
|
||||||
|
GROUP BY artist_id, platform
|
||||||
|
HAVING bool_or(url LIKE 'sidecar:%')
|
||||||
|
AND bool_or(url NOT LIKE 'sidecar:%')
|
||||||
|
""")).fetchall()
|
||||||
|
|
||||||
|
for artist_id, platform in groups:
|
||||||
|
rows = conn.execute(
|
||||||
|
text("""
|
||||||
|
SELECT id, url FROM source
|
||||||
|
WHERE artist_id = :a AND platform = :p
|
||||||
|
ORDER BY id ASC
|
||||||
|
"""),
|
||||||
|
{"a": artist_id, "p": platform},
|
||||||
|
).fetchall()
|
||||||
|
|
||||||
|
synthetic_ids = [sid for sid, url in rows if url.startswith("sidecar:")]
|
||||||
|
real_rows = [(sid, url) for sid, url in rows if not url.startswith("sidecar:")]
|
||||||
|
if not synthetic_ids or not real_rows:
|
||||||
|
continue # belt+suspenders; the GROUP BY already filtered
|
||||||
|
|
||||||
|
# Canonical real: lowest-id non-sidecar Source.
|
||||||
|
canonical_id = real_rows[0][0]
|
||||||
|
|
||||||
|
# STEP A: PRE-merge Post collisions on (canonical, external_post_id).
|
||||||
|
# Mirror alembic 0022's pre-merge logic — when synth has Post X
|
||||||
|
# epid=N and real has Post Y epid=N, the bulk UPDATE below would
|
||||||
|
# trip uq_post_source_external_id row-by-row. Group all Posts
|
||||||
|
# under (canonical + synthetics) by epid; for any group >1,
|
||||||
|
# pick a keep (prefer one already under canonical, else lowest
|
||||||
|
# id) and merge the rest into it.
|
||||||
|
all_posts = conn.execute(
|
||||||
|
text("""
|
||||||
|
SELECT external_post_id, id, source_id
|
||||||
|
FROM post
|
||||||
|
WHERE source_id = :canonical OR source_id = ANY(:synths)
|
||||||
|
ORDER BY external_post_id, id
|
||||||
|
"""),
|
||||||
|
{"canonical": canonical_id, "synths": synthetic_ids},
|
||||||
|
).fetchall()
|
||||||
|
by_epid: dict = {}
|
||||||
|
for epid, post_id, src_id in all_posts:
|
||||||
|
by_epid.setdefault(epid, []).append((post_id, src_id))
|
||||||
|
for _epid, posts in by_epid.items():
|
||||||
|
if len(posts) <= 1:
|
||||||
|
continue
|
||||||
|
canonical_side = [p for p in posts if p[1] == canonical_id]
|
||||||
|
keep_id = canonical_side[0][0] if canonical_side else posts[0][0]
|
||||||
|
drop_ids = [p[0] for p in posts if p[0] != keep_id]
|
||||||
|
for drop_id in drop_ids:
|
||||||
|
# Pre-delete image_provenance rows under drop_ whose
|
||||||
|
# image_record_id already has provenance under keep —
|
||||||
|
# avoids tripping uq_image_provenance_image_post (0021)
|
||||||
|
# row-by-row during the repoint UPDATE.
|
||||||
|
conn.execute(
|
||||||
|
text("""
|
||||||
|
DELETE FROM image_provenance
|
||||||
|
WHERE post_id = :drop_
|
||||||
|
AND image_record_id IN (
|
||||||
|
SELECT image_record_id FROM image_provenance
|
||||||
|
WHERE post_id = :keep
|
||||||
|
)
|
||||||
|
"""),
|
||||||
|
{"keep": keep_id, "drop_": drop_id},
|
||||||
|
)
|
||||||
|
conn.execute(
|
||||||
|
text("""
|
||||||
|
UPDATE image_provenance SET post_id = :keep
|
||||||
|
WHERE post_id = :drop_
|
||||||
|
"""),
|
||||||
|
{"keep": keep_id, "drop_": drop_id},
|
||||||
|
)
|
||||||
|
conn.execute(
|
||||||
|
text("""
|
||||||
|
UPDATE image_record SET primary_post_id = :keep
|
||||||
|
WHERE primary_post_id = :drop_
|
||||||
|
"""),
|
||||||
|
{"keep": keep_id, "drop_": drop_id},
|
||||||
|
)
|
||||||
|
conn.execute(
|
||||||
|
text("DELETE FROM post WHERE id = :drop_"),
|
||||||
|
{"drop_": drop_id},
|
||||||
|
)
|
||||||
|
|
||||||
|
# STEP B: Bulk reparent the remaining Posts off the synthetics.
|
||||||
|
conn.execute(
|
||||||
|
text("""
|
||||||
|
UPDATE post SET source_id = :canonical
|
||||||
|
WHERE source_id = ANY(:synths)
|
||||||
|
"""),
|
||||||
|
{"canonical": canonical_id, "synths": synthetic_ids},
|
||||||
|
)
|
||||||
|
|
||||||
|
# STEP C: Reparent ImageProvenance.source_id (denormalized FK;
|
||||||
|
# no UNIQUE on source_id, safe bulk).
|
||||||
|
conn.execute(
|
||||||
|
text("""
|
||||||
|
UPDATE image_provenance SET source_id = :canonical
|
||||||
|
WHERE source_id = ANY(:synths)
|
||||||
|
"""),
|
||||||
|
{"canonical": canonical_id, "synths": synthetic_ids},
|
||||||
|
)
|
||||||
|
|
||||||
|
# STEP D: Reparent any DownloadEvent.source_id. Synthetics are
|
||||||
|
# enabled=false so the scheduler never created events for them;
|
||||||
|
# this is belt+suspenders for any rows planted by manual force
|
||||||
|
# or older code paths.
|
||||||
|
conn.execute(
|
||||||
|
text("""
|
||||||
|
UPDATE download_event SET source_id = :canonical
|
||||||
|
WHERE source_id = ANY(:synths)
|
||||||
|
"""),
|
||||||
|
{"canonical": canonical_id, "synths": synthetic_ids},
|
||||||
|
)
|
||||||
|
|
||||||
|
# STEP E: Drop the now-empty synthetics.
|
||||||
|
conn.execute(
|
||||||
|
text("DELETE FROM source WHERE id = ANY(:synths)"),
|
||||||
|
{"synths": synthetic_ids},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Lossy migration — synthetic Sources deleted, Posts repointed and
|
||||||
|
# potentially merged. No safe downgrade.
|
||||||
|
pass
|
||||||
@@ -0,0 +1,71 @@
|
|||||||
|
"""drop artist + copyright ml thresholds; lower general default to 0.50
|
||||||
|
|
||||||
|
Revision ID: 0029
|
||||||
|
Revises: 0028
|
||||||
|
Create Date: 2026-06-01
|
||||||
|
|
||||||
|
Operator-flagged 2026-06-01: the view modal's Suggestions panel hides
|
||||||
|
most general-category predictions because the default threshold is
|
||||||
|
0.95. Lowering the default to 0.50 (matches character) so general
|
||||||
|
suggestions surface more aggressively; the value remains tunable in
|
||||||
|
Settings → ML.
|
||||||
|
|
||||||
|
Same change retires two ML suggestion categories whose Tag.kind
|
||||||
|
surfaces are unused:
|
||||||
|
|
||||||
|
- `artist`: retired in FC-2d-vii-c — artist identity is acquisition-
|
||||||
|
derived (image_record.artist_id), never ML-inferred. The threshold
|
||||||
|
column was a leftover from before that retirement.
|
||||||
|
- `copyright`: retired 2026-06-01 — the app uses `fandom` for the
|
||||||
|
franchise/copyright concept (per TagsView.vue's doc comment); no
|
||||||
|
Tag rows of kind=copyright exist, and the threshold column never
|
||||||
|
fed anything user-visible.
|
||||||
|
|
||||||
|
Both columns are dropped from ml_settings; the existing row's
|
||||||
|
suggestion_threshold_general value is bumped from 0.95 to 0.50 iff
|
||||||
|
it's still at the old default, so deployed installs pick up the new
|
||||||
|
UX without overriding any operator tuning.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
from sqlalchemy import text
|
||||||
|
|
||||||
|
revision: str = "0029"
|
||||||
|
down_revision: Union[str, None] = "0028"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
# Bump the general threshold for installs still at the old default.
|
||||||
|
op.execute(text(
|
||||||
|
"UPDATE ml_settings "
|
||||||
|
"SET suggestion_threshold_general = 0.50 "
|
||||||
|
"WHERE id = 1 AND suggestion_threshold_general = 0.95"
|
||||||
|
))
|
||||||
|
op.drop_column("ml_settings", "suggestion_threshold_artist")
|
||||||
|
op.drop_column("ml_settings", "suggestion_threshold_copyright")
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Restore the columns with their prior defaults. The bump from
|
||||||
|
# 0.95 → 0.50 isn't reversible without remembering whether the
|
||||||
|
# operator had explicitly set 0.95 (unlikely — that was just the
|
||||||
|
# default) so we leave the current general value as-is.
|
||||||
|
from sqlalchemy import Column, Float
|
||||||
|
|
||||||
|
op.add_column(
|
||||||
|
"ml_settings",
|
||||||
|
Column(
|
||||||
|
"suggestion_threshold_artist",
|
||||||
|
Float, nullable=False, server_default="0.30",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.add_column(
|
||||||
|
"ml_settings",
|
||||||
|
Column(
|
||||||
|
"suggestion_threshold_copyright",
|
||||||
|
Float, nullable=False, server_default="0.50",
|
||||||
|
),
|
||||||
|
)
|
||||||
@@ -0,0 +1,145 @@
|
|||||||
|
"""nullable post.source_id + denormalized post.artist_id; retire sidecar synthetics
|
||||||
|
|
||||||
|
Revision ID: 0030
|
||||||
|
Revises: 0029
|
||||||
|
Create Date: 2026-06-01
|
||||||
|
|
||||||
|
Operator-asked 2026-06-01 after the Dymkens orphan investigation: the
|
||||||
|
sidecar synthetic Source pattern (`sidecar:<platform>:<slug>` rows
|
||||||
|
with enabled=false) was technically correct but misled the operator
|
||||||
|
into thinking they had phantom subscriptions. The synthetics existed
|
||||||
|
solely to satisfy `Post.source_id NOT NULL` for filesystem-imported
|
||||||
|
content with no real subscription.
|
||||||
|
|
||||||
|
This migration makes the data model honest:
|
||||||
|
|
||||||
|
1. **Post gets a denormalized `artist_id` column** so artist filters
|
||||||
|
work without traversing `Post → Source.artist_id`. Backfilled from
|
||||||
|
the existing Source linkage, then NOT NULL'd.
|
||||||
|
2. **`Post.source_id` becomes nullable**, FK ondelete `CASCADE` → `SET
|
||||||
|
NULL`. Deleting a Source detaches its Posts instead of destroying
|
||||||
|
imported content (semantically: subscription ends, archive stays).
|
||||||
|
3. **`ImageProvenance.source_id` becomes nullable** with the same FK
|
||||||
|
semantic change.
|
||||||
|
4. **Sidecar synthetic Sources are deleted** — first NULL out the
|
||||||
|
FKs from Post + ImageProvenance pointing at them (so the implicit
|
||||||
|
CASCADE doesn't fire), then delete. DownloadEvent FK is unchanged
|
||||||
|
(still CASCADE'd, NOT NULL'd) — synthetics have `enabled=false`
|
||||||
|
so no events exist for them.
|
||||||
|
|
||||||
|
Uniqueness handling: the existing `uq_post_source_external_id`
|
||||||
|
(source_id, external_post_id) keeps working for source-bound Posts
|
||||||
|
(Postgres treats NULL != NULL so NULL-source rows aren't deduped by
|
||||||
|
it). A second partial unique index covers the NULL-source case on
|
||||||
|
(artist_id, external_post_id) so filesystem-imported posts still
|
||||||
|
dedupe within an artist.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from sqlalchemy import text
|
||||||
|
|
||||||
|
revision: str = "0030"
|
||||||
|
down_revision: Union[str, None] = "0029"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
conn = op.get_bind()
|
||||||
|
|
||||||
|
# Step 1: add Post.artist_id, initially nullable for backfill.
|
||||||
|
# FK naming follows the Base.metadata naming_convention
|
||||||
|
# (fk_<table>_<column>_<referred_table>) — alembic 0001 set this up.
|
||||||
|
op.add_column(
|
||||||
|
"post",
|
||||||
|
sa.Column("artist_id", sa.Integer, nullable=True),
|
||||||
|
)
|
||||||
|
op.create_foreign_key(
|
||||||
|
"fk_post_artist_id_artist", "post", "artist",
|
||||||
|
["artist_id"], ["id"], ondelete="CASCADE",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Step 2: backfill from Source.artist_id (every existing Post has a
|
||||||
|
# Source today, so every row gets populated).
|
||||||
|
conn.execute(text("""
|
||||||
|
UPDATE post p
|
||||||
|
SET artist_id = s.artist_id
|
||||||
|
FROM source s
|
||||||
|
WHERE p.source_id = s.id AND p.artist_id IS NULL
|
||||||
|
"""))
|
||||||
|
|
||||||
|
# Sanity: count any remaining NULLs. Should be zero pre-this-migration.
|
||||||
|
remaining = conn.execute(text(
|
||||||
|
"SELECT COUNT(*) FROM post WHERE artist_id IS NULL"
|
||||||
|
)).scalar_one()
|
||||||
|
if remaining:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"alembic 0030: {remaining} post rows have no resolvable "
|
||||||
|
f"artist_id after backfill. Investigate before continuing."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Step 3: enforce NOT NULL + add index for artist-filter queries.
|
||||||
|
op.alter_column("post", "artist_id", nullable=False)
|
||||||
|
op.create_index("ix_post_artist_id", "post", ["artist_id"])
|
||||||
|
|
||||||
|
# Step 4: relax post.source_id + flip FK to SET NULL. The original FK
|
||||||
|
# name from alembic 0001 is `fk_post_source_id_source` per the
|
||||||
|
# NAMING_CONVENTION in models/base.py.
|
||||||
|
op.alter_column("post", "source_id", nullable=True)
|
||||||
|
op.drop_constraint("fk_post_source_id_source", "post", type_="foreignkey")
|
||||||
|
op.create_foreign_key(
|
||||||
|
"fk_post_source_id_source", "post", "source",
|
||||||
|
["source_id"], ["id"], ondelete="SET NULL",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Step 5: relax image_provenance.source_id + flip FK to SET NULL.
|
||||||
|
op.alter_column("image_provenance", "source_id", nullable=True)
|
||||||
|
op.drop_constraint(
|
||||||
|
"fk_image_provenance_source_id_source", "image_provenance",
|
||||||
|
type_="foreignkey",
|
||||||
|
)
|
||||||
|
op.create_foreign_key(
|
||||||
|
"fk_image_provenance_source_id_source", "image_provenance", "source",
|
||||||
|
["source_id"], ["id"], ondelete="SET NULL",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Step 6: partial unique index on (artist_id, external_post_id) for
|
||||||
|
# NULL-source Posts. The existing uq_post_source_external_id keeps
|
||||||
|
# guarding source-bound rows; NULL-source rows now dedupe within
|
||||||
|
# an artist.
|
||||||
|
op.execute(
|
||||||
|
"CREATE UNIQUE INDEX uq_post_artist_external_id_null_source "
|
||||||
|
"ON post (artist_id, external_post_id) "
|
||||||
|
"WHERE source_id IS NULL"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Step 7: retire sidecar synthetic Sources. NULL out the references
|
||||||
|
# FIRST (the new FK is SET NULL so CASCADE wouldn't fire anyway, but
|
||||||
|
# being explicit makes the intent clear). Then delete the synthetic
|
||||||
|
# source rows. Any DownloadEvent rows under synthetics CASCADE-die
|
||||||
|
# with the source — synthetics have enabled=false so there shouldn't
|
||||||
|
# be any in practice.
|
||||||
|
conn.execute(text("""
|
||||||
|
UPDATE post
|
||||||
|
SET source_id = NULL
|
||||||
|
WHERE source_id IN (SELECT id FROM source WHERE url LIKE 'sidecar:%')
|
||||||
|
"""))
|
||||||
|
conn.execute(text("""
|
||||||
|
UPDATE image_provenance
|
||||||
|
SET source_id = NULL
|
||||||
|
WHERE source_id IN (SELECT id FROM source WHERE url LIKE 'sidecar:%')
|
||||||
|
"""))
|
||||||
|
deleted = conn.execute(text(
|
||||||
|
"DELETE FROM source WHERE url LIKE 'sidecar:%' RETURNING id"
|
||||||
|
)).rowcount
|
||||||
|
print(f"alembic 0030: deleted {deleted} sidecar synthetic source rows")
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Lossy migration — the deleted sidecar synthetics can't be
|
||||||
|
# restored from the orphan post.source_id / image_provenance.source_id
|
||||||
|
# values, and the partial unique index encodes a constraint that
|
||||||
|
# NULL-source Posts may now exist. No safe downgrade.
|
||||||
|
pass
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
"""source.backfill_runs_remaining: sticky deep-scan mode
|
||||||
|
|
||||||
|
Revision ID: 0031
|
||||||
|
Revises: 0030
|
||||||
|
Create Date: 2026-06-01
|
||||||
|
|
||||||
|
Tick vs backfill mode for subscription downloads. When
|
||||||
|
`backfill_runs_remaining > 0`, the next N download runs use
|
||||||
|
`skip: True` + 30-min timeout (walk full history). When 0, runs use
|
||||||
|
`skip: "exit:20"` + 14.5-min timeout (catch-up mode, exits early once
|
||||||
|
20 contiguous archived items are seen).
|
||||||
|
|
||||||
|
Operator-flagged 2026-06-01 (Knuxy run #38887): a creator with ~550
|
||||||
|
archived posts saturates the 870s catch-up timeout even when there is
|
||||||
|
no new content, because gallery-dl's default `skip: True` keeps walking.
|
||||||
|
Tick mode short-circuits that; backfill mode is the explicit opt-in for
|
||||||
|
deep history scans.
|
||||||
|
|
||||||
|
Default 0 (all existing subscriptions start in tick mode).
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0031"
|
||||||
|
down_revision: Union[str, None] = "0030"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"source",
|
||||||
|
sa.Column(
|
||||||
|
"backfill_runs_remaining",
|
||||||
|
sa.Integer,
|
||||||
|
nullable=False,
|
||||||
|
server_default="0",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("source", "backfill_runs_remaining")
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
"""source.error_type: surface ErrorType taxonomy in FailingSourcesCard
|
||||||
|
|
||||||
|
Revision ID: 0032
|
||||||
|
Revises: 0031
|
||||||
|
Create Date: 2026-06-02
|
||||||
|
|
||||||
|
Audit 2026-06-02: the backend computes 13 ErrorType categories (auth_error,
|
||||||
|
rate_limited, not_found, access_denied, validation_failed, etc.) and
|
||||||
|
stamps each one on DownloadEvent.metadata, but the Source row only carried
|
||||||
|
the free-text last_error. Operators couldn't bulk-triage failing sources
|
||||||
|
("all auth_error → rotate cookies, all rate_limited → just wait") without
|
||||||
|
opening Logs per row.
|
||||||
|
|
||||||
|
This column receives the last error_type from _update_source_health
|
||||||
|
and gets cleared on a successful run. Nullable + indexed so the failing-
|
||||||
|
sources rollup can filter/group cheaply.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0032"
|
||||||
|
down_revision: Union[str, None] = "0031"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"source",
|
||||||
|
sa.Column("error_type", sa.String(length=32), nullable=True),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_source_error_type", "source", ["error_type"],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_source_error_type", table_name="source")
|
||||||
|
op.drop_column("source", "error_type")
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
"""suggestion_threshold default 0.50 → 0.70
|
||||||
|
|
||||||
|
Revision ID: 0033
|
||||||
|
Revises: 0032
|
||||||
|
Create Date: 2026-06-02
|
||||||
|
|
||||||
|
Operator-flagged 2026-06-02 — the 0.50 default (set on 2026-06-01) is
|
||||||
|
too noisy in practice; raise to 0.70 for both suggestion categories.
|
||||||
|
|
||||||
|
Only conditionally updates singletons whose current value is still the
|
||||||
|
2026-06-01 default (0.50). Operators who deliberately tuned their row
|
||||||
|
to some other value (0.55, 0.65, 0.80, etc. via the Settings UI) keep
|
||||||
|
their pick — the migration only catches the unchanged-default case.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0033"
|
||||||
|
down_revision: Union[str, None] = "0032"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.execute(
|
||||||
|
"UPDATE ml_settings "
|
||||||
|
"SET suggestion_threshold_character = 0.70 "
|
||||||
|
"WHERE id = 1 AND suggestion_threshold_character = 0.50"
|
||||||
|
)
|
||||||
|
op.execute(
|
||||||
|
"UPDATE ml_settings "
|
||||||
|
"SET suggestion_threshold_general = 0.70 "
|
||||||
|
"WHERE id = 1 AND suggestion_threshold_general = 0.50"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.execute(
|
||||||
|
"UPDATE ml_settings "
|
||||||
|
"SET suggestion_threshold_character = 0.50 "
|
||||||
|
"WHERE id = 1 AND suggestion_threshold_character = 0.70"
|
||||||
|
)
|
||||||
|
op.execute(
|
||||||
|
"UPDATE ml_settings "
|
||||||
|
"SET suggestion_threshold_general = 0.50 "
|
||||||
|
"WHERE id = 1 AND suggestion_threshold_general = 0.70"
|
||||||
|
)
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
"""artist_visit: per-artist last-viewed timestamp for the "+N new" badge
|
||||||
|
|
||||||
|
Revision ID: 0034
|
||||||
|
Revises: 0033
|
||||||
|
Create Date: 2026-06-03
|
||||||
|
|
||||||
|
Powers the artists-directory "+N new since last visit" badge + ArtistView
|
||||||
|
banner. Single row per artist (no user_id yet — rule #47 multi-user ACL
|
||||||
|
is aspirational; widens to (user_id, artist_id) PK when User lands).
|
||||||
|
|
||||||
|
Seed every existing artist with `last_viewed_at = NOW()` so the badge
|
||||||
|
starts at 0 across the board — no noisy "you have 5000 unseen images"
|
||||||
|
on first deploy. New artists auto-get a row via
|
||||||
|
`ArtistService.find_or_create`.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0034"
|
||||||
|
down_revision: Union[str, None] = "0033"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"artist_visit",
|
||||||
|
sa.Column(
|
||||||
|
"artist_id",
|
||||||
|
sa.Integer,
|
||||||
|
sa.ForeignKey("artist.id", ondelete="CASCADE"),
|
||||||
|
primary_key=True,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"last_viewed_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("NOW()"),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
# Seed: every existing artist starts "fully caught up". Without this,
|
||||||
|
# every operator with N artists would see N badges (worth of every
|
||||||
|
# image ever imported) on first deploy.
|
||||||
|
op.execute(
|
||||||
|
"INSERT INTO artist_visit (artist_id, last_viewed_at) "
|
||||||
|
"SELECT id, NOW() FROM artist"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_table("artist_visit")
|
||||||
@@ -0,0 +1,70 @@
|
|||||||
|
"""image_record.effective_date: materialized gallery sort key + index
|
||||||
|
|
||||||
|
Revision ID: 0035
|
||||||
|
Revises: 0034
|
||||||
|
Create Date: 2026-06-04
|
||||||
|
|
||||||
|
The gallery ordered/cursored on COALESCE(post.post_date,
|
||||||
|
image_record.created_at) across the Post outer join. That expression spans
|
||||||
|
two tables, so no index can serve it — every /scroll sorted a large slice
|
||||||
|
of the library, and the frontend fired ten of them serially per initial
|
||||||
|
load. Materialize the value into image_record.effective_date and index
|
||||||
|
(effective_date DESC, id DESC) so the cursor scroll is an index range scan.
|
||||||
|
|
||||||
|
Backfill = COALESCE(primary post's post_date, created_at) so existing rows
|
||||||
|
keep their exact ordering. New rows get the created_at-equivalent server
|
||||||
|
default; services/importer.py overrides it with the post's date when a
|
||||||
|
primary post with a date is linked.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0035"
|
||||||
|
down_revision: Union[str, None] = "0034"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
# Add nullable first so the backfill can populate before NOT NULL.
|
||||||
|
op.add_column(
|
||||||
|
"image_record",
|
||||||
|
sa.Column("effective_date", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
)
|
||||||
|
# Pure set-based UPDATEs (no per-row params) — immune to the 65535
|
||||||
|
# bind-parameter ceiling regardless of library size.
|
||||||
|
op.execute(
|
||||||
|
"""
|
||||||
|
UPDATE image_record AS ir
|
||||||
|
SET effective_date = COALESCE(p.post_date, ir.created_at)
|
||||||
|
FROM post AS p
|
||||||
|
WHERE ir.primary_post_id = p.id
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
op.execute(
|
||||||
|
"""
|
||||||
|
UPDATE image_record
|
||||||
|
SET effective_date = created_at
|
||||||
|
WHERE effective_date IS NULL
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
op.alter_column(
|
||||||
|
"image_record",
|
||||||
|
"effective_date",
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("now()"),
|
||||||
|
)
|
||||||
|
# DESC/DESC matches the gallery's ORDER BY effective_date DESC, id DESC
|
||||||
|
# so the scroll is a forward index scan; raw SQL because alembic's
|
||||||
|
# column list doesn't express per-column DESC cleanly.
|
||||||
|
op.execute(
|
||||||
|
"CREATE INDEX ix_image_record_effective_date "
|
||||||
|
"ON image_record (effective_date DESC, id DESC)"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_image_record_effective_date", table_name="image_record")
|
||||||
|
op.drop_column("image_record", "effective_date")
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
"""image_record.siglip_embedding: HNSW cosine index for "more like this"
|
||||||
|
|
||||||
|
Revision ID: 0036
|
||||||
|
Revises: 0035
|
||||||
|
Create Date: 2026-06-04
|
||||||
|
|
||||||
|
Gallery Phase 3 (visual similarity search) ranks images by
|
||||||
|
`siglip_embedding.cosine_distance(source_embedding)`. Without an index that's
|
||||||
|
a sequential scan computing a 1152-dim distance for every row — fine at small
|
||||||
|
scale, but it grows linearly with the library. Add an HNSW index with
|
||||||
|
`vector_cosine_ops` so the top-N nearest search is sub-50ms ANN.
|
||||||
|
|
||||||
|
1152 dims is under pgvector's 2000-dim HNSW limit, so HNSW (no training,
|
||||||
|
better recall than IVFFlat) is the right choice. ONE-TIME COST: building the
|
||||||
|
index over the existing embeddings (~57k vectors on the operator's library)
|
||||||
|
locks image_record for ~30-60s during this migration on deploy — acceptable
|
||||||
|
for a single-operator homelab. NULL embeddings (videos / not-yet-embedded
|
||||||
|
rows) are simply not indexed.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0036"
|
||||||
|
down_revision: Union[str, None] = "0035"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
# Raw SQL: alembic's create_index doesn't express the `USING hnsw (...
|
||||||
|
# vector_cosine_ops)` access-method + opclass cleanly. Must match the
|
||||||
|
# query's cosine_distance operator class to be usable by the planner.
|
||||||
|
op.execute(
|
||||||
|
"CREATE INDEX ix_image_record_siglip_hnsw "
|
||||||
|
"ON image_record USING hnsw (siglip_embedding vector_cosine_ops)"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_image_record_siglip_hnsw", table_name="image_record")
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
"""patreon_seen_media: per-source ledger of already-ingested Patreon media
|
||||||
|
|
||||||
|
Revision ID: 0037
|
||||||
|
Revises: 0036
|
||||||
|
Create Date: 2026-06-05
|
||||||
|
|
||||||
|
Native Patreon ingester (build step 2a). Replaces gallery-dl's
|
||||||
|
archive.sqlite3 with our own queryable table. The downloader upserts one
|
||||||
|
row per (source, media) so routine walks skip media we've already
|
||||||
|
processed; a future "recovery" mode bypasses the ledger to re-walk.
|
||||||
|
|
||||||
|
`filehash` is a 32-hex Patreon CDN MD5, OR a video sentinel of the form
|
||||||
|
``video:<post_id>:<media_id>`` — hence String(128). The unique
|
||||||
|
constraint on (source_id, filehash) is the dedup upsert key.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0037"
|
||||||
|
down_revision: Union[str, None] = "0036"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"patreon_seen_media",
|
||||||
|
sa.Column("id", sa.Integer, primary_key=True),
|
||||||
|
sa.Column(
|
||||||
|
"source_id",
|
||||||
|
sa.Integer,
|
||||||
|
sa.ForeignKey("source.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
index=True,
|
||||||
|
),
|
||||||
|
sa.Column("filehash", sa.String(128), nullable=False),
|
||||||
|
sa.Column("post_id", sa.String(64), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"seen_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("NOW()"),
|
||||||
|
),
|
||||||
|
sa.UniqueConstraint(
|
||||||
|
"source_id", "filehash", name="uq_patreon_seen_media_source_id"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_table("patreon_seen_media")
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
"""patreon_failed_media: per-source dead-letter ledger for failing Patreon media
|
||||||
|
|
||||||
|
Revision ID: 0038
|
||||||
|
Revises: 0037
|
||||||
|
Create Date: 2026-06-06
|
||||||
|
|
||||||
|
Plan #705 (#7). Media that keeps failing to download/validate (404'd CDN,
|
||||||
|
deleted post, geo-blocked Mux, persistently-corrupt bytes) gets recorded here
|
||||||
|
with an attempt counter; once it crosses the dead-letter threshold the ingester
|
||||||
|
skips it on routine walks (recovery still re-attempts). A clean download clears
|
||||||
|
the row. UNIQUE (source_id, filehash) is the upsert key (same media key the
|
||||||
|
seen-ledger uses).
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0038"
|
||||||
|
down_revision: Union[str, None] = "0037"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"patreon_failed_media",
|
||||||
|
sa.Column("id", sa.Integer, primary_key=True),
|
||||||
|
sa.Column(
|
||||||
|
"source_id",
|
||||||
|
sa.Integer,
|
||||||
|
sa.ForeignKey("source.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
index=True,
|
||||||
|
),
|
||||||
|
sa.Column("filehash", sa.String(128), nullable=False),
|
||||||
|
sa.Column("attempts", sa.Integer, nullable=False, server_default="1"),
|
||||||
|
sa.Column("last_error", sa.Text, nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"first_failed_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("NOW()"),
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"last_failed_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("NOW()"),
|
||||||
|
),
|
||||||
|
sa.UniqueConstraint(
|
||||||
|
"source_id", "filehash", name="uq_patreon_failed_media_source_id"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_table("patreon_failed_media")
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
"""library_audit_run: resume cursor + progress timestamp for chunked scans
|
||||||
|
|
||||||
|
Revision ID: 0039
|
||||||
|
Revises: 0038
|
||||||
|
Create Date: 2026-06-07
|
||||||
|
|
||||||
|
scan_library_for_rule used to run one 2h pass that timed out on large libraries
|
||||||
|
and monopolized the concurrency-1 maintenance queue (operator-flagged). It now
|
||||||
|
runs short time-boxed chunks that re-enqueue: `resume_after_id` persists the
|
||||||
|
keyset cursor so the next chunk continues where it left off, and
|
||||||
|
`last_progress_at` lets the recovery sweep tell a progressing multi-chunk audit
|
||||||
|
from a genuinely stuck one.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0039"
|
||||||
|
down_revision: Union[str, None] = "0038"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"library_audit_run",
|
||||||
|
sa.Column(
|
||||||
|
"resume_after_id", sa.Integer, nullable=False, server_default="0"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.add_column(
|
||||||
|
"library_audit_run",
|
||||||
|
sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("library_audit_run", "last_progress_at")
|
||||||
|
op.drop_column("library_audit_run", "resume_after_id")
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
"""series chapters: chapter layer over series_page (FC-6.1)
|
||||||
|
|
||||||
|
Revision ID: 0040
|
||||||
|
Revises: 0039
|
||||||
|
Create Date: 2026-06-07
|
||||||
|
|
||||||
|
A series (Tag kind='series') gains an ordered chapter layer. Reading order
|
||||||
|
becomes (series_chapter.chapter_number, series_page.page_number). Every existing
|
||||||
|
series is backfilled into a single auto-chapter (chapter_number=1) holding its
|
||||||
|
current flat pages, so no data is lost and the old flat ordering is preserved.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0040"
|
||||||
|
down_revision: Union[str, None] = "0039"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"series_chapter",
|
||||||
|
sa.Column("id", sa.Integer, primary_key=True),
|
||||||
|
sa.Column(
|
||||||
|
"series_tag_id",
|
||||||
|
sa.Integer,
|
||||||
|
sa.ForeignKey("tag.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("chapter_number", sa.Integer, nullable=False),
|
||||||
|
sa.Column("title", sa.Text, nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"is_placeholder", sa.Boolean, nullable=False, server_default="false"
|
||||||
|
),
|
||||||
|
sa.Column("stated_page_start", sa.Integer, nullable=True),
|
||||||
|
sa.Column("stated_page_end", sa.Integer, nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"created_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("now()"),
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"updated_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("now()"),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_series_chapter_series_tag_id", "series_chapter", ["series_tag_id"]
|
||||||
|
)
|
||||||
|
|
||||||
|
# New columns on series_page; chapter_id starts nullable so we can backfill.
|
||||||
|
op.add_column(
|
||||||
|
"series_page", sa.Column("chapter_id", sa.Integer, nullable=True)
|
||||||
|
)
|
||||||
|
op.add_column(
|
||||||
|
"series_page", sa.Column("stated_page", sa.Integer, nullable=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
conn = op.get_bind()
|
||||||
|
# One auto-chapter per existing series (any series_tag_id present in pages).
|
||||||
|
conn.execute(
|
||||||
|
sa.text(
|
||||||
|
"INSERT INTO series_chapter "
|
||||||
|
"(series_tag_id, chapter_number, is_placeholder, created_at, updated_at) "
|
||||||
|
"SELECT DISTINCT series_tag_id, 1, false, now(), now() "
|
||||||
|
"FROM series_page"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
# Point every existing page at its series' auto-chapter.
|
||||||
|
conn.execute(
|
||||||
|
sa.text(
|
||||||
|
"UPDATE series_page sp "
|
||||||
|
"SET chapter_id = sc.id "
|
||||||
|
"FROM series_chapter sc "
|
||||||
|
"WHERE sc.series_tag_id = sp.series_tag_id"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Now lock chapter_id down: NOT NULL + FK (cascade) + index.
|
||||||
|
op.alter_column("series_page", "chapter_id", nullable=False)
|
||||||
|
op.create_foreign_key(
|
||||||
|
"fk_series_page_chapter_id",
|
||||||
|
"series_page",
|
||||||
|
"series_chapter",
|
||||||
|
["chapter_id"],
|
||||||
|
["id"],
|
||||||
|
ondelete="CASCADE",
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_series_page_chapter_id", "series_page", ["chapter_id"]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_series_page_chapter_id", table_name="series_page")
|
||||||
|
op.drop_constraint(
|
||||||
|
"fk_series_page_chapter_id", "series_page", type_="foreignkey"
|
||||||
|
)
|
||||||
|
op.drop_column("series_page", "stated_page")
|
||||||
|
op.drop_column("series_page", "chapter_id")
|
||||||
|
op.drop_index("ix_series_chapter_series_tag_id", table_name="series_chapter")
|
||||||
|
op.drop_table("series_chapter")
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
"""series suggestions: assisted-continuation matcher (FC-6.3)
|
||||||
|
|
||||||
|
Revision ID: 0041
|
||||||
|
Revises: 0040
|
||||||
|
Create Date: 2026-06-07
|
||||||
|
|
||||||
|
A confirm-only queue of "this post may continue this series" hints, plus two
|
||||||
|
import_settings knobs (enable + score threshold) for the matcher.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0041"
|
||||||
|
down_revision: Union[str, None] = "0040"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"series_suggestion",
|
||||||
|
sa.Column("id", sa.Integer, primary_key=True),
|
||||||
|
sa.Column(
|
||||||
|
"post_id",
|
||||||
|
sa.Integer,
|
||||||
|
sa.ForeignKey("post.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"series_tag_id",
|
||||||
|
sa.Integer,
|
||||||
|
sa.ForeignKey("tag.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("score", sa.Float, nullable=False),
|
||||||
|
sa.Column("signals", sa.JSON, nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"status", sa.String(16), nullable=False, server_default="pending"
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"created_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("now()"),
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"updated_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.text("now()"),
|
||||||
|
),
|
||||||
|
sa.UniqueConstraint(
|
||||||
|
"post_id", "series_tag_id", name="uq_series_suggestion_post_series"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_series_suggestion_post_id", "series_suggestion", ["post_id"]
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_series_suggestion_series_tag_id",
|
||||||
|
"series_suggestion",
|
||||||
|
["series_tag_id"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_series_suggestion_status", "series_suggestion", ["status"]
|
||||||
|
)
|
||||||
|
|
||||||
|
op.add_column(
|
||||||
|
"import_settings",
|
||||||
|
sa.Column(
|
||||||
|
"series_suggest_enabled",
|
||||||
|
sa.Boolean,
|
||||||
|
nullable=False,
|
||||||
|
server_default=sa.true(),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.add_column(
|
||||||
|
"import_settings",
|
||||||
|
sa.Column(
|
||||||
|
"series_suggest_threshold",
|
||||||
|
sa.Float,
|
||||||
|
nullable=False,
|
||||||
|
server_default="0.5",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("import_settings", "series_suggest_threshold")
|
||||||
|
op.drop_column("import_settings", "series_suggest_enabled")
|
||||||
|
op.drop_index("ix_series_suggestion_status", table_name="series_suggestion")
|
||||||
|
op.drop_index(
|
||||||
|
"ix_series_suggestion_series_tag_id", table_name="series_suggestion"
|
||||||
|
)
|
||||||
|
op.drop_index("ix_series_suggestion_post_id", table_name="series_suggestion")
|
||||||
|
op.drop_table("series_suggestion")
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
"""series chapter stated_part: operator-facing Part N label (FC-6.4)
|
||||||
|
|
||||||
|
Revision ID: 0042
|
||||||
|
Revises: 0041
|
||||||
|
Create Date: 2026-06-07
|
||||||
|
|
||||||
|
A chapter's positional chapter_number is auto-managed (rewritten 1..N on
|
||||||
|
reorder/delete), so it can't double as the installment number the operator wants
|
||||||
|
to type (e.g. a series authored from a post that is Part 2). Add a nullable
|
||||||
|
stated_part alongside it — the same split as series_page.page_number (order) vs
|
||||||
|
series_page.stated_page (printed number). Nullable; the UI falls back to
|
||||||
|
chapter_number when unset.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0042"
|
||||||
|
down_revision: Union[str, None] = "0041"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"series_chapter", sa.Column("stated_part", sa.Integer, nullable=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("series_chapter", "stated_part")
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
"""post_attachment: per-post sha uniqueness (empty-post flood fix)
|
||||||
|
|
||||||
|
Revision ID: 0043
|
||||||
|
Revises: 0042
|
||||||
|
Create Date: 2026-06-08
|
||||||
|
|
||||||
|
PostAttachment.sha256 was GLOBALLY unique, so a non-art file the creator attaches
|
||||||
|
to many posts (a standard pdf/zip/link-card) only ever got ONE row — on the first
|
||||||
|
post — leaving every later post a bare shell (no image, no attachment). The native
|
||||||
|
Patreon backfill of Anduo surfaced 1589 such shells (operator-flagged 2026-06-08).
|
||||||
|
|
||||||
|
Switch to PER-POST uniqueness: the on-disk blob stays sha-deduped, but each post
|
||||||
|
gets its own row. Replace the unique sha256 index with a plain lookup index plus
|
||||||
|
two partial uniques — (post_id, sha256) for real posts and (sha256) for the
|
||||||
|
NULL-post filesystem case (still one row per file there).
|
||||||
|
|
||||||
|
Existing data has ≤1 row per sha (the old global unique), so the new partial
|
||||||
|
uniques can't be violated on upgrade — no data backfill needed here. The bare-post
|
||||||
|
shells themselves are removed by the separate prune-empty-posts cleanup tool.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0043"
|
||||||
|
down_revision: Union[str, None] = "0042"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
# Drop the global unique index; recreate it as a plain (non-unique) lookup
|
||||||
|
# index so sha-based reads keep their index (matches the model's index=True).
|
||||||
|
op.drop_index("ix_post_attachment_sha256", table_name="post_attachment")
|
||||||
|
op.create_index(
|
||||||
|
"ix_post_attachment_sha256", "post_attachment", ["sha256"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"uq_post_attachment_post_sha", "post_attachment",
|
||||||
|
["post_id", "sha256"], unique=True,
|
||||||
|
postgresql_where=sa.text("post_id IS NOT NULL"),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"uq_post_attachment_null_post_sha", "post_attachment",
|
||||||
|
["sha256"], unique=True,
|
||||||
|
postgresql_where=sa.text("post_id IS NULL"),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index(
|
||||||
|
"uq_post_attachment_null_post_sha", table_name="post_attachment"
|
||||||
|
)
|
||||||
|
op.drop_index(
|
||||||
|
"uq_post_attachment_post_sha", table_name="post_attachment"
|
||||||
|
)
|
||||||
|
op.drop_index("ix_post_attachment_sha256", table_name="post_attachment")
|
||||||
|
op.create_index(
|
||||||
|
"ix_post_attachment_sha256", "post_attachment", ["sha256"],
|
||||||
|
unique=True,
|
||||||
|
)
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
"""ml_settings.tagger_store_floor
|
||||||
|
|
||||||
|
The ingest confidence floor below which tagger predictions are not stored,
|
||||||
|
promoted from the TAGGER_STORE_FLOOR env var to a DB-backed, UI-tunable
|
||||||
|
setting. Default 0.70 (was an env default of 0.05): the suggestion path
|
||||||
|
already filters at 0.70 and the centroid/learned path covers low-confidence
|
||||||
|
preferred tags, so the sub-0.70 tail was redundant weight — it had grown
|
||||||
|
image_record's TOAST to ~100 GB. See plan-task #764.
|
||||||
|
|
||||||
|
Revision ID: 0044
|
||||||
|
Revises: 0043
|
||||||
|
Create Date: 2026-06-10
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0044"
|
||||||
|
down_revision: Union[str, None] = "0043"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"ml_settings",
|
||||||
|
sa.Column(
|
||||||
|
"tagger_store_floor", sa.Float(),
|
||||||
|
nullable=False, server_default="0.7",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("ml_settings", "tagger_store_floor")
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
"""image_prediction table (DDL only — backfill runs as a background task)
|
||||||
|
|
||||||
|
Normalizes the per-image tagger predictions out of the JSON blob into a
|
||||||
|
queryable table (#768). This migration creates ONLY the table + indexes — it
|
||||||
|
is pure DDL and commits instantly, so web boots immediately.
|
||||||
|
|
||||||
|
The data backfill from the existing image_record.tagger_predictions JSON is
|
||||||
|
deliberately NOT done here. Doing it inline made the whole migration one
|
||||||
|
transaction over the ~100 GB TOAST: nothing committed until the very end, it
|
||||||
|
was invisible/unmonitorable mid-run, and an early MATERIALIZED-CTE form spilled
|
||||||
|
the full 100 GB to temp. Instead the backfill is the
|
||||||
|
backend.app.tasks.admin.backfill_image_predictions_task — batched by id window,
|
||||||
|
committed per chunk (visible progress + resumable), idempotent
|
||||||
|
(ON CONFLICT DO NOTHING). Trigger it from Settings → Maintenance once web is up.
|
||||||
|
|
||||||
|
The old image_record.tagger_predictions column is left in place (vestigial) and
|
||||||
|
dropped in a follow-up once the backfill + code cutover are verified — dropping
|
||||||
|
it needs an ACCESS EXCLUSIVE lock on the hot image_record table (the 0044 lock
|
||||||
|
class), so it's deferred to a quiesced-worker window.
|
||||||
|
|
||||||
|
Revision ID: 0045
|
||||||
|
Revises: 0044
|
||||||
|
Create Date: 2026-06-10
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0045"
|
||||||
|
down_revision: Union[str, None] = "0044"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"image_prediction",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column(
|
||||||
|
"image_record_id", sa.Integer(),
|
||||||
|
sa.ForeignKey("image_record.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("raw_name", sa.String(length=255), nullable=False),
|
||||||
|
sa.Column("category", sa.String(length=64), nullable=False),
|
||||||
|
sa.Column("score", sa.Float(), nullable=False),
|
||||||
|
sa.UniqueConstraint(
|
||||||
|
"image_record_id", "raw_name", name="image_raw_name",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_image_prediction_image", "image_prediction", ["image_record_id"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_image_prediction_name_score", "image_prediction",
|
||||||
|
["raw_name", "score"],
|
||||||
|
)
|
||||||
|
# No data backfill here — see the module docstring. The one-time copy from
|
||||||
|
# image_record.tagger_predictions runs as backfill_image_predictions_task
|
||||||
|
# (batched, resumable, idempotent), kept out of this transaction so web boots
|
||||||
|
# without waiting on a ~100 GB pass.
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_image_prediction_name_score", "image_prediction")
|
||||||
|
op.drop_index("ix_image_prediction_image", "image_prediction")
|
||||||
|
op.drop_table("image_prediction")
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
"""drop image_record.tagger_predictions (predictions normalized to image_prediction)
|
||||||
|
|
||||||
|
Final step of #768. The per-tag predictions now live in the image_prediction
|
||||||
|
table (backfilled from the JSON, read by suggestions + allowlist, written by
|
||||||
|
tag_and_embed). The old JSON column is dead weight — and it's the ~100 GB of
|
||||||
|
sub-0.70 score tail that bloated image_record's TOAST and broke DB backups
|
||||||
|
(#739). Dropping it is a fast catalog change; it does NOT reclaim the disk on
|
||||||
|
its own — run `VACUUM FULL image_record` (or pg_repack) afterward, off-hours,
|
||||||
|
to return the space to the OS so backups go small.
|
||||||
|
|
||||||
|
DROP COLUMN needs a brief ACCESS EXCLUSIVE lock on image_record; env.py's
|
||||||
|
lock_timeout guards it, so quiesce the ml-worker if a tagging run is in flight
|
||||||
|
(see the migration-lock reference). tagger_model_version is kept — it's the
|
||||||
|
"has this been tagged / is it current?" signal the backfill sweep reads.
|
||||||
|
|
||||||
|
Revision ID: 0046
|
||||||
|
Revises: 0045
|
||||||
|
Create Date: 2026-06-11
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0046"
|
||||||
|
down_revision: Union[str, None] = "0045"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.drop_column("image_record", "tagger_predictions")
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Re-add the column empty. The JSON data is not restored (it lived only in
|
||||||
|
# this column); a downgrade would re-tag or backfill from image_prediction
|
||||||
|
# separately if ever needed.
|
||||||
|
op.add_column(
|
||||||
|
"image_record",
|
||||||
|
sa.Column("tagger_predictions", sa.JSON(), nullable=True),
|
||||||
|
)
|
||||||
@@ -0,0 +1,175 @@
|
|||||||
|
"""series chapters become cosmetic dividers; pages become one series-global run
|
||||||
|
|
||||||
|
FC-6.x reframe (#789). A series is now ONE flat, series-global ordered run of
|
||||||
|
pages; chapters stop owning pages and become labeled dividers anchored to the
|
||||||
|
page that begins them.
|
||||||
|
|
||||||
|
Migration (order matters — series_page.chapter_id cascades, so it must be
|
||||||
|
dropped BEFORE any chapter row is deleted, or pages would cascade away):
|
||||||
|
a. Renumber series_page.page_number to a series-global 1..N (ordered by the
|
||||||
|
OLD (chapter_number, page_number)).
|
||||||
|
b. Add series_chapter.anchor_page_id and populate it with each chapter's first
|
||||||
|
page (lowest new page_number).
|
||||||
|
c. Drop series_page.chapter_id (severs the cascade link).
|
||||||
|
d. Prune chapters that shouldn't become dividers: empty/placeholder ones (no
|
||||||
|
anchor) and the redundant unlabeled chapter that would sit at page 1.
|
||||||
|
e. Reshape series_chapter into the divider: drop chapter_number,
|
||||||
|
is_placeholder, stated_page_start/end; make anchor_page_id NOT NULL +
|
||||||
|
UNIQUE + FK→series_page ON DELETE CASCADE.
|
||||||
|
|
||||||
|
Revision ID: 0047
|
||||||
|
Revises: 0046
|
||||||
|
Create Date: 2026-06-11
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0047"
|
||||||
|
down_revision: Union[str, None] = "0046"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
# a. series-global page numbering, preserving the old reading order.
|
||||||
|
op.execute(
|
||||||
|
"""
|
||||||
|
WITH ordered AS (
|
||||||
|
SELECT sp.id,
|
||||||
|
ROW_NUMBER() OVER (
|
||||||
|
PARTITION BY sp.series_tag_id
|
||||||
|
ORDER BY sc.chapter_number, sp.page_number, sp.id
|
||||||
|
) AS rn
|
||||||
|
FROM series_page sp
|
||||||
|
JOIN series_chapter sc ON sc.id = sp.chapter_id
|
||||||
|
)
|
||||||
|
UPDATE series_page sp
|
||||||
|
SET page_number = ordered.rn
|
||||||
|
FROM ordered
|
||||||
|
WHERE sp.id = ordered.id
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
|
||||||
|
# b. anchor each existing chapter at its first page (lowest new page_number).
|
||||||
|
op.add_column(
|
||||||
|
"series_chapter",
|
||||||
|
sa.Column("anchor_page_id", sa.Integer(), nullable=True),
|
||||||
|
)
|
||||||
|
op.execute(
|
||||||
|
"""
|
||||||
|
WITH firsts AS (
|
||||||
|
SELECT DISTINCT ON (sp.chapter_id)
|
||||||
|
sp.chapter_id, sp.id AS page_id
|
||||||
|
FROM series_page sp
|
||||||
|
ORDER BY sp.chapter_id, sp.page_number, sp.id
|
||||||
|
)
|
||||||
|
UPDATE series_chapter sc
|
||||||
|
SET anchor_page_id = firsts.page_id
|
||||||
|
FROM firsts
|
||||||
|
WHERE firsts.chapter_id = sc.id
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
|
||||||
|
# c. sever the ownership link (drops the FK + index with the column) BEFORE
|
||||||
|
# pruning chapters, so deleting a chapter can't cascade-delete its pages.
|
||||||
|
op.drop_column("series_page", "chapter_id")
|
||||||
|
|
||||||
|
# d. prune chapters that don't become dividers: placeholders / empty ones
|
||||||
|
# (no anchor), and the unlabeled chapter that would land redundantly at
|
||||||
|
# page 1 (the series just starts — no divider needed there).
|
||||||
|
op.execute(
|
||||||
|
"""
|
||||||
|
DELETE FROM series_chapter sc
|
||||||
|
USING (
|
||||||
|
SELECT sc2.id
|
||||||
|
FROM series_chapter sc2
|
||||||
|
LEFT JOIN series_page sp ON sp.id = sc2.anchor_page_id
|
||||||
|
WHERE sc2.anchor_page_id IS NULL
|
||||||
|
OR (sp.page_number = 1
|
||||||
|
AND sc2.title IS NULL
|
||||||
|
AND sc2.stated_part IS NULL)
|
||||||
|
) gone
|
||||||
|
WHERE sc.id = gone.id
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
|
||||||
|
# e. reshape into the divider model.
|
||||||
|
op.drop_column("series_chapter", "chapter_number")
|
||||||
|
op.drop_column("series_chapter", "is_placeholder")
|
||||||
|
op.drop_column("series_chapter", "stated_page_start")
|
||||||
|
op.drop_column("series_chapter", "stated_page_end")
|
||||||
|
op.alter_column("series_chapter", "anchor_page_id", nullable=False)
|
||||||
|
op.create_unique_constraint(
|
||||||
|
"uq_series_chapter_anchor_page", "series_chapter", ["anchor_page_id"]
|
||||||
|
)
|
||||||
|
op.create_foreign_key(
|
||||||
|
"fk_series_chapter_anchor_page",
|
||||||
|
"series_chapter",
|
||||||
|
"series_page",
|
||||||
|
["anchor_page_id"],
|
||||||
|
["id"],
|
||||||
|
ondelete="CASCADE",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Lossy: dividers can't be reconstructed as owning chapters. Collapse back to
|
||||||
|
# exactly one chapter per series that owns all its pages in order.
|
||||||
|
op.add_column(
|
||||||
|
"series_page", sa.Column("chapter_id", sa.Integer(), nullable=True)
|
||||||
|
)
|
||||||
|
op.drop_constraint(
|
||||||
|
"fk_series_chapter_anchor_page", "series_chapter", type_="foreignkey"
|
||||||
|
)
|
||||||
|
op.drop_constraint(
|
||||||
|
"uq_series_chapter_anchor_page", "series_chapter", type_="unique"
|
||||||
|
)
|
||||||
|
op.drop_column("series_chapter", "anchor_page_id")
|
||||||
|
op.add_column(
|
||||||
|
"series_chapter",
|
||||||
|
sa.Column(
|
||||||
|
"chapter_number", sa.Integer(), nullable=False, server_default="1"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.add_column(
|
||||||
|
"series_chapter",
|
||||||
|
sa.Column(
|
||||||
|
"is_placeholder", sa.Boolean(), nullable=False,
|
||||||
|
server_default="false",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.add_column(
|
||||||
|
"series_chapter",
|
||||||
|
sa.Column("stated_page_start", sa.Integer(), nullable=True),
|
||||||
|
)
|
||||||
|
op.add_column(
|
||||||
|
"series_chapter",
|
||||||
|
sa.Column("stated_page_end", sa.Integer(), nullable=True),
|
||||||
|
)
|
||||||
|
op.execute("DELETE FROM series_chapter")
|
||||||
|
op.execute(
|
||||||
|
"""
|
||||||
|
INSERT INTO series_chapter (series_tag_id, chapter_number)
|
||||||
|
SELECT DISTINCT series_tag_id, 1 FROM series_page
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
op.execute(
|
||||||
|
"""
|
||||||
|
UPDATE series_page sp
|
||||||
|
SET chapter_id = sc.id
|
||||||
|
FROM series_chapter sc
|
||||||
|
WHERE sc.series_tag_id = sp.series_tag_id
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
op.alter_column("series_page", "chapter_id", nullable=False)
|
||||||
|
op.create_foreign_key(
|
||||||
|
"fk_series_page_chapter",
|
||||||
|
"series_page",
|
||||||
|
"series_chapter",
|
||||||
|
["chapter_id"],
|
||||||
|
["id"],
|
||||||
|
ondelete="CASCADE",
|
||||||
|
)
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
"""series_page pending staging: status + nullable page_number (#789 Phase 2)
|
||||||
|
|
||||||
|
Pages added from a post no longer append straight into the run — they land
|
||||||
|
'pending' with a NULL page_number, staged grouped by their source post so the
|
||||||
|
operator can drop junk (text-free alts, bumpers) and place the keepers into the
|
||||||
|
sequence. A page only gets a series-global page_number once it's 'placed'.
|
||||||
|
|
||||||
|
Revision ID: 0048
|
||||||
|
Revises: 0047
|
||||||
|
Create Date: 2026-06-11
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0048"
|
||||||
|
down_revision: Union[str, None] = "0047"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"series_page",
|
||||||
|
sa.Column(
|
||||||
|
"status", sa.String(length=16), nullable=False,
|
||||||
|
server_default="placed",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.alter_column(
|
||||||
|
"series_page", "page_number",
|
||||||
|
existing_type=sa.Integer(), nullable=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Lossy: pending pages are unsorted staging rows with no order — drop them.
|
||||||
|
op.execute("DELETE FROM series_page WHERE status = 'pending'")
|
||||||
|
op.alter_column(
|
||||||
|
"series_page", "page_number",
|
||||||
|
existing_type=sa.Integer(), nullable=False,
|
||||||
|
)
|
||||||
|
op.drop_column("series_page", "status")
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
"""external_link table — off-platform file-host links found in post bodies
|
||||||
|
|
||||||
|
Creators host the real files on mega.nz / Google Drive / MediaFire / Dropbox /
|
||||||
|
Pixeldrain and link them in the post text. This table records each such link
|
||||||
|
(so nothing is silently dropped), and doubles as the dedup + dead-letter ledger
|
||||||
|
the download worker (a later slice) walks. `url` keeps the FULL link including
|
||||||
|
the `#fragment` — mega.nz's decryption key lives there; truncating it makes the
|
||||||
|
file undownloadable.
|
||||||
|
|
||||||
|
CHECK whitelists for host + status include the full enum up front (incl. the
|
||||||
|
download-worker statuses) so the worker slice needs no constraint migration.
|
||||||
|
|
||||||
|
Revision ID: 0049
|
||||||
|
Revises: 0048
|
||||||
|
Create Date: 2026-06-14
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0049"
|
||||||
|
down_revision: Union[str, None] = "0048"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"external_link",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column(
|
||||||
|
"post_id", sa.Integer(),
|
||||||
|
sa.ForeignKey("post.id", ondelete="CASCADE"), nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"artist_id", sa.Integer(),
|
||||||
|
sa.ForeignKey("artist.id", ondelete="SET NULL"), nullable=True,
|
||||||
|
),
|
||||||
|
sa.Column("host", sa.String(length=16), nullable=False),
|
||||||
|
sa.Column("url", sa.Text(), nullable=False),
|
||||||
|
sa.Column("label", sa.Text(), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"status", sa.String(length=16), nullable=False,
|
||||||
|
server_default="pending",
|
||||||
|
),
|
||||||
|
sa.Column("attempts", sa.Integer(), nullable=False, server_default="0"),
|
||||||
|
sa.Column("last_error", sa.Text(), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"attachment_id", sa.Integer(),
|
||||||
|
sa.ForeignKey("post_attachment.id", ondelete="SET NULL"),
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"created_at", sa.DateTime(timezone=True), nullable=False,
|
||||||
|
server_default=sa.func.now(),
|
||||||
|
),
|
||||||
|
sa.Column("completed_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
sa.Column("duration_seconds", sa.Float(), nullable=True),
|
||||||
|
sa.CheckConstraint(
|
||||||
|
"host IN ('mega','gdrive','mediafire','dropbox','pixeldrain')",
|
||||||
|
name="ck_external_link_host",
|
||||||
|
),
|
||||||
|
sa.CheckConstraint(
|
||||||
|
"status IN ('pending','downloading','downloaded','failed',"
|
||||||
|
"'skipped','dead')",
|
||||||
|
name="ck_external_link_status",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_external_link_post_id", "external_link", ["post_id"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_external_link_artist_id", "external_link", ["artist_id"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_external_link_status", "external_link", ["status"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"uq_external_link_post_url", "external_link", ["post_id", "url"],
|
||||||
|
unique=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("uq_external_link_post_url", table_name="external_link")
|
||||||
|
op.drop_index("ix_external_link_status", table_name="external_link")
|
||||||
|
op.drop_index("ix_external_link_artist_id", table_name="external_link")
|
||||||
|
op.drop_index("ix_external_link_post_id", table_name="external_link")
|
||||||
|
op.drop_table("external_link")
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
"""import_settings: per-host enable toggles for external file-host downloads
|
||||||
|
|
||||||
|
Operator levers (#830): disable a single host (e.g. mega.nz when it's
|
||||||
|
rate-limiting/banning) without touching the others. The worker reads these via
|
||||||
|
getattr and defaults to enabled, so the toggles default TRUE (works out of the
|
||||||
|
box, rule #26).
|
||||||
|
|
||||||
|
Revision ID: 0050
|
||||||
|
Revises: 0049
|
||||||
|
Create Date: 2026-06-14
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0050"
|
||||||
|
down_revision: Union[str, None] = "0049"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
_HOSTS = ("mega", "gdrive", "mediafire", "dropbox", "pixeldrain")
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
for host in _HOSTS:
|
||||||
|
op.add_column(
|
||||||
|
"import_settings",
|
||||||
|
sa.Column(
|
||||||
|
f"extdl_{host}_enabled", sa.Boolean(), nullable=False,
|
||||||
|
server_default=sa.true(),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
for host in _HOSTS:
|
||||||
|
op.drop_column("import_settings", f"extdl_{host}_enabled")
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
"""image_record: source_url + source_filehash (inline-image localization)
|
||||||
|
|
||||||
|
#830 Phase 2. To render a post body faithfully we serve LOCAL copies of inline
|
||||||
|
images instead of hotlinking the public CDN. The join key between a body
|
||||||
|
`<img src=CDN>` and the local file is the CDN's 32-hex filehash (the same
|
||||||
|
identity extract_media dedups by). Persist it (indexed) plus the full source
|
||||||
|
URL for provenance/debugging. Both NULL for filesystem-imported / pre-existing
|
||||||
|
rows — those fall back to hotlinking until re-downloaded.
|
||||||
|
|
||||||
|
Revision ID: 0051
|
||||||
|
Revises: 0050
|
||||||
|
Create Date: 2026-06-14
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0051"
|
||||||
|
down_revision: Union[str, None] = "0050"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column("image_record", sa.Column("source_url", sa.Text(), nullable=True))
|
||||||
|
op.add_column(
|
||||||
|
"image_record", sa.Column("source_filehash", sa.String(length=32), nullable=True)
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_image_record_source_filehash", "image_record", ["source_filehash"]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_image_record_source_filehash", table_name="image_record")
|
||||||
|
op.drop_column("image_record", "source_filehash")
|
||||||
|
op.drop_column("image_record", "source_url")
|
||||||
@@ -33,12 +33,6 @@ def create_app() -> Quart:
|
|||||||
|
|
||||||
app = Quart(__name__)
|
app = Quart(__name__)
|
||||||
app.secret_key = cfg.secret_key
|
app.secret_key = cfg.secret_key
|
||||||
# FC-5: legacy IR ingest JSON can run to tens of MB (hundreds of
|
|
||||||
# thousands of image_tag_associations). Werkzeug's default form
|
|
||||||
# memory cap is 500KB; raise both ceilings so the multipart upload
|
|
||||||
# for /api/migrate/ir_ingest doesn't 413.
|
|
||||||
app.config["MAX_CONTENT_LENGTH"] = 1024 * 1024 * 1024 # 1 GB
|
|
||||||
app.config["MAX_FORM_MEMORY_SIZE"] = 1024 * 1024 * 1024 # 1 GB
|
|
||||||
|
|
||||||
for bp in all_blueprints():
|
for bp in all_blueprints():
|
||||||
app.register_blueprint(bp)
|
app.register_blueprint(bp)
|
||||||
|
|||||||
@@ -26,7 +26,6 @@ def all_blueprints() -> list[Blueprint]:
|
|||||||
from .extension import extension_bp
|
from .extension import extension_bp
|
||||||
from .gallery import gallery_bp
|
from .gallery import gallery_bp
|
||||||
from .import_admin import import_admin_bp
|
from .import_admin import import_admin_bp
|
||||||
from .migrate import migrate_bp
|
|
||||||
from .ml_admin import ml_admin_bp
|
from .ml_admin import ml_admin_bp
|
||||||
from .platforms import platforms_bp
|
from .platforms import platforms_bp
|
||||||
from .posts import posts_bp
|
from .posts import posts_bp
|
||||||
@@ -54,7 +53,6 @@ def all_blueprints() -> list[Blueprint]:
|
|||||||
admin_bp,
|
admin_bp,
|
||||||
cleanup_bp,
|
cleanup_bp,
|
||||||
import_admin_bp,
|
import_admin_bp,
|
||||||
migrate_bp,
|
|
||||||
suggestions_bp,
|
suggestions_bp,
|
||||||
allowlist_bp,
|
allowlist_bp,
|
||||||
aliases_bp,
|
aliases_bp,
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
"""Shared API response helpers."""
|
||||||
|
|
||||||
|
from quart import jsonify
|
||||||
|
|
||||||
|
|
||||||
|
def error_response(
|
||||||
|
error: str, *, status: int = 400, detail: str | None = None, **extra,
|
||||||
|
):
|
||||||
|
"""JSON error body + HTTP status. `detail` is included only when given;
|
||||||
|
`extra` keys are merged into the body. Returns the (response, status)
|
||||||
|
tuple Quart expects. Imported as `_bad` by the blueprints."""
|
||||||
|
body = {"error": error}
|
||||||
|
if detail is not None:
|
||||||
|
body["detail"] = detail
|
||||||
|
body.update(extra)
|
||||||
|
return jsonify(body), status
|
||||||
+149
-7
@@ -6,6 +6,8 @@ Five action surfaces:
|
|||||||
DELETE /api/admin/tags/<int:tag_id> (Tier B)
|
DELETE /api/admin/tags/<int:tag_id> (Tier B)
|
||||||
POST /api/admin/tags/<int:dest_id>/merge (Tier B)
|
POST /api/admin/tags/<int:dest_id>/merge (Tier B)
|
||||||
POST /api/admin/tags/prune-unused (Tier A)
|
POST /api/admin/tags/prune-unused (Tier A)
|
||||||
|
POST /api/admin/posts/prune-bare (Tier A)
|
||||||
|
POST /api/admin/tags/purge-legacy (Tier A)
|
||||||
GET /api/admin/tags/<int:tag_id>/usage-count (helper)
|
GET /api/admin/tags/<int:tag_id>/usage-count (helper)
|
||||||
|
|
||||||
Tier-C ops take a dry_run body flag (returns projection inline,
|
Tier-C ops take a dry_run body flag (returns projection inline,
|
||||||
@@ -18,21 +20,16 @@ from __future__ import annotations
|
|||||||
import hashlib
|
import hashlib
|
||||||
|
|
||||||
from quart import Blueprint, jsonify, request
|
from quart import Blueprint, jsonify, request
|
||||||
from sqlalchemy import select
|
from sqlalchemy import select, text
|
||||||
|
|
||||||
from ..extensions import get_session
|
from ..extensions import get_session
|
||||||
from ..models import Artist
|
from ..models import Artist
|
||||||
from ..services.cleanup_service import project_artist_cascade, project_bulk_image_delete
|
from ..services.cleanup_service import project_artist_cascade, project_bulk_image_delete
|
||||||
|
from ._responses import error_response as _bad
|
||||||
|
|
||||||
admin_bp = Blueprint("admin", __name__, url_prefix="/api/admin")
|
admin_bp = Blueprint("admin", __name__, url_prefix="/api/admin")
|
||||||
|
|
||||||
|
|
||||||
def _bad(error: str, *, status: int = 400, **extra):
|
|
||||||
body = {"error": error}
|
|
||||||
body.update(extra)
|
|
||||||
return jsonify(body), status
|
|
||||||
|
|
||||||
|
|
||||||
def _bulk_image_confirm_token(image_ids: list[int]) -> str:
|
def _bulk_image_confirm_token(image_ids: list[int]) -> str:
|
||||||
"""Stable 8-hex token derived from the sorted id list. Mutates
|
"""Stable 8-hex token derived from the sorted id list. Mutates
|
||||||
when the selection changes; stays the same across modal opens of
|
when the selection changes; stays the same across modal opens of
|
||||||
@@ -206,3 +203,148 @@ async def tags_prune_unused():
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
return jsonify(result)
|
return jsonify(result)
|
||||||
|
|
||||||
|
|
||||||
|
@admin_bp.route("/posts/prune-bare", methods=["POST"])
|
||||||
|
async def posts_prune_bare():
|
||||||
|
"""Tier-A: delete bare posts — Post rows with no linked images (primary OR
|
||||||
|
provenance) and no attachments. Dry-run preview list IS the prompt: UI calls
|
||||||
|
with dry_run=true first, shows the count + sample, operator confirms by
|
||||||
|
re-calling with dry_run=false. Same preview/apply-parity predicate as the
|
||||||
|
prune itself, so the preview can't diverge from the delete."""
|
||||||
|
from ..services.cleanup_service import prune_bare_posts
|
||||||
|
|
||||||
|
body = await request.get_json(silent=True) or {}
|
||||||
|
dry_run = bool(body.get("dry_run", False))
|
||||||
|
|
||||||
|
async with get_session() as session:
|
||||||
|
result = await session.run_sync(
|
||||||
|
lambda sync_sess: prune_bare_posts(
|
||||||
|
sync_sess, dry_run=dry_run,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return jsonify(result)
|
||||||
|
|
||||||
|
|
||||||
|
@admin_bp.route("/tags/purge-legacy", methods=["POST"])
|
||||||
|
async def tags_purge_legacy():
|
||||||
|
"""Tier-A: delete legacy IR-migration tags — archive/post/artist
|
||||||
|
kinds (e.g. `BlenderKnight:Hannah_BJ_Loops`) PLUS general tags with
|
||||||
|
a legacy name prefix (`source:*`, from IR's source kind that fell
|
||||||
|
back to general). dry-run preview returns per-kind + per-prefix
|
||||||
|
counts + a sample so the UI shows exactly what'll go before the
|
||||||
|
operator confirms with dry_run=false."""
|
||||||
|
from ..services.cleanup_service import purge_legacy_tags
|
||||||
|
|
||||||
|
body = await request.get_json(silent=True) or {}
|
||||||
|
dry_run = bool(body.get("dry_run", False))
|
||||||
|
|
||||||
|
async with get_session() as session:
|
||||||
|
result = await session.run_sync(
|
||||||
|
lambda sync_sess: purge_legacy_tags(sync_sess, dry_run=dry_run)
|
||||||
|
)
|
||||||
|
return jsonify(result)
|
||||||
|
|
||||||
|
|
||||||
|
@admin_bp.route("/tags/reset-content", methods=["POST"])
|
||||||
|
async def tags_reset_content():
|
||||||
|
"""Tier-A: delete ALL general + character tags (the Camie-suggestable
|
||||||
|
content vocabulary) so the operator can re-tag from scratch via
|
||||||
|
auto-suggest. fandom + series tags + series_page ordering are preserved,
|
||||||
|
and image_prediction rows are untouched so suggestions repopulate.
|
||||||
|
dry-run preview returns per-kind counts + applications + a sample so the
|
||||||
|
UI shows exactly what'll go before the operator confirms (dry_run=false).
|
||||||
|
Irreversible except via DB backup restore."""
|
||||||
|
from ..services.cleanup_service import reset_content_tagging
|
||||||
|
|
||||||
|
body = await request.get_json(silent=True) or {}
|
||||||
|
dry_run = bool(body.get("dry_run", False))
|
||||||
|
|
||||||
|
async with get_session() as session:
|
||||||
|
result = await session.run_sync(
|
||||||
|
lambda sync_sess: reset_content_tagging(sync_sess, dry_run=dry_run)
|
||||||
|
)
|
||||||
|
return jsonify(result)
|
||||||
|
|
||||||
|
|
||||||
|
@admin_bp.route("/tags/normalize", methods=["POST"])
|
||||||
|
async def tags_normalize():
|
||||||
|
"""#714: retro-normalize existing tags to the #701 canonical form (Title
|
||||||
|
Case + collapsed whitespace) and merge case/whitespace-variant duplicates.
|
||||||
|
|
||||||
|
dry_run=true (default) returns a projection inline — group/collision/rename
|
||||||
|
counts + a sample of the changes — so the UI shows exactly what'll happen.
|
||||||
|
dry_run=false dispatches the long-running maintenance task (the merge FK
|
||||||
|
repoints can touch many tags); the UI tails the activity dashboard for the
|
||||||
|
summary. Idempotent; back up first (the merges are irreversible)."""
|
||||||
|
from ..services.tag_service import normalize_existing_tags
|
||||||
|
|
||||||
|
body = await request.get_json(silent=True) or {}
|
||||||
|
dry_run = bool(body.get("dry_run", True))
|
||||||
|
|
||||||
|
if dry_run:
|
||||||
|
async with get_session() as session:
|
||||||
|
result = await normalize_existing_tags(session, dry_run=True)
|
||||||
|
return jsonify(result)
|
||||||
|
|
||||||
|
from ..tasks.admin import normalize_tags_task
|
||||||
|
|
||||||
|
async_result = normalize_tags_task.delay()
|
||||||
|
return jsonify({"task_id": async_result.id, "status": "queued"}), 202
|
||||||
|
|
||||||
|
|
||||||
|
@admin_bp.route("/maintenance/db-stats", methods=["GET"])
|
||||||
|
async def db_stats():
|
||||||
|
"""Per-table bloat readout (pg_stat_user_tables) for the high-churn tables
|
||||||
|
so the operator can see when a VACUUM is worth running."""
|
||||||
|
from ..tasks.maintenance import VACUUM_TABLES
|
||||||
|
|
||||||
|
wanted = set(VACUUM_TABLES)
|
||||||
|
async with get_session() as session:
|
||||||
|
rows = (await session.execute(text(
|
||||||
|
"SELECT relname, n_live_tup, n_dead_tup, last_vacuum, "
|
||||||
|
"last_autovacuum, last_analyze FROM pg_stat_user_tables"
|
||||||
|
))).all()
|
||||||
|
|
||||||
|
def _iso(v):
|
||||||
|
return v.isoformat() if v is not None else None
|
||||||
|
|
||||||
|
out = []
|
||||||
|
for r in rows:
|
||||||
|
if r.relname not in wanted:
|
||||||
|
continue
|
||||||
|
live = r.n_live_tup or 0
|
||||||
|
dead = r.n_dead_tup or 0
|
||||||
|
total = live + dead
|
||||||
|
out.append({
|
||||||
|
"table": r.relname,
|
||||||
|
"live": live,
|
||||||
|
"dead": dead,
|
||||||
|
"dead_pct": round(100 * dead / total, 1) if total else 0.0,
|
||||||
|
"last_vacuum": _iso(r.last_vacuum),
|
||||||
|
"last_autovacuum": _iso(r.last_autovacuum),
|
||||||
|
"last_analyze": _iso(r.last_analyze),
|
||||||
|
})
|
||||||
|
out.sort(key=lambda t: t["dead"], reverse=True)
|
||||||
|
return jsonify({"tables": out})
|
||||||
|
|
||||||
|
|
||||||
|
@admin_bp.route("/maintenance/vacuum", methods=["POST"])
|
||||||
|
async def trigger_vacuum():
|
||||||
|
"""Operator-triggered VACUUM (ANALYZE) over the high-churn tables — the
|
||||||
|
same maintenance-queue task the weekly Beat schedule runs."""
|
||||||
|
from ..tasks.maintenance import vacuum_analyze
|
||||||
|
|
||||||
|
vacuum_analyze.delay()
|
||||||
|
return jsonify({"status": "queued"}), 202
|
||||||
|
|
||||||
|
|
||||||
|
@admin_bp.route("/maintenance/reextract-archives", methods=["POST"])
|
||||||
|
async def trigger_reextract_archives():
|
||||||
|
"""Operator-triggered re-extract (#713): PostAttachments that are actually
|
||||||
|
archives but were filed opaquely (pre magic-byte gate) get extracted and
|
||||||
|
their members linked to the post. Idempotent; runs on the maintenance queue."""
|
||||||
|
from ..tasks.admin import reextract_archive_attachments_task
|
||||||
|
|
||||||
|
async_result = reextract_archive_attachments_task.delay()
|
||||||
|
return jsonify({"task_id": async_result.id, "status": "queued"}), 202
|
||||||
|
|||||||
@@ -31,18 +31,13 @@ from sqlalchemy import select
|
|||||||
from ..extensions import get_session
|
from ..extensions import get_session
|
||||||
from ..models import LibraryAuditRun
|
from ..models import LibraryAuditRun
|
||||||
from ..services import cleanup_service
|
from ..services import cleanup_service
|
||||||
|
from ._responses import error_response as _bad
|
||||||
|
|
||||||
cleanup_bp = Blueprint("cleanup", __name__, url_prefix="/api/cleanup")
|
cleanup_bp = Blueprint("cleanup", __name__, url_prefix="/api/cleanup")
|
||||||
|
|
||||||
IMAGES_ROOT = Path("/images")
|
IMAGES_ROOT = Path("/images")
|
||||||
|
|
||||||
|
|
||||||
def _bad(error: str, *, status: int = 400, **extra):
|
|
||||||
body = {"error": error}
|
|
||||||
body.update(extra)
|
|
||||||
return jsonify(body), status
|
|
||||||
|
|
||||||
|
|
||||||
def _min_dim_token(min_w: int, min_h: int) -> str:
|
def _min_dim_token(min_w: int, min_h: int) -> str:
|
||||||
# SHA-256 (not MD5) — Web Crypto's subtle.digest rejects MD5; both
|
# SHA-256 (not MD5) — Web Crypto's subtle.digest rejects MD5; both
|
||||||
# sides use SHA-256 truncated to 8 hex chars.
|
# sides use SHA-256 truncated to 8 hex chars.
|
||||||
@@ -159,12 +154,15 @@ async def audit_history():
|
|||||||
limit = min(int(request.args.get("limit", "20")), 100)
|
limit = min(int(request.args.get("limit", "20")), 100)
|
||||||
except ValueError:
|
except ValueError:
|
||||||
return _bad("invalid_limit")
|
return _bad("invalid_limit")
|
||||||
|
# Optional rule filter so a card can reconnect to ITS latest run on mount
|
||||||
|
# (?rule=transparency&limit=1) — the audit survives navigation; the UI
|
||||||
|
# rehydrates from this rather than losing the in-flight scan.
|
||||||
|
rule = request.args.get("rule") or None
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
rows = (await session.execute(
|
stmt = select(LibraryAuditRun).order_by(LibraryAuditRun.id.desc())
|
||||||
select(LibraryAuditRun)
|
if rule is not None:
|
||||||
.order_by(LibraryAuditRun.id.desc())
|
stmt = stmt.where(LibraryAuditRun.rule == rule)
|
||||||
.limit(limit)
|
rows = (await session.execute(stmt.limit(limit))).scalars().all()
|
||||||
)).scalars().all()
|
|
||||||
return jsonify({"runs": [_serialize_audit_run(r) for r in rows]})
|
return jsonify({"runs": [_serialize_audit_run(r) for r in rows]})
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ from ..services.credential_service import (
|
|||||||
UnknownPlatformError,
|
UnknownPlatformError,
|
||||||
WrongAuthTypeError,
|
WrongAuthTypeError,
|
||||||
)
|
)
|
||||||
|
from ._responses import error_response as _bad
|
||||||
|
|
||||||
credentials_bp = Blueprint("credentials", __name__, url_prefix="/api/credentials")
|
credentials_bp = Blueprint("credentials", __name__, url_prefix="/api/credentials")
|
||||||
|
|
||||||
@@ -38,14 +39,6 @@ def _get_crypto() -> CredentialCrypto:
|
|||||||
return _crypto
|
return _crypto
|
||||||
|
|
||||||
|
|
||||||
def _bad(error: str, *, status: int = 400, detail: str | None = None, **extra):
|
|
||||||
body = {"error": error}
|
|
||||||
if detail is not None:
|
|
||||||
body["detail"] = detail
|
|
||||||
body.update(extra)
|
|
||||||
return jsonify(body), status
|
|
||||||
|
|
||||||
|
|
||||||
async def _ext_key_ok(session) -> bool:
|
async def _ext_key_ok(session) -> bool:
|
||||||
"""If X-Extension-Key is supplied, it must match the stored value.
|
"""If X-Extension-Key is supplied, it must match the stored value.
|
||||||
Missing header → True (browser path; accepted per homelab posture).
|
Missing header → True (browser path; accepted per homelab posture).
|
||||||
@@ -124,3 +117,58 @@ async def delete_credential(platform: str):
|
|||||||
except LookupError:
|
except LookupError:
|
||||||
return _bad("not_found", status=404)
|
return _bad("not_found", status=404)
|
||||||
return "", 204
|
return "", 204
|
||||||
|
|
||||||
|
|
||||||
|
@credentials_bp.route("/<platform>/verify", methods=["POST"])
|
||||||
|
async def verify_credential(platform: str):
|
||||||
|
"""Test the stored credential against one of the platform's enabled sources,
|
||||||
|
WITHOUT downloading. Routes through the platform's backend
|
||||||
|
(download_backends.verify_credential) — native ingester for Patreon, an
|
||||||
|
authenticated API page; gallery-dl --simulate for the rest. On success
|
||||||
|
stamps last_verified. Returns {valid: bool|null, reason, last_verified?};
|
||||||
|
valid=null means "couldn't test" (no credential, no enabled source, or an
|
||||||
|
inconclusive network/drift result)."""
|
||||||
|
from ..models import Artist, Source
|
||||||
|
from ..services.download_backends import verify_source_credential
|
||||||
|
|
||||||
|
async with get_session() as session:
|
||||||
|
if not await _ext_key_ok(session):
|
||||||
|
return _bad("unauthorized", status=401)
|
||||||
|
svc = CredentialService(session, _get_crypto())
|
||||||
|
record = await svc.get(platform)
|
||||||
|
if record is None:
|
||||||
|
return jsonify({"valid": None, "reason": "No credential stored for this platform."})
|
||||||
|
|
||||||
|
# Pick an enabled source for this platform to point the probe at.
|
||||||
|
row = (await session.execute(
|
||||||
|
select(Source, Artist)
|
||||||
|
.join(Artist, Artist.id == Source.artist_id)
|
||||||
|
.where(Source.platform == platform, Source.enabled.is_(True))
|
||||||
|
.order_by(Source.id.asc())
|
||||||
|
)).first()
|
||||||
|
if row is None:
|
||||||
|
return jsonify({
|
||||||
|
"valid": None,
|
||||||
|
"reason": "No enabled source for this platform to verify against — add a subscription first.",
|
||||||
|
})
|
||||||
|
source, artist = row
|
||||||
|
|
||||||
|
cookies_path = await svc.get_cookies_path(platform)
|
||||||
|
auth_token = await svc.get_token(platform)
|
||||||
|
|
||||||
|
ok, message = await verify_source_credential(
|
||||||
|
platform=platform,
|
||||||
|
url=source.url,
|
||||||
|
artist_slug=artist.slug,
|
||||||
|
config_overrides=source.config_overrides or {},
|
||||||
|
cookies_path=str(cookies_path) if cookies_path else None,
|
||||||
|
auth_token=auth_token,
|
||||||
|
images_root=Path("/images"),
|
||||||
|
)
|
||||||
|
|
||||||
|
last_verified = None
|
||||||
|
if ok:
|
||||||
|
async with get_session() as session:
|
||||||
|
ts = await CredentialService(session, _get_crypto()).mark_verified(platform)
|
||||||
|
last_verified = ts.isoformat() if ts else None
|
||||||
|
return jsonify({"valid": ok, "reason": message, "last_verified": last_verified})
|
||||||
|
|||||||
@@ -44,6 +44,9 @@ def _list_record(event: DownloadEvent, source: Source | None, artist: Artist | N
|
|||||||
"bytes_downloaded": event.bytes_downloaded,
|
"bytes_downloaded": event.bytes_downloaded,
|
||||||
"error": event.error,
|
"error": event.error,
|
||||||
"summary": _summary_from_metadata(event.metadata_),
|
"summary": _summary_from_metadata(event.metadata_),
|
||||||
|
# plan #709: mid-walk live counts for a RUNNING native-ingester event
|
||||||
|
# (None otherwise; phase 3 overwrites metadata with run_stats on finish).
|
||||||
|
"live": (event.metadata_ or {}).get("live"),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -126,6 +129,54 @@ async def downloads_stats():
|
|||||||
return jsonify(out)
|
return jsonify(out)
|
||||||
|
|
||||||
|
|
||||||
|
@downloads_bp.route("/activity", methods=["GET"])
|
||||||
|
async def downloads_activity():
|
||||||
|
"""Hourly download-event counts over the last `?hours=` (default 24).
|
||||||
|
|
||||||
|
Returns a fixed-length, oldest-first bucket array so the UI can render
|
||||||
|
a sparkline directly. Bucketing is done in Python against UTC to dodge
|
||||||
|
session-timezone ambiguity in SQL date_trunc.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
hours = int(request.args.get("hours", "24"))
|
||||||
|
except ValueError:
|
||||||
|
return jsonify({"error": "invalid_hours"}), 400
|
||||||
|
hours = max(1, min(168, hours))
|
||||||
|
|
||||||
|
now = datetime.now(UTC)
|
||||||
|
end = now.replace(minute=0, second=0, microsecond=0)
|
||||||
|
start = end - timedelta(hours=hours - 1)
|
||||||
|
buckets = [
|
||||||
|
{"hour": (start + timedelta(hours=i)).isoformat(),
|
||||||
|
"ok": 0, "error": 0, "other": 0, "total": 0}
|
||||||
|
for i in range(hours)
|
||||||
|
]
|
||||||
|
|
||||||
|
async with get_session() as session:
|
||||||
|
rows = (await session.execute(
|
||||||
|
select(DownloadEvent.started_at, DownloadEvent.status)
|
||||||
|
.where(DownloadEvent.started_at >= start)
|
||||||
|
)).all()
|
||||||
|
|
||||||
|
for started_at, status in rows:
|
||||||
|
if started_at is None:
|
||||||
|
continue
|
||||||
|
sa = started_at if started_at.tzinfo else started_at.replace(tzinfo=UTC)
|
||||||
|
idx = int((sa - start).total_seconds() // 3600)
|
||||||
|
if not (0 <= idx < hours):
|
||||||
|
continue
|
||||||
|
b = buckets[idx]
|
||||||
|
if status == "ok":
|
||||||
|
b["ok"] += 1
|
||||||
|
elif status == "error":
|
||||||
|
b["error"] += 1
|
||||||
|
else:
|
||||||
|
b["other"] += 1
|
||||||
|
b["total"] += 1
|
||||||
|
|
||||||
|
return jsonify({"hours": hours, "buckets": buckets})
|
||||||
|
|
||||||
|
|
||||||
@downloads_bp.route("/<int:event_id>", methods=["GET"])
|
@downloads_bp.route("/<int:event_id>", methods=["GET"])
|
||||||
async def get_download(event_id: int):
|
async def get_download(event_id: int):
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
@@ -139,3 +190,20 @@ async def get_download(event_id: int):
|
|||||||
return jsonify({"error": "not_found"}), 404
|
return jsonify({"error": "not_found"}), 404
|
||||||
event, source, artist = row
|
event, source, artist = row
|
||||||
return jsonify(_detail_record(event, source, artist))
|
return jsonify(_detail_record(event, source, artist))
|
||||||
|
|
||||||
|
|
||||||
|
@downloads_bp.route("/recover-stalled", methods=["POST"])
|
||||||
|
async def recover_stalled():
|
||||||
|
"""Trigger the recover_stalled_download_events sweep on demand.
|
||||||
|
|
||||||
|
The same sweep runs every 5 min via Beat (see celery_app.beat_schedule);
|
||||||
|
this endpoint exists so the operator can force-clear stuck pending/
|
||||||
|
running download_events from the Subscriptions → Downloads maintenance
|
||||||
|
menu without waiting for the next scheduled tick.
|
||||||
|
"""
|
||||||
|
# Local import: avoids registering maintenance tasks during blueprint
|
||||||
|
# import (Celery task discovery races with the API import otherwise).
|
||||||
|
from ..tasks.maintenance import recover_stalled_download_events
|
||||||
|
|
||||||
|
recover_stalled_download_events.delay()
|
||||||
|
return jsonify({"queued": True}), 202
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ from ..services.extension_service import (
|
|||||||
UnknownPlatformError,
|
UnknownPlatformError,
|
||||||
)
|
)
|
||||||
from ..services.source_service import KNOWN_PLATFORMS
|
from ..services.source_service import KNOWN_PLATFORMS
|
||||||
|
from ._responses import error_response as _bad
|
||||||
|
|
||||||
extension_bp = Blueprint("extension", __name__, url_prefix="/api/extension")
|
extension_bp = Blueprint("extension", __name__, url_prefix="/api/extension")
|
||||||
|
|
||||||
@@ -30,12 +31,6 @@ XPI_DIR = Path("/app/frontend/dist/extension")
|
|||||||
_XPI_VERSION_RE = re.compile(r"fabledcurator-(?P<version>[\w.-]+)\.xpi$")
|
_XPI_VERSION_RE = re.compile(r"fabledcurator-(?P<version>[\w.-]+)\.xpi$")
|
||||||
|
|
||||||
|
|
||||||
def _bad(error: str, *, status: int = 400, **extra):
|
|
||||||
body = {"error": error}
|
|
||||||
body.update(extra)
|
|
||||||
return jsonify(body), status
|
|
||||||
|
|
||||||
|
|
||||||
async def _ext_key_required(session) -> bool:
|
async def _ext_key_required(session) -> bool:
|
||||||
"""Unlike /api/credentials (which accepts the browser path with no
|
"""Unlike /api/credentials (which accepts the browser path with no
|
||||||
header), quick-add-source writes server state and must be explicitly
|
header), quick-add-source writes server state and must be explicitly
|
||||||
@@ -62,6 +57,24 @@ def _sha256(path: Path) -> str:
|
|||||||
return h.hexdigest()
|
return h.hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
@extension_bp.route("/probe", methods=["GET"])
|
||||||
|
async def probe_source():
|
||||||
|
"""Read-only resolution of a creator-page URL: tells the extension
|
||||||
|
whether this URL is already a Source, is for an Artist that exists
|
||||||
|
but with a different URL, is brand new, or doesn't match any known
|
||||||
|
platform pattern. Drives the content-script chip's color/copy
|
||||||
|
BEFORE the operator clicks, so the button can show 'already added'
|
||||||
|
without requiring an add-attempt."""
|
||||||
|
url = (request.args.get("url") or "").strip()
|
||||||
|
if not url:
|
||||||
|
return _bad("invalid_body", detail="url query parameter is required")
|
||||||
|
async with get_session() as session:
|
||||||
|
if not await _ext_key_required(session):
|
||||||
|
return _bad("unauthorized", status=401)
|
||||||
|
result = await ExtensionService(session).probe(url)
|
||||||
|
return jsonify(result)
|
||||||
|
|
||||||
|
|
||||||
@extension_bp.route("/quick-add-source", methods=["POST"])
|
@extension_bp.route("/quick-add-source", methods=["POST"])
|
||||||
async def quick_add_source():
|
async def quick_add_source():
|
||||||
body = await request.get_json(silent=True)
|
body = await request.get_json(silent=True)
|
||||||
|
|||||||
+131
-44
@@ -1,4 +1,6 @@
|
|||||||
"""Gallery API: cursor scroll, timeline, jump, image detail."""
|
"""Gallery API: cursor scroll, timeline, jump, image detail, facets."""
|
||||||
|
|
||||||
|
from datetime import UTC, datetime, timedelta
|
||||||
|
|
||||||
from quart import Blueprint, jsonify, request
|
from quart import Blueprint, jsonify, request
|
||||||
|
|
||||||
@@ -8,47 +10,88 @@ from ..services.gallery_service import GalleryService
|
|||||||
gallery_bp = Blueprint("gallery", __name__, url_prefix="/api/gallery")
|
gallery_bp = Blueprint("gallery", __name__, url_prefix="/api/gallery")
|
||||||
|
|
||||||
|
|
||||||
|
def _image_json(i):
|
||||||
|
"""Serialize a GalleryImage for the scroll/similar list responses."""
|
||||||
|
return {
|
||||||
|
"id": i.id,
|
||||||
|
"sha256": i.sha256,
|
||||||
|
"mime": i.mime,
|
||||||
|
"width": i.width,
|
||||||
|
"height": i.height,
|
||||||
|
"created_at": i.created_at.isoformat(),
|
||||||
|
"posted_at": i.posted_at.isoformat() if i.posted_at else None,
|
||||||
|
"thumbnail_url": i.thumbnail_url,
|
||||||
|
"artist": i.artist,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_date(raw):
|
||||||
|
"""Parse a YYYY-MM-DD query value to a UTC midnight datetime, or None.
|
||||||
|
Raises ValueError (→ 400) on a malformed value."""
|
||||||
|
if not raw:
|
||||||
|
return None
|
||||||
|
return datetime.strptime(raw, "%Y-%m-%d").replace(tzinfo=UTC)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_filters():
|
||||||
|
"""Parse the composable gallery filters from query args, returning
|
||||||
|
``(filters_dict, sort)``. Raises ValueError (→ 400) on malformed ids/dates.
|
||||||
|
|
||||||
|
`tag_id` accepts a single id or a comma-separated list (AND); `media` is
|
||||||
|
image|video; `sort` is newest|oldest; `platform` selects one platform
|
||||||
|
(or the UNSOURCED_PLATFORM sentinel); `untagged`/`no_artist` are boolean
|
||||||
|
flags; `date_from`/`date_to` are inclusive calendar-day bounds (date_to is
|
||||||
|
widened by a day so the whole day is covered by the service's half-open
|
||||||
|
`< date_to`)."""
|
||||||
|
tag_raw = request.args.get("tag_id")
|
||||||
|
tag_ids = (
|
||||||
|
[int(x) for x in tag_raw.split(",") if x.strip()] if tag_raw else None
|
||||||
|
) or None
|
||||||
|
post_id_raw = request.args.get("post_id")
|
||||||
|
post_id = int(post_id_raw) if post_id_raw else None
|
||||||
|
artist_id_raw = request.args.get("artist_id")
|
||||||
|
artist_id = int(artist_id_raw) if artist_id_raw else None
|
||||||
|
media = request.args.get("media")
|
||||||
|
media_type = media if media in ("image", "video") else None
|
||||||
|
sort = request.args.get("sort")
|
||||||
|
sort = sort if sort in ("newest", "oldest") else "newest"
|
||||||
|
platform = request.args.get("platform") or None
|
||||||
|
untagged = request.args.get("untagged") in ("1", "true", "yes")
|
||||||
|
no_artist = request.args.get("no_artist") in ("1", "true", "yes")
|
||||||
|
date_from = _parse_date(request.args.get("date_from"))
|
||||||
|
date_to = _parse_date(request.args.get("date_to"))
|
||||||
|
if date_to is not None:
|
||||||
|
date_to += timedelta(days=1) # inclusive of the date_to calendar day
|
||||||
|
filters = {
|
||||||
|
"tag_ids": tag_ids, "post_id": post_id, "artist_id": artist_id,
|
||||||
|
"media_type": media_type, "platform": platform,
|
||||||
|
"untagged": untagged, "no_artist": no_artist,
|
||||||
|
"date_from": date_from, "date_to": date_to,
|
||||||
|
}
|
||||||
|
return filters, sort
|
||||||
|
|
||||||
|
|
||||||
@gallery_bp.route("/scroll", methods=["GET"])
|
@gallery_bp.route("/scroll", methods=["GET"])
|
||||||
async def scroll():
|
async def scroll():
|
||||||
cursor = request.args.get("cursor") or None
|
cursor = request.args.get("cursor") or None
|
||||||
try:
|
try:
|
||||||
limit = int(request.args.get("limit", "50"))
|
limit = int(request.args.get("limit", "50"))
|
||||||
|
filters, sort = _parse_filters()
|
||||||
except ValueError:
|
except ValueError:
|
||||||
return jsonify({"error": "limit must be an integer"}), 400
|
return jsonify({"error": "invalid filter or limit parameter"}), 400
|
||||||
tag_id_raw = request.args.get("tag_id")
|
|
||||||
tag_id = int(tag_id_raw) if tag_id_raw else None
|
|
||||||
post_id_raw = request.args.get("post_id")
|
|
||||||
post_id = int(post_id_raw) if post_id_raw else None
|
|
||||||
artist_id_raw = request.args.get("artist_id")
|
|
||||||
artist_id = int(artist_id_raw) if artist_id_raw else None
|
|
||||||
|
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
svc = GalleryService(session)
|
svc = GalleryService(session)
|
||||||
try:
|
try:
|
||||||
page = await svc.scroll(
|
page = await svc.scroll(
|
||||||
cursor=cursor, limit=limit, tag_id=tag_id,
|
cursor=cursor, limit=limit, sort=sort, **filters,
|
||||||
post_id=post_id, artist_id=artist_id,
|
|
||||||
)
|
)
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
return jsonify({"error": str(exc)}), 400
|
return jsonify({"error": str(exc)}), 400
|
||||||
|
|
||||||
return jsonify(
|
return jsonify(
|
||||||
{
|
{
|
||||||
"images": [
|
"images": [_image_json(i) for i in page.images],
|
||||||
{
|
|
||||||
"id": i.id,
|
|
||||||
"sha256": i.sha256,
|
|
||||||
"mime": i.mime,
|
|
||||||
"width": i.width,
|
|
||||||
"height": i.height,
|
|
||||||
"created_at": i.created_at.isoformat(),
|
|
||||||
"posted_at": i.posted_at.isoformat() if i.posted_at else None,
|
|
||||||
"effective_date": i.effective_date.isoformat(),
|
|
||||||
"thumbnail_url": i.thumbnail_url,
|
|
||||||
"artist": i.artist,
|
|
||||||
}
|
|
||||||
for i in page.images
|
|
||||||
],
|
|
||||||
"next_cursor": page.next_cursor,
|
"next_cursor": page.next_cursor,
|
||||||
"date_groups": [
|
"date_groups": [
|
||||||
{"year": y, "month": m, "image_ids": ids} for y, m, ids in page.date_groups
|
{"year": y, "month": m, "image_ids": ids} for y, m, ids in page.date_groups
|
||||||
@@ -57,20 +100,46 @@ async def scroll():
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@gallery_bp.route("/timeline", methods=["GET"])
|
@gallery_bp.route("/similar", methods=["GET"])
|
||||||
async def timeline():
|
async def similar():
|
||||||
tag_id_raw = request.args.get("tag_id")
|
"""Visual "more like this": images ranked by cosine distance to the
|
||||||
tag_id = int(tag_id_raw) if tag_id_raw else None
|
`similar_to` image's embedding. Composes with the scope filters (AND) but
|
||||||
post_id_raw = request.args.get("post_id")
|
ignores post_id and sort. Bounded top-N, no cursor."""
|
||||||
post_id = int(post_id_raw) if post_id_raw else None
|
try:
|
||||||
artist_id_raw = request.args.get("artist_id")
|
similar_to = int(request.args["similar_to"])
|
||||||
artist_id = int(artist_id_raw) if artist_id_raw else None
|
limit = int(request.args.get("limit", "100"))
|
||||||
|
filters, _sort = _parse_filters()
|
||||||
|
except (KeyError, ValueError):
|
||||||
|
return jsonify({"error": "similar_to query param required"}), 400
|
||||||
|
# post_id is the exclusive post-detail view — not a similarity scope.
|
||||||
|
scope = {k: v for k, v in filters.items() if k != "post_id"}
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
svc = GalleryService(session)
|
svc = GalleryService(session)
|
||||||
try:
|
try:
|
||||||
buckets = await svc.timeline(
|
images = await svc.similar(image_id=similar_to, limit=limit, **scope)
|
||||||
tag_id=tag_id, post_id=post_id, artist_id=artist_id
|
except ValueError as exc:
|
||||||
)
|
return jsonify({"error": str(exc)}), 400
|
||||||
|
if images is None:
|
||||||
|
return jsonify({"error": "not found"}), 404
|
||||||
|
return jsonify(
|
||||||
|
{
|
||||||
|
"images": [_image_json(i) for i in images],
|
||||||
|
"next_cursor": None,
|
||||||
|
"date_groups": [],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@gallery_bp.route("/timeline", methods=["GET"])
|
||||||
|
async def timeline():
|
||||||
|
try:
|
||||||
|
filters, _sort = _parse_filters()
|
||||||
|
except ValueError:
|
||||||
|
return jsonify({"error": "invalid filter parameter"}), 400
|
||||||
|
async with get_session() as session:
|
||||||
|
svc = GalleryService(session)
|
||||||
|
try:
|
||||||
|
buckets = await svc.timeline(**filters)
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
return jsonify({"error": str(exc)}), 400
|
return jsonify({"error": str(exc)}), 400
|
||||||
return jsonify(
|
return jsonify(
|
||||||
@@ -78,25 +147,43 @@ async def timeline():
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@gallery_bp.route("/facets", methods=["GET"])
|
||||||
|
async def facets():
|
||||||
|
try:
|
||||||
|
filters, _sort = _parse_filters()
|
||||||
|
except ValueError:
|
||||||
|
return jsonify({"error": "invalid filter parameter"}), 400
|
||||||
|
async with get_session() as session:
|
||||||
|
svc = GalleryService(session)
|
||||||
|
try:
|
||||||
|
f = await svc.facets(**filters)
|
||||||
|
except ValueError as exc:
|
||||||
|
return jsonify({"error": str(exc)}), 400
|
||||||
|
return jsonify(
|
||||||
|
{
|
||||||
|
"total": f.total,
|
||||||
|
"platforms": f.platforms,
|
||||||
|
"untagged": f.untagged,
|
||||||
|
"no_artist": f.no_artist,
|
||||||
|
"date_min": f.date_min.isoformat() if f.date_min else None,
|
||||||
|
"date_max": f.date_max.isoformat() if f.date_max else None,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@gallery_bp.route("/jump", methods=["GET"])
|
@gallery_bp.route("/jump", methods=["GET"])
|
||||||
async def jump():
|
async def jump():
|
||||||
try:
|
try:
|
||||||
year = int(request.args["year"])
|
year = int(request.args["year"])
|
||||||
month = int(request.args["month"])
|
month = int(request.args["month"])
|
||||||
|
filters, sort = _parse_filters()
|
||||||
except (KeyError, ValueError):
|
except (KeyError, ValueError):
|
||||||
return jsonify({"error": "year and month query params required"}), 400
|
return jsonify({"error": "year and month query params required"}), 400
|
||||||
tag_id_raw = request.args.get("tag_id")
|
|
||||||
tag_id = int(tag_id_raw) if tag_id_raw else None
|
|
||||||
post_id_raw = request.args.get("post_id")
|
|
||||||
post_id = int(post_id_raw) if post_id_raw else None
|
|
||||||
artist_id_raw = request.args.get("artist_id")
|
|
||||||
artist_id = int(artist_id_raw) if artist_id_raw else None
|
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
svc = GalleryService(session)
|
svc = GalleryService(session)
|
||||||
try:
|
try:
|
||||||
cursor = await svc.jump_cursor(
|
cursor = await svc.jump_cursor(
|
||||||
year=year, month=month, tag_id=tag_id,
|
year=year, month=month, sort=sort, **filters,
|
||||||
post_id=post_id, artist_id=artist_id,
|
|
||||||
)
|
)
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
return jsonify({"error": str(exc)}), 400
|
return jsonify({"error": str(exc)}), 400
|
||||||
|
|||||||
@@ -35,10 +35,26 @@ async def trigger_scan():
|
|||||||
@import_admin_bp.route("/status", methods=["GET"])
|
@import_admin_bp.route("/status", methods=["GET"])
|
||||||
async def status():
|
async def status():
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
|
# Active batch = running batch that still has outstanding work.
|
||||||
|
# Plain "most recent running" picks freshly-created scans that
|
||||||
|
# enqueued zero new files and hides the older batch that's
|
||||||
|
# actually being processed. Mirrors the EXISTS predicate
|
||||||
|
# /api/system/stats already uses (api/settings.py:145-160).
|
||||||
|
# Audit 2026-06-02 — /api/import/status and /api/system/stats
|
||||||
|
# used to disagree on the active-batch predicate; the UI banner
|
||||||
|
# said "Scanning…" indefinitely while the stats card said idle.
|
||||||
active = (
|
active = (
|
||||||
await session.execute(
|
await session.execute(
|
||||||
select(ImportBatch)
|
select(ImportBatch)
|
||||||
.where(ImportBatch.status == "running")
|
.where(
|
||||||
|
ImportBatch.status == "running",
|
||||||
|
select(ImportTask.id)
|
||||||
|
.where(
|
||||||
|
ImportTask.batch_id == ImportBatch.id,
|
||||||
|
ImportTask.status.in_(["pending", "queued", "processing"]),
|
||||||
|
)
|
||||||
|
.exists(),
|
||||||
|
)
|
||||||
.order_by(ImportBatch.started_at.desc())
|
.order_by(ImportBatch.started_at.desc())
|
||||||
.limit(1)
|
.limit(1)
|
||||||
)
|
)
|
||||||
@@ -159,9 +175,7 @@ def _refetch_task_sync(session, task_id: int) -> dict:
|
|||||||
return {"status": "not_found"}
|
return {"status": "not_found"}
|
||||||
if task.status != "failed":
|
if task.status != "failed":
|
||||||
return {"status": "not_failed"}
|
return {"status": "not_failed"}
|
||||||
settings = session.execute(
|
settings = ImportSettings.load_sync(session)
|
||||||
select(ImportSettings).where(ImportSettings.id == 1)
|
|
||||||
).scalar_one()
|
|
||||||
return attempt_refetch(session, task, Path(settings.import_scan_path))
|
return attempt_refetch(session, task, Path(settings.import_scan_path))
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,112 +0,0 @@
|
|||||||
"""FC-5: /api/migrate — trigger and poll migration runs.
|
|
||||||
|
|
||||||
Ingest kinds (gs_ingest, ir_ingest) accept multipart/form-data with an
|
|
||||||
`export_file` field. All other kinds accept JSON. Backup + rollback
|
|
||||||
were retired in FC-3h (2026-05-24); use /api/system/backup/* instead.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import json
|
|
||||||
|
|
||||||
from quart import Blueprint, jsonify, request
|
|
||||||
from sqlalchemy import select
|
|
||||||
|
|
||||||
from ..extensions import get_session
|
|
||||||
from ..models import MigrationRun
|
|
||||||
from ..tasks.migration import run_migration
|
|
||||||
|
|
||||||
migrate_bp = Blueprint("migrate", __name__, url_prefix="/api/migrate")
|
|
||||||
|
|
||||||
# 'backup' + 'rollback' retired 2026-05-24 (FC-3h); see /api/system/backup/*.
|
|
||||||
_VALID_KINDS = frozenset({
|
|
||||||
"gs_ingest", "ir_ingest", "tag_apply",
|
|
||||||
"ml_queue", "verify", "cleanup",
|
|
||||||
})
|
|
||||||
_INGEST_KINDS = frozenset({"gs_ingest", "ir_ingest"})
|
|
||||||
|
|
||||||
|
|
||||||
def _bad(error: str, *, status: int = 400, **extra):
|
|
||||||
body = {"error": error}
|
|
||||||
body.update(extra)
|
|
||||||
return jsonify(body), status
|
|
||||||
|
|
||||||
|
|
||||||
def _run_to_dict(run: MigrationRun) -> dict:
|
|
||||||
return {
|
|
||||||
"id": run.id,
|
|
||||||
"kind": run.kind,
|
|
||||||
"status": run.status,
|
|
||||||
"dry_run": run.dry_run,
|
|
||||||
"started_at": run.started_at.isoformat(),
|
|
||||||
"finished_at": run.finished_at.isoformat() if run.finished_at else None,
|
|
||||||
"counts": run.counts or {},
|
|
||||||
"error": run.error,
|
|
||||||
"metadata": run.metadata_ or {},
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
@migrate_bp.route("/<kind>", methods=["POST"])
|
|
||||||
async def create_run(kind: str):
|
|
||||||
if kind not in _VALID_KINDS:
|
|
||||||
return _bad("unknown_kind", detail=f"kind must be one of {sorted(_VALID_KINDS)}")
|
|
||||||
|
|
||||||
# Ingest kinds accept multipart/form-data; everything else takes JSON.
|
|
||||||
if kind in _INGEST_KINDS:
|
|
||||||
form = await request.form
|
|
||||||
files = await request.files
|
|
||||||
if "export_file" not in files:
|
|
||||||
return _bad("missing_export_file", detail="multipart export_file required")
|
|
||||||
export_file = files["export_file"]
|
|
||||||
try:
|
|
||||||
raw = export_file.read()
|
|
||||||
data = json.loads(raw.decode("utf-8"))
|
|
||||||
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
|
|
||||||
return _bad("invalid_export_file", detail=str(exc))
|
|
||||||
dry_run = str(form.get("dry_run", "false")).lower() in ("true", "1", "yes")
|
|
||||||
params: dict = {"data": data, "dry_run": dry_run}
|
|
||||||
else:
|
|
||||||
body = await request.get_json()
|
|
||||||
if body is None:
|
|
||||||
body = {}
|
|
||||||
if not isinstance(body, dict):
|
|
||||||
return _bad("invalid_body")
|
|
||||||
dry_run = bool(body.get("dry_run", False))
|
|
||||||
params = dict(body)
|
|
||||||
|
|
||||||
async with get_session() as session:
|
|
||||||
run = MigrationRun(kind=kind, status="pending", dry_run=dry_run)
|
|
||||||
session.add(run)
|
|
||||||
await session.commit()
|
|
||||||
await session.refresh(run)
|
|
||||||
run_id = run.id
|
|
||||||
|
|
||||||
run_migration.delay(run_id, kind, params)
|
|
||||||
return jsonify({"run_id": run_id, "status": "pending"}), 202
|
|
||||||
|
|
||||||
|
|
||||||
@migrate_bp.route("/runs/<int:run_id>", methods=["GET"])
|
|
||||||
async def get_run(run_id: int):
|
|
||||||
async with get_session() as session:
|
|
||||||
run = (await session.execute(
|
|
||||||
select(MigrationRun).where(MigrationRun.id == run_id)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if run is None:
|
|
||||||
return _bad("not_found", status=404)
|
|
||||||
return jsonify(_run_to_dict(run))
|
|
||||||
|
|
||||||
|
|
||||||
@migrate_bp.route("/runs", methods=["GET"])
|
|
||||||
async def list_runs():
|
|
||||||
try:
|
|
||||||
limit = int(request.args.get("limit", "10"))
|
|
||||||
except ValueError:
|
|
||||||
return _bad("invalid_limit")
|
|
||||||
if limit < 1 or limit > 100:
|
|
||||||
return _bad("invalid_limit")
|
|
||||||
|
|
||||||
async with get_session() as session:
|
|
||||||
rows = (await session.execute(
|
|
||||||
select(MigrationRun)
|
|
||||||
.order_by(MigrationRun.id.desc())
|
|
||||||
.limit(limit)
|
|
||||||
)).scalars().all()
|
|
||||||
return jsonify([_run_to_dict(r) for r in rows])
|
|
||||||
@@ -9,12 +9,11 @@ ml_admin_bp = Blueprint("ml_admin", __name__, url_prefix="/api/ml")
|
|||||||
|
|
||||||
|
|
||||||
_EDITABLE = (
|
_EDITABLE = (
|
||||||
"suggestion_threshold_artist",
|
|
||||||
"suggestion_threshold_character",
|
"suggestion_threshold_character",
|
||||||
"suggestion_threshold_copyright",
|
|
||||||
"suggestion_threshold_general",
|
"suggestion_threshold_general",
|
||||||
"centroid_similarity_threshold",
|
"centroid_similarity_threshold",
|
||||||
"min_reference_images",
|
"min_reference_images",
|
||||||
|
"tagger_store_floor",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -28,12 +27,11 @@ async def get_settings():
|
|||||||
).scalar_one()
|
).scalar_one()
|
||||||
return jsonify(
|
return jsonify(
|
||||||
{
|
{
|
||||||
"suggestion_threshold_artist": s.suggestion_threshold_artist,
|
|
||||||
"suggestion_threshold_character": s.suggestion_threshold_character,
|
"suggestion_threshold_character": s.suggestion_threshold_character,
|
||||||
"suggestion_threshold_copyright": s.suggestion_threshold_copyright,
|
|
||||||
"suggestion_threshold_general": s.suggestion_threshold_general,
|
"suggestion_threshold_general": s.suggestion_threshold_general,
|
||||||
"centroid_similarity_threshold": s.centroid_similarity_threshold,
|
"centroid_similarity_threshold": s.centroid_similarity_threshold,
|
||||||
"min_reference_images": s.min_reference_images,
|
"min_reference_images": s.min_reference_images,
|
||||||
|
"tagger_store_floor": s.tagger_store_floor,
|
||||||
"tagger_model_version": s.tagger_model_version,
|
"tagger_model_version": s.tagger_model_version,
|
||||||
"embedder_model_version": s.embedder_model_version,
|
"embedder_model_version": s.embedder_model_version,
|
||||||
}
|
}
|
||||||
@@ -51,13 +49,45 @@ async def patch_settings():
|
|||||||
s = (
|
s = (
|
||||||
await session.execute(select(MLSettings).where(MLSettings.id == 1))
|
await session.execute(select(MLSettings).where(MLSettings.id == 1))
|
||||||
).scalar_one()
|
).scalar_one()
|
||||||
|
|
||||||
|
# Merge the patch over current values, then validate the result as a
|
||||||
|
# whole — the store-floor invariant couples three fields, so they
|
||||||
|
# can't be checked one at a time.
|
||||||
|
proposed = {f: getattr(s, f) for f in _EDITABLE}
|
||||||
for field in _EDITABLE:
|
for field in _EDITABLE:
|
||||||
if field in body:
|
if field in body:
|
||||||
setattr(s, field, body[field])
|
proposed[field] = body[field]
|
||||||
|
|
||||||
|
err = _validate(proposed)
|
||||||
|
if err is not None:
|
||||||
|
return jsonify({"error": err}), 400
|
||||||
|
|
||||||
|
for field in _EDITABLE:
|
||||||
|
setattr(s, field, proposed[field])
|
||||||
await session.commit()
|
await session.commit()
|
||||||
return await get_settings()
|
return await get_settings()
|
||||||
|
|
||||||
|
|
||||||
|
def _validate(p: dict) -> str | None:
|
||||||
|
"""Returns an error string if the proposed settings are invalid, else None.
|
||||||
|
|
||||||
|
Invariant (plan-task #764): the per-category suggestion thresholds can't
|
||||||
|
drop below tagger_store_floor — nothing below the floor is stored, so a
|
||||||
|
lower threshold would silently surface nothing in that gap. The UI clamps
|
||||||
|
the sliders to the floor; this is the server-side backstop.
|
||||||
|
"""
|
||||||
|
floor = p["tagger_store_floor"]
|
||||||
|
if not (0.0 <= floor <= 1.0):
|
||||||
|
return "tagger_store_floor must be between 0 and 1"
|
||||||
|
for cat in ("character", "general"):
|
||||||
|
if p[f"suggestion_threshold_{cat}"] < floor:
|
||||||
|
return (
|
||||||
|
f"suggestion_threshold_{cat} cannot be below tagger_store_floor "
|
||||||
|
f"({floor}) — predictions below the floor are not stored"
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
@ml_admin_bp.route("/backfill", methods=["POST"])
|
@ml_admin_bp.route("/backfill", methods=["POST"])
|
||||||
async def trigger_backfill():
|
async def trigger_backfill():
|
||||||
from ..tasks.ml import backfill
|
from ..tasks.ml import backfill
|
||||||
|
|||||||
+25
-10
@@ -5,18 +5,11 @@ from quart import Blueprint, jsonify, request
|
|||||||
from ..extensions import get_session
|
from ..extensions import get_session
|
||||||
from ..services.post_feed_service import PostFeedService
|
from ..services.post_feed_service import PostFeedService
|
||||||
from ..services.source_service import KNOWN_PLATFORMS
|
from ..services.source_service import KNOWN_PLATFORMS
|
||||||
|
from ._responses import error_response as _bad
|
||||||
|
|
||||||
posts_bp = Blueprint("posts", __name__, url_prefix="/api/posts")
|
posts_bp = Blueprint("posts", __name__, url_prefix="/api/posts")
|
||||||
|
|
||||||
|
|
||||||
def _bad(error: str, *, status: int = 400, detail: str | None = None, **extra):
|
|
||||||
body = {"error": error}
|
|
||||||
if detail is not None:
|
|
||||||
body["detail"] = detail
|
|
||||||
body.update(extra)
|
|
||||||
return jsonify(body), status
|
|
||||||
|
|
||||||
|
|
||||||
@posts_bp.route("", methods=["GET"])
|
@posts_bp.route("", methods=["GET"])
|
||||||
async def list_posts():
|
async def list_posts():
|
||||||
args = request.args
|
args = request.args
|
||||||
@@ -24,7 +17,10 @@ async def list_posts():
|
|||||||
cursor = args.get("cursor") or None
|
cursor = args.get("cursor") or None
|
||||||
artist_id_raw = args.get("artist_id")
|
artist_id_raw = args.get("artist_id")
|
||||||
platform = args.get("platform") or None
|
platform = args.get("platform") or None
|
||||||
|
q = (args.get("q") or "").strip() or None
|
||||||
limit_raw = args.get("limit", "24")
|
limit_raw = args.get("limit", "24")
|
||||||
|
direction = args.get("direction", "older")
|
||||||
|
around_raw = args.get("around")
|
||||||
|
|
||||||
try:
|
try:
|
||||||
limit = int(limit_raw)
|
limit = int(limit_raw)
|
||||||
@@ -33,6 +29,16 @@ async def list_posts():
|
|||||||
if limit < 1 or limit > 100:
|
if limit < 1 or limit > 100:
|
||||||
return _bad("invalid_limit", detail="limit must be between 1 and 100")
|
return _bad("invalid_limit", detail="limit must be between 1 and 100")
|
||||||
|
|
||||||
|
if direction not in ("older", "newer"):
|
||||||
|
return _bad("invalid_direction", detail="direction must be 'older' or 'newer'")
|
||||||
|
|
||||||
|
around_id = None
|
||||||
|
if around_raw is not None:
|
||||||
|
try:
|
||||||
|
around_id = int(around_raw)
|
||||||
|
except ValueError:
|
||||||
|
return _bad("invalid_around", detail="around must be an integer post id")
|
||||||
|
|
||||||
artist_id = None
|
artist_id = None
|
||||||
if artist_id_raw is not None:
|
if artist_id_raw is not None:
|
||||||
try:
|
try:
|
||||||
@@ -47,10 +53,19 @@ async def list_posts():
|
|||||||
)
|
)
|
||||||
|
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
|
svc = PostFeedService(session)
|
||||||
|
if around_id is not None:
|
||||||
|
result = await svc.around(
|
||||||
|
post_id=around_id, artist_id=artist_id,
|
||||||
|
platform=platform, q=q, limit=limit,
|
||||||
|
)
|
||||||
|
if result is None:
|
||||||
|
return _bad("not_found", status=404, detail=f"post id={around_id}")
|
||||||
|
return jsonify(result)
|
||||||
try:
|
try:
|
||||||
page = await PostFeedService(session).scroll(
|
page = await svc.scroll(
|
||||||
cursor=cursor, artist_id=artist_id,
|
cursor=cursor, artist_id=artist_id,
|
||||||
platform=platform, limit=limit,
|
platform=platform, q=q, limit=limit, direction=direction,
|
||||||
)
|
)
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
# Service raises ValueError for malformed cursors only;
|
# Service raises ValueError for malformed cursors only;
|
||||||
|
|||||||
@@ -25,15 +25,29 @@ _EDITABLE_FIELDS = (
|
|||||||
"download_schedule_default_seconds",
|
"download_schedule_default_seconds",
|
||||||
"download_event_retention_days",
|
"download_event_retention_days",
|
||||||
"download_failure_warning_threshold",
|
"download_failure_warning_threshold",
|
||||||
|
"series_suggest_enabled",
|
||||||
|
"series_suggest_threshold",
|
||||||
|
"extdl_mega_enabled",
|
||||||
|
"extdl_gdrive_enabled",
|
||||||
|
"extdl_mediafire_enabled",
|
||||||
|
"extdl_dropbox_enabled",
|
||||||
|
"extdl_pixeldrain_enabled",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Per-host external-download toggles — all plain booleans, validated uniformly.
|
||||||
|
_EXTDL_TOGGLE_FIELDS = (
|
||||||
|
"extdl_mega_enabled",
|
||||||
|
"extdl_gdrive_enabled",
|
||||||
|
"extdl_mediafire_enabled",
|
||||||
|
"extdl_dropbox_enabled",
|
||||||
|
"extdl_pixeldrain_enabled",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@settings_bp.route("/settings/import", methods=["GET"])
|
@settings_bp.route("/settings/import", methods=["GET"])
|
||||||
async def get_import_settings():
|
async def get_import_settings():
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
row = (
|
row = await ImportSettings.load(session)
|
||||||
await session.execute(select(ImportSettings).where(ImportSettings.id == 1))
|
|
||||||
).scalar_one()
|
|
||||||
return jsonify({
|
return jsonify({
|
||||||
"min_width": row.min_width,
|
"min_width": row.min_width,
|
||||||
"min_height": row.min_height,
|
"min_height": row.min_height,
|
||||||
@@ -48,6 +62,13 @@ async def get_import_settings():
|
|||||||
"download_schedule_default_seconds": row.download_schedule_default_seconds,
|
"download_schedule_default_seconds": row.download_schedule_default_seconds,
|
||||||
"download_event_retention_days": row.download_event_retention_days,
|
"download_event_retention_days": row.download_event_retention_days,
|
||||||
"download_failure_warning_threshold": row.download_failure_warning_threshold,
|
"download_failure_warning_threshold": row.download_failure_warning_threshold,
|
||||||
|
"series_suggest_enabled": row.series_suggest_enabled,
|
||||||
|
"series_suggest_threshold": row.series_suggest_threshold,
|
||||||
|
"extdl_mega_enabled": row.extdl_mega_enabled,
|
||||||
|
"extdl_gdrive_enabled": row.extdl_gdrive_enabled,
|
||||||
|
"extdl_mediafire_enabled": row.extdl_mediafire_enabled,
|
||||||
|
"extdl_dropbox_enabled": row.extdl_dropbox_enabled,
|
||||||
|
"extdl_pixeldrain_enabled": row.extdl_pixeldrain_enabled,
|
||||||
})
|
})
|
||||||
|
|
||||||
|
|
||||||
@@ -98,10 +119,24 @@ async def update_import_settings():
|
|||||||
if not isinstance(v, int) or isinstance(v, bool) or v < 1 or v > 100:
|
if not isinstance(v, int) or isinstance(v, bool) or v < 1 or v > 100:
|
||||||
return _bad_int("download_failure_warning_threshold", 1, 100)
|
return _bad_int("download_failure_warning_threshold", 1, 100)
|
||||||
|
|
||||||
|
if "series_suggest_enabled" in body and not isinstance(
|
||||||
|
body["series_suggest_enabled"], bool
|
||||||
|
):
|
||||||
|
return jsonify(
|
||||||
|
{"error": "series_suggest_enabled must be a boolean"}
|
||||||
|
), 400
|
||||||
|
for tog in _EXTDL_TOGGLE_FIELDS:
|
||||||
|
if tog in body and not isinstance(body[tog], bool):
|
||||||
|
return jsonify({"error": f"{tog} must be a boolean"}), 400
|
||||||
|
if "series_suggest_threshold" in body:
|
||||||
|
v = body["series_suggest_threshold"]
|
||||||
|
if not isinstance(v, (int, float)) or isinstance(v, bool) or v < 0 or v > 1:
|
||||||
|
return jsonify(
|
||||||
|
{"error": "series_suggest_threshold must be a number in [0, 1]"}
|
||||||
|
), 400
|
||||||
|
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
row = (
|
row = await ImportSettings.load(session)
|
||||||
await session.execute(select(ImportSettings).where(ImportSettings.id == 1))
|
|
||||||
).scalar_one()
|
|
||||||
for field in _EDITABLE_FIELDS:
|
for field in _EDITABLE_FIELDS:
|
||||||
if field in body:
|
if field in body:
|
||||||
setattr(row, field, body[field])
|
setattr(row, field, body[field])
|
||||||
|
|||||||
+165
-10
@@ -5,6 +5,7 @@ from sqlalchemy import select
|
|||||||
|
|
||||||
from ..extensions import get_session
|
from ..extensions import get_session
|
||||||
from ..models import DownloadEvent, Source
|
from ..models import DownloadEvent, Source
|
||||||
|
from ..services.scheduler_service import active_platform_cooldowns, scheduler_status
|
||||||
from ..services.source_service import (
|
from ..services.source_service import (
|
||||||
KNOWN_PLATFORMS,
|
KNOWN_PLATFORMS,
|
||||||
ArtistNotFoundError,
|
ArtistNotFoundError,
|
||||||
@@ -14,18 +15,11 @@ from ..services.source_service import (
|
|||||||
SourceService,
|
SourceService,
|
||||||
UnknownPlatformError,
|
UnknownPlatformError,
|
||||||
)
|
)
|
||||||
|
from ._responses import error_response as _bad
|
||||||
|
|
||||||
sources_bp = Blueprint("sources", __name__, url_prefix="/api/sources")
|
sources_bp = Blueprint("sources", __name__, url_prefix="/api/sources")
|
||||||
|
|
||||||
|
|
||||||
def _bad(error: str, *, status: int = 400, detail: str | None = None, **extra):
|
|
||||||
body = {"error": error}
|
|
||||||
if detail is not None:
|
|
||||||
body["detail"] = detail
|
|
||||||
body.update(extra)
|
|
||||||
return jsonify(body), status
|
|
||||||
|
|
||||||
|
|
||||||
@sources_bp.route("", methods=["GET"])
|
@sources_bp.route("", methods=["GET"])
|
||||||
async def list_sources():
|
async def list_sources():
|
||||||
artist_id_raw = request.args.get("artist_id")
|
artist_id_raw = request.args.get("artist_id")
|
||||||
@@ -35,11 +29,19 @@ async def list_sources():
|
|||||||
artist_id = int(artist_id_raw)
|
artist_id = int(artist_id_raw)
|
||||||
except ValueError:
|
except ValueError:
|
||||||
return _bad("invalid_artist_id", detail="artist_id must be an integer")
|
return _bad("invalid_artist_id", detail="artist_id must be an integer")
|
||||||
|
failing = request.args.get("failing", "").lower() in ("1", "true", "yes")
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
records = await SourceService(session).list(artist_id=artist_id)
|
records = await SourceService(session).list(artist_id=artist_id, failing=failing)
|
||||||
return jsonify([r.to_dict() for r in records])
|
return jsonify([r.to_dict() for r in records])
|
||||||
|
|
||||||
|
|
||||||
|
@sources_bp.route("/schedule-status", methods=["GET"])
|
||||||
|
async def schedule_status():
|
||||||
|
"""FC-dashboards: scheduler health for the Subscriptions hub."""
|
||||||
|
async with get_session() as session:
|
||||||
|
return jsonify(await scheduler_status(session))
|
||||||
|
|
||||||
|
|
||||||
@sources_bp.route("/<int:source_id>", methods=["GET"])
|
@sources_bp.route("/<int:source_id>", methods=["GET"])
|
||||||
async def get_source(source_id: int):
|
async def get_source(source_id: int):
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
@@ -83,6 +85,22 @@ async def create_source():
|
|||||||
return _bad("empty_url", detail=str(exc))
|
return _bad("empty_url", detail=str(exc))
|
||||||
except DuplicateSourceError as exc:
|
except DuplicateSourceError as exc:
|
||||||
return _bad("duplicate", status=409, existing_id=exc.existing_id)
|
return _bad("duplicate", status=409, existing_id=exc.existing_id)
|
||||||
|
|
||||||
|
# Immediate kickoff: a new enabled source is armed for backfill (#693)
|
||||||
|
# but would otherwise sit idle until the next scheduler tick (~60s).
|
||||||
|
# Enqueue the first walk now, skipping only if the platform is in a
|
||||||
|
# rate-limit cooldown (the scheduler picks it up when that clears).
|
||||||
|
dispatch_id = None
|
||||||
|
if record.enabled:
|
||||||
|
cooldowns = await active_platform_cooldowns(session)
|
||||||
|
if record.platform not in cooldowns:
|
||||||
|
session.add(DownloadEvent(source_id=record.id, status="pending"))
|
||||||
|
await session.commit()
|
||||||
|
dispatch_id = record.id
|
||||||
|
|
||||||
|
if dispatch_id is not None:
|
||||||
|
from ..tasks.download import download_source
|
||||||
|
download_source.delay(dispatch_id)
|
||||||
return jsonify(record.to_dict()), 201
|
return jsonify(record.to_dict()), 201
|
||||||
|
|
||||||
|
|
||||||
@@ -118,12 +136,136 @@ async def delete_source(source_id: int):
|
|||||||
return "", 204
|
return "", 204
|
||||||
|
|
||||||
|
|
||||||
|
@sources_bp.route("/<int:source_id>/backfill", methods=["POST"])
|
||||||
|
async def set_backfill(source_id: int):
|
||||||
|
"""Plan #693/#697: start/stop a run-until-done backfill, or start a recovery.
|
||||||
|
Body: `{"action": "start" | "stop" | "recover"}` (default "start"). 'start'
|
||||||
|
walks the full post history in time-boxed chunks until it reaches the bottom
|
||||||
|
(then the source shows 'complete'); 'recover' is the same walk but bypasses
|
||||||
|
the Patreon seen-ledger to re-fetch dropped-and-deleted near-dups under the
|
||||||
|
current pHash threshold; 'stop' cancels either back to tick mode. Returns the
|
||||||
|
updated source dict (incl. backfill_state / backfill_chunks /
|
||||||
|
backfill_bypass_seen)."""
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from ..services.credential_service import CredentialService
|
||||||
|
from ..services.download_backends import (
|
||||||
|
uses_native_ingester,
|
||||||
|
verify_source_credential,
|
||||||
|
)
|
||||||
|
from .credentials import _get_crypto
|
||||||
|
|
||||||
|
payload = await request.get_json(silent=True) or {}
|
||||||
|
action = payload.get("action", "start")
|
||||||
|
if action not in ("start", "stop", "recover"):
|
||||||
|
return _bad(
|
||||||
|
"invalid_action",
|
||||||
|
detail="action must be 'start', 'stop', or 'recover'",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Pre-flight (plan #703 #2): before arming a deep walk on a native-ingester
|
||||||
|
# platform (where verify is one cheap API page), refuse if the credential is
|
||||||
|
# DEFINITIVELY rejected — don't burn chunks against expired cookies. Proceed
|
||||||
|
# on valid OR inconclusive (a network blip shouldn't block). Gated to native
|
||||||
|
# platforms: gallery-dl verify is a slow --simulate subprocess, too heavy for
|
||||||
|
# an arm action. The credential read happens in a session that's CLOSED
|
||||||
|
# before the verify network call (don't hold a DB conn across the request).
|
||||||
|
if action in ("start", "recover"):
|
||||||
|
async with get_session() as session:
|
||||||
|
rec = await SourceService(session).get(source_id)
|
||||||
|
if rec is None:
|
||||||
|
return _bad("not_found", status=404)
|
||||||
|
native = uses_native_ingester(rec.platform)
|
||||||
|
if native:
|
||||||
|
cred = CredentialService(session, _get_crypto())
|
||||||
|
cookies_path = await cred.get_cookies_path(rec.platform)
|
||||||
|
auth_token = await cred.get_token(rec.platform)
|
||||||
|
if native:
|
||||||
|
ok, message = await verify_source_credential(
|
||||||
|
platform=rec.platform,
|
||||||
|
url=rec.url,
|
||||||
|
artist_slug=rec.artist_slug,
|
||||||
|
config_overrides=rec.config_overrides or {},
|
||||||
|
cookies_path=str(cookies_path) if cookies_path else None,
|
||||||
|
auth_token=auth_token,
|
||||||
|
images_root=Path("/images"),
|
||||||
|
)
|
||||||
|
if ok is False:
|
||||||
|
return _bad("credential_rejected", detail=message, status=409)
|
||||||
|
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
svc = SourceService(session)
|
||||||
|
if action == "start":
|
||||||
|
record = await svc.start_backfill(source_id)
|
||||||
|
elif action == "recover":
|
||||||
|
record = await svc.start_recovery(source_id)
|
||||||
|
else:
|
||||||
|
record = await svc.stop_backfill(source_id)
|
||||||
|
except LookupError:
|
||||||
|
return _bad("not_found", status=404)
|
||||||
|
return jsonify(record.to_dict())
|
||||||
|
|
||||||
|
|
||||||
|
@sources_bp.route("/<int:source_id>/preview", methods=["POST"])
|
||||||
|
async def preview_source_endpoint(source_id: int):
|
||||||
|
"""Plan #708 B4: dry-run — count what a backfill WOULD download for a native
|
||||||
|
platform (Patreon today), without downloading. Walks the first few feed pages
|
||||||
|
and counts media not already in the seen/dead ledgers. Returns
|
||||||
|
{total_new, posts_scanned, pages_scanned, has_more, sample[]} or 409 + reason
|
||||||
|
(unresolvable campaign id / auth / drift). 400 for gallery-dl platforms (no
|
||||||
|
cheap dry-run — their verify is a slow --simulate)."""
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from ..services.credential_service import CredentialService
|
||||||
|
from ..services.download_backends import preview_source, uses_native_ingester
|
||||||
|
from ..tasks._sync_engine import sync_session_factory
|
||||||
|
from .credentials import _get_crypto
|
||||||
|
|
||||||
|
async with get_session() as session:
|
||||||
|
rec = await SourceService(session).get(source_id)
|
||||||
|
if rec is None:
|
||||||
|
return _bad("not_found", status=404)
|
||||||
|
if not uses_native_ingester(rec.platform):
|
||||||
|
return _bad(
|
||||||
|
"unsupported",
|
||||||
|
detail="Preview is only available for native-ingester platforms.",
|
||||||
|
status=400,
|
||||||
|
)
|
||||||
|
cred = CredentialService(session, _get_crypto())
|
||||||
|
cookies_path = await cred.get_cookies_path(rec.platform)
|
||||||
|
|
||||||
|
# The walk + ledger reads are sync (run off the request loop); the process
|
||||||
|
# sync engine is the same one the download task uses.
|
||||||
|
result = await preview_source(
|
||||||
|
platform=rec.platform,
|
||||||
|
url=rec.url,
|
||||||
|
source_id=source_id,
|
||||||
|
config_overrides=rec.config_overrides or {},
|
||||||
|
cookies_path=str(cookies_path) if cookies_path else None,
|
||||||
|
images_root=Path("/images"),
|
||||||
|
sync_session_factory=sync_session_factory(),
|
||||||
|
)
|
||||||
|
if "error" in result:
|
||||||
|
return _bad("preview_failed", detail=result["error"], status=409)
|
||||||
|
return jsonify(result)
|
||||||
|
|
||||||
|
|
||||||
@sources_bp.route("/<int:source_id>/check", methods=["POST"])
|
@sources_bp.route("/<int:source_id>/check", methods=["POST"])
|
||||||
async def check_source(source_id: int):
|
async def check_source(source_id: int):
|
||||||
"""FC-3c: enqueue a download for this source.
|
"""FC-3c: enqueue a download for this source.
|
||||||
|
|
||||||
Returns 202 with the new DownloadEvent id. If a pending/running
|
Returns 202 with the new DownloadEvent id. If a pending/running
|
||||||
event already exists for this source, returns 409 with that id."""
|
event already exists for this source, returns 409 with that id. If
|
||||||
|
the source's platform is currently in a rate-limit cooldown, returns
|
||||||
|
**202 with `{status: "deferred", cooldown_until, platform}`** and
|
||||||
|
does NOT create an event or dispatch — the bulk retry path uses this
|
||||||
|
to avoid bowling N sources right back into the rate limit the
|
||||||
|
cooldown is preventing. Single-click "retry this one source" passes
|
||||||
|
`?force=true` to override the cooldown (operator-explicit, useful
|
||||||
|
for rapid auth-fix testing). The in-flight guard always applies.
|
||||||
|
"""
|
||||||
|
force = (request.args.get("force") or "").lower() in ("1", "true", "yes")
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
source = (await session.execute(
|
source = (await session.execute(
|
||||||
select(Source).where(Source.id == source_id)
|
select(Source).where(Source.id == source_id)
|
||||||
@@ -133,6 +275,19 @@ async def check_source(source_id: int):
|
|||||||
if not source.enabled:
|
if not source.enabled:
|
||||||
return _bad("source_disabled", detail="enable the source first")
|
return _bad("source_disabled", detail="enable the source first")
|
||||||
|
|
||||||
|
# Cooldown gate (unless explicitly overridden). Checked before the
|
||||||
|
# in-flight guard because a deferred retry doesn't need to create
|
||||||
|
# or check for an event at all.
|
||||||
|
if not force:
|
||||||
|
cooldowns = await active_platform_cooldowns(session)
|
||||||
|
expires_at = cooldowns.get(source.platform)
|
||||||
|
if expires_at is not None:
|
||||||
|
return jsonify({
|
||||||
|
"status": "deferred",
|
||||||
|
"platform": source.platform,
|
||||||
|
"cooldown_until": expires_at.isoformat(),
|
||||||
|
}), 202
|
||||||
|
|
||||||
in_flight = (await session.execute(
|
in_flight = (await session.execute(
|
||||||
select(DownloadEvent.id).where(
|
select(DownloadEvent.id).where(
|
||||||
DownloadEvent.source_id == source_id,
|
DownloadEvent.source_id == source_id,
|
||||||
|
|||||||
@@ -11,8 +11,21 @@ suggestions_bp = Blueprint("suggestions", __name__, url_prefix="/api")
|
|||||||
|
|
||||||
@suggestions_bp.route("/images/<int:image_id>/suggestions", methods=["GET"])
|
@suggestions_bp.route("/images/<int:image_id>/suggestions", methods=["GET"])
|
||||||
async def get_suggestions(image_id: int):
|
async def get_suggestions(image_id: int):
|
||||||
|
# ?min=<float> overrides the configured per-category thresholds so the typed
|
||||||
|
# tag-input dropdown can surface EVERY stored prediction (min=0), including
|
||||||
|
# low-confidence actions/features, in canonical formatting. Omitted → the
|
||||||
|
# curated above-threshold list the Suggestions panel uses.
|
||||||
|
override = None
|
||||||
|
raw_min = request.args.get("min")
|
||||||
|
if raw_min is not None:
|
||||||
|
try:
|
||||||
|
override = min(1.0, max(0.0, float(raw_min)))
|
||||||
|
except ValueError:
|
||||||
|
return jsonify({"error": "min must be a float in [0,1]"}), 400
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
sl = await SuggestionService(session).for_image(image_id)
|
sl = await SuggestionService(session).for_image(
|
||||||
|
image_id, threshold_override=override
|
||||||
|
)
|
||||||
return jsonify(
|
return jsonify(
|
||||||
{
|
{
|
||||||
"by_category": {
|
"by_category": {
|
||||||
@@ -24,6 +37,11 @@ async def get_suggestions(image_id: int):
|
|||||||
"score": round(s.score, 4),
|
"score": round(s.score, 4),
|
||||||
"source": s.source,
|
"source": s.source,
|
||||||
"creates_new_tag": s.creates_new_tag,
|
"creates_new_tag": s.creates_new_tag,
|
||||||
|
# raw model key (alias is stored under this) + whether an
|
||||||
|
# operator alias produced this suggestion — drive the
|
||||||
|
# modal's "Treat as alias"/"Remove alias" affordances.
|
||||||
|
"raw_name": s.raw_name,
|
||||||
|
"via_alias": s.via_alias,
|
||||||
}
|
}
|
||||||
for s in items
|
for s in items
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ from sqlalchemy import desc, func, select
|
|||||||
from ..config import get_config
|
from ..config import get_config
|
||||||
from ..extensions import get_session
|
from ..extensions import get_session
|
||||||
from ..models import TaskRun
|
from ..models import TaskRun
|
||||||
|
from ..services.scheduler_service import scheduler_status
|
||||||
|
|
||||||
system_activity_bp = Blueprint(
|
system_activity_bp = Blueprint(
|
||||||
"system_activity", __name__, url_prefix="/api/system/activity",
|
"system_activity", __name__, url_prefix="/api/system/activity",
|
||||||
@@ -30,7 +31,7 @@ system_activity_bp = Blueprint(
|
|||||||
# absent.
|
# absent.
|
||||||
_QUEUE_NAMES = (
|
_QUEUE_NAMES = (
|
||||||
"default", "import", "thumbnail", "ml",
|
"default", "import", "thumbnail", "ml",
|
||||||
"download", "scan", "maintenance",
|
"download", "scan", "maintenance", "maintenance_long",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Cache module-level so all requests share the cache between polls.
|
# Cache module-level so all requests share the cache between polls.
|
||||||
@@ -81,17 +82,22 @@ def _read_workers_sync() -> dict:
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def _queues_cached() -> dict:
|
||||||
|
"""Per-queue Redis LLEN, cached 2s. Shared by /queues and /summary."""
|
||||||
|
now = time.time()
|
||||||
|
if _QUEUE_CACHE["data"] is None or (now - _QUEUE_CACHE["ts"]) > _QUEUE_CACHE_TTL:
|
||||||
|
_QUEUE_CACHE["data"] = await asyncio.to_thread(_read_queues_sync)
|
||||||
|
_QUEUE_CACHE["ts"] = now
|
||||||
|
return _QUEUE_CACHE["data"]
|
||||||
|
|
||||||
|
|
||||||
@system_activity_bp.route("/queues", methods=["GET"])
|
@system_activity_bp.route("/queues", methods=["GET"])
|
||||||
async def get_queues():
|
async def get_queues():
|
||||||
"""Per-queue Redis LLEN. Cached 2s.
|
"""Per-queue Redis LLEN. Cached 2s.
|
||||||
|
|
||||||
Response: {queues: {name: depth_or_null}, fetched_at: iso8601}
|
Response: {queues: {name: depth_or_null}, fetched_at: iso8601}
|
||||||
"""
|
"""
|
||||||
now = time.time()
|
return jsonify(await _queues_cached())
|
||||||
if _QUEUE_CACHE["data"] is None or (now - _QUEUE_CACHE["ts"]) > _QUEUE_CACHE_TTL:
|
|
||||||
_QUEUE_CACHE["data"] = await asyncio.to_thread(_read_queues_sync)
|
|
||||||
_QUEUE_CACHE["ts"] = now
|
|
||||||
return jsonify(_QUEUE_CACHE["data"])
|
|
||||||
|
|
||||||
|
|
||||||
@system_activity_bp.route("/workers", methods=["GET"])
|
@system_activity_bp.route("/workers", methods=["GET"])
|
||||||
@@ -107,11 +113,41 @@ async def get_workers():
|
|||||||
return jsonify(_WORKER_CACHE["data"])
|
return jsonify(_WORKER_CACHE["data"])
|
||||||
|
|
||||||
|
|
||||||
|
@system_activity_bp.route("/summary", methods=["GET"])
|
||||||
|
async def get_summary():
|
||||||
|
"""One-call rollup for the always-on TopNav pipeline indicator:
|
||||||
|
scheduler health, per-queue pending depths, currently-running count, and
|
||||||
|
recent (24h) failure count. Cheap — cached queue LLENs + two TaskRun
|
||||||
|
counts — so it's safe to poll app-wide."""
|
||||||
|
queues_data = await _queues_cached()
|
||||||
|
depths = queues_data.get("queues", {})
|
||||||
|
queued_total = sum(v for v in depths.values() if isinstance(v, int))
|
||||||
|
since = datetime.now(UTC) - timedelta(hours=24)
|
||||||
|
async with get_session() as session:
|
||||||
|
scheduler = await scheduler_status(session)
|
||||||
|
running = (await session.execute(
|
||||||
|
select(func.count(TaskRun.id)).where(TaskRun.status == "running")
|
||||||
|
)).scalar_one()
|
||||||
|
failing = (await session.execute(
|
||||||
|
select(func.count(TaskRun.id))
|
||||||
|
.where(TaskRun.status.in_(["error", "timeout"]))
|
||||||
|
.where(TaskRun.finished_at >= since)
|
||||||
|
)).scalar_one()
|
||||||
|
return jsonify({
|
||||||
|
"scheduler": scheduler,
|
||||||
|
"queues": depths,
|
||||||
|
"queued_total": queued_total,
|
||||||
|
"running": int(running),
|
||||||
|
"failing": int(failing),
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
@system_activity_bp.route("/runs", methods=["GET"])
|
@system_activity_bp.route("/runs", methods=["GET"])
|
||||||
async def list_runs():
|
async def list_runs():
|
||||||
"""Paginated task_run history. Query params:
|
"""Paginated task_run history. Query params:
|
||||||
queue=<name> filter to one queue
|
queue=<name> filter to one queue
|
||||||
status=<status> filter to one status (running/ok/error/timeout/retry)
|
status=<status> filter to one status (running/ok/error/timeout/retry)
|
||||||
|
task=<substr> case-insensitive substring match on task_name
|
||||||
limit=<int> default 50, max 200
|
limit=<int> default 50, max 200
|
||||||
before_id=<int> cursor for keyset pagination
|
before_id=<int> cursor for keyset pagination
|
||||||
|
|
||||||
@@ -126,6 +162,7 @@ async def list_runs():
|
|||||||
|
|
||||||
queue = request.args.get("queue")
|
queue = request.args.get("queue")
|
||||||
status = request.args.get("status")
|
status = request.args.get("status")
|
||||||
|
task = request.args.get("task")
|
||||||
before_id_raw = request.args.get("before_id")
|
before_id_raw = request.args.get("before_id")
|
||||||
before_id = int(before_id_raw) if before_id_raw else None
|
before_id = int(before_id_raw) if before_id_raw else None
|
||||||
|
|
||||||
@@ -135,6 +172,11 @@ async def list_runs():
|
|||||||
stmt = stmt.where(TaskRun.queue == queue)
|
stmt = stmt.where(TaskRun.queue == queue)
|
||||||
if status:
|
if status:
|
||||||
stmt = stmt.where(TaskRun.status == status)
|
stmt = stmt.where(TaskRun.status == status)
|
||||||
|
if task:
|
||||||
|
# Task names contain literal underscores (download_source,
|
||||||
|
# vacuum_analyze) — escape LIKE wildcards so a search for
|
||||||
|
# "vacuum_analyze" doesn't treat "_" as a single-char match.
|
||||||
|
stmt = stmt.where(TaskRun.task_name.ilike(f"%{_escape_like(task)}%", escape="\\"))
|
||||||
if before_id is not None:
|
if before_id is not None:
|
||||||
stmt = stmt.where(TaskRun.id < before_id)
|
stmt = stmt.where(TaskRun.id < before_id)
|
||||||
stmt = stmt.limit(limit + 1)
|
stmt = stmt.limit(limit + 1)
|
||||||
@@ -190,6 +232,12 @@ async def list_failures():
|
|||||||
})
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def _escape_like(value: str) -> str:
|
||||||
|
"""Escape SQL LIKE/ILIKE metacharacters so user search text is matched
|
||||||
|
literally. Pairs with `escape="\\"` on the .ilike() call."""
|
||||||
|
return value.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
|
||||||
|
|
||||||
|
|
||||||
def _row_to_dict(r: TaskRun) -> dict:
|
def _row_to_dict(r: TaskRun) -> dict:
|
||||||
return {
|
return {
|
||||||
"id": r.id,
|
"id": r.id,
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ from sqlalchemy import desc, select
|
|||||||
|
|
||||||
from ..extensions import get_session
|
from ..extensions import get_session
|
||||||
from ..models import BackupRun, ImportSettings
|
from ..models import BackupRun, ImportSettings
|
||||||
|
from ._responses import error_response as _bad
|
||||||
|
|
||||||
system_backup_bp = Blueprint(
|
system_backup_bp = Blueprint(
|
||||||
"system_backup", __name__, url_prefix="/api/system/backup",
|
"system_backup", __name__, url_prefix="/api/system/backup",
|
||||||
@@ -29,12 +30,6 @@ _BACKUP_SETTINGS_FIELDS = (
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _bad(error: str, *, status: int = 400, **extra):
|
|
||||||
body = {"error": error}
|
|
||||||
body.update(extra)
|
|
||||||
return jsonify(body), status
|
|
||||||
|
|
||||||
|
|
||||||
def _row_to_dict(r: BackupRun) -> dict:
|
def _row_to_dict(r: BackupRun) -> dict:
|
||||||
return {
|
return {
|
||||||
"id": r.id,
|
"id": r.id,
|
||||||
@@ -232,9 +227,7 @@ async def delete_run(run_id: int):
|
|||||||
@system_backup_bp.route("/settings", methods=["GET"])
|
@system_backup_bp.route("/settings", methods=["GET"])
|
||||||
async def get_settings():
|
async def get_settings():
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
row = (await session.execute(
|
row = await ImportSettings.load(session)
|
||||||
select(ImportSettings).where(ImportSettings.id == 1)
|
|
||||||
)).scalar_one()
|
|
||||||
return jsonify({
|
return jsonify({
|
||||||
"backup_db_nightly_enabled": row.backup_db_nightly_enabled,
|
"backup_db_nightly_enabled": row.backup_db_nightly_enabled,
|
||||||
"backup_db_nightly_hour_utc": row.backup_db_nightly_hour_utc,
|
"backup_db_nightly_hour_utc": row.backup_db_nightly_hour_utc,
|
||||||
@@ -254,9 +247,7 @@ async def patch_settings():
|
|||||||
return err
|
return err
|
||||||
|
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
row = (await session.execute(
|
row = await ImportSettings.load(session)
|
||||||
select(ImportSettings).where(ImportSettings.id == 1)
|
|
||||||
)).scalar_one()
|
|
||||||
for field in _BACKUP_SETTINGS_FIELDS:
|
for field in _BACKUP_SETTINGS_FIELDS:
|
||||||
if field in body:
|
if field in body:
|
||||||
setattr(row, field, body[field])
|
setattr(row, field, body[field])
|
||||||
|
|||||||
+319
-35
@@ -8,12 +8,16 @@ from ..extensions import get_session
|
|||||||
from ..models import Tag, TagKind
|
from ..models import Tag, TagKind
|
||||||
from ..models.tag_allowlist import TagAllowlist
|
from ..models.tag_allowlist import TagAllowlist
|
||||||
from ..services.bulk_tag_service import BulkTagService
|
from ..services.bulk_tag_service import BulkTagService
|
||||||
|
from ..services.ml.aliases import AliasService
|
||||||
|
from ..services.series_match_service import SeriesMatchService
|
||||||
from ..services.series_service import SeriesError, SeriesService
|
from ..services.series_service import SeriesError, SeriesService
|
||||||
from ..services.tag_directory_service import TagDirectoryService
|
from ..services.tag_directory_service import TagDirectoryService
|
||||||
|
from ..services.tag_query import serialize_tag
|
||||||
from ..services.tag_service import (
|
from ..services.tag_service import (
|
||||||
TagMergeConflict,
|
TagMergeConflict,
|
||||||
TagService,
|
TagService,
|
||||||
TagValidationError,
|
TagValidationError,
|
||||||
|
normalize_tag_name,
|
||||||
)
|
)
|
||||||
from ..utils.tag_prefix import parse_kind_prefix
|
from ..utils.tag_prefix import parse_kind_prefix
|
||||||
|
|
||||||
@@ -70,17 +74,7 @@ async def autocomplete():
|
|||||||
hits = await svc.autocomplete(q, kind=kind, limit=limit)
|
hits = await svc.autocomplete(q, kind=kind, limit=limit)
|
||||||
|
|
||||||
return jsonify(
|
return jsonify(
|
||||||
[
|
[{**serialize_tag(h), "image_count": h.image_count} for h in hits]
|
||||||
{
|
|
||||||
"id": h.id,
|
|
||||||
"name": h.name,
|
|
||||||
"kind": h.kind,
|
|
||||||
"fandom_id": h.fandom_id,
|
|
||||||
"fandom_name": h.fandom_name,
|
|
||||||
"image_count": h.image_count,
|
|
||||||
}
|
|
||||||
for h in hits
|
|
||||||
]
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -141,6 +135,11 @@ async def create_tag():
|
|||||||
|
|
||||||
fandom_id = body.get("fandom_id")
|
fandom_id = body.get("fandom_id")
|
||||||
|
|
||||||
|
# #701: Title-Case operator-entered tags. Only here (the explicit create
|
||||||
|
# endpoint), NOT in the shared find_or_create — the ML tagger uses that path
|
||||||
|
# and must keep the booru vocabulary's casing for allowlist matching.
|
||||||
|
name = normalize_tag_name(name)
|
||||||
|
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
svc = TagService(session)
|
svc = TagService(session)
|
||||||
try:
|
try:
|
||||||
@@ -158,17 +157,7 @@ async def list_tags_for_image(image_id: int):
|
|||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
svc = TagService(session)
|
svc = TagService(session)
|
||||||
tags = await svc.list_for_image(image_id)
|
tags = await svc.list_for_image(image_id)
|
||||||
return jsonify(
|
return jsonify([serialize_tag(t) for t in tags])
|
||||||
[
|
|
||||||
{
|
|
||||||
"id": t.id,
|
|
||||||
"name": t.name,
|
|
||||||
"kind": t.kind.value,
|
|
||||||
"fandom_id": t.fandom_id,
|
|
||||||
}
|
|
||||||
for t in tags
|
|
||||||
]
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@tags_bp.route("/images/<int:image_id>/tags", methods=["POST"])
|
@tags_bp.route("/images/<int:image_id>/tags", methods=["POST"])
|
||||||
@@ -194,15 +183,65 @@ async def remove_tag_from_image(image_id: int, tag_id: int):
|
|||||||
return "", 204
|
return "", 204
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/tags/<int:tag_id>", methods=["GET"])
|
||||||
|
async def get_tag(tag_id: int):
|
||||||
|
"""Resolve a single tag (used by the gallery to label its active
|
||||||
|
tag-filter chip)."""
|
||||||
|
async with get_session() as session:
|
||||||
|
tag = await session.get(Tag, tag_id)
|
||||||
|
if tag is None:
|
||||||
|
return jsonify({"error": "tag not found"}), 404
|
||||||
|
return jsonify(
|
||||||
|
{
|
||||||
|
"id": tag.id,
|
||||||
|
"name": tag.name,
|
||||||
|
"kind": tag.kind.value,
|
||||||
|
"fandom_id": tag.fandom_id,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/tags/<int:tag_id>/aliases", methods=["GET"])
|
||||||
|
async def list_tag_aliases(tag_id: int):
|
||||||
|
"""Model keys that fold into this tag (tag-side alias view). Remove via the
|
||||||
|
shared DELETE /api/aliases/<string>/<category>."""
|
||||||
|
async with get_session() as session:
|
||||||
|
if await session.get(Tag, tag_id) is None:
|
||||||
|
return jsonify({"error": "tag not found"}), 404
|
||||||
|
rows = await AliasService(session).list_for_tag(tag_id)
|
||||||
|
return jsonify(
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"alias_string": r.alias_string,
|
||||||
|
"alias_category": r.alias_category,
|
||||||
|
}
|
||||||
|
for r in rows
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@tags_bp.route("/tags/<int:tag_id>", methods=["PATCH"])
|
@tags_bp.route("/tags/<int:tag_id>", methods=["PATCH"])
|
||||||
async def rename_tag(tag_id: int):
|
async def update_tag(tag_id: int):
|
||||||
body = await request.get_json()
|
"""Rename and/or re-fandom a tag. Body may carry `name` and/or
|
||||||
if not body or "name" not in body:
|
`fandom_id` (a fandom tag id, or null to clear — character tags only).
|
||||||
return jsonify({"error": "name required"}), 400
|
`merge: true` resolves a collision by merging into the existing tag.
|
||||||
|
"""
|
||||||
|
body = await request.get_json() or {}
|
||||||
|
has_name = "name" in body
|
||||||
|
has_fandom = "fandom_id" in body
|
||||||
|
if not has_name and not has_fandom:
|
||||||
|
return jsonify({"error": "name or fandom_id required"}), 400
|
||||||
|
do_merge = bool(body.get("merge"))
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
svc = TagService(session)
|
svc = TagService(session)
|
||||||
try:
|
try:
|
||||||
tag = await svc.rename(tag_id, body["name"])
|
tag = None
|
||||||
|
if has_name:
|
||||||
|
tag = await svc.rename(tag_id, body["name"])
|
||||||
|
if has_fandom:
|
||||||
|
tag = await svc.set_fandom(
|
||||||
|
tag_id, body["fandom_id"], merge=do_merge
|
||||||
|
)
|
||||||
except TagMergeConflict as exc:
|
except TagMergeConflict as exc:
|
||||||
return jsonify(
|
return jsonify(
|
||||||
{
|
{
|
||||||
@@ -219,7 +258,12 @@ async def rename_tag(tag_id: int):
|
|||||||
return jsonify({"error": str(exc)}), 400
|
return jsonify({"error": str(exc)}), 400
|
||||||
await session.commit()
|
await session.commit()
|
||||||
return jsonify(
|
return jsonify(
|
||||||
{"id": tag.id, "name": tag.name, "kind": tag.kind.value}
|
{
|
||||||
|
"id": tag.id,
|
||||||
|
"name": tag.name,
|
||||||
|
"kind": tag.kind.value,
|
||||||
|
"fandom_id": tag.fandom_id,
|
||||||
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -245,6 +289,12 @@ async def merge_tag(source_id: int):
|
|||||||
from ..tasks.ml import apply_allowlist_tags
|
from ..tasks.ml import apply_allowlist_tags
|
||||||
|
|
||||||
apply_allowlist_tags.delay(tag_id=result.target_id)
|
apply_allowlist_tags.delay(tag_id=result.target_id)
|
||||||
|
# Tag merge invalidates the target's centroid (the merged-in source
|
||||||
|
# tag's images now contribute to it). Daily list_drifted catches it
|
||||||
|
# within 24h, but eager recompute closes the suggestion-quality dip
|
||||||
|
# in the meantime. Audit 2026-06-02.
|
||||||
|
from ..tasks.ml import recompute_centroid
|
||||||
|
recompute_centroid.delay(result.target_id)
|
||||||
return jsonify(
|
return jsonify(
|
||||||
{
|
{
|
||||||
"target": {
|
"target": {
|
||||||
@@ -320,6 +370,31 @@ def _series_err(exc: SeriesError):
|
|||||||
return jsonify({"error": msg}), status
|
return jsonify({"error": msg}), status
|
||||||
|
|
||||||
|
|
||||||
|
def _opt_int(body, key: str):
|
||||||
|
"""(value, error) — value is None when absent, error is (json, status)."""
|
||||||
|
if not body or body.get(key) is None:
|
||||||
|
return None, None
|
||||||
|
try:
|
||||||
|
return int(body[key]), None
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None, (jsonify({"error": f"{key} must be an integer"}), 400)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_int_list(body, key: str, *, max_ids: int = 500):
|
||||||
|
"""(list, error) for a required list of ints under `key`."""
|
||||||
|
if not body or key not in body:
|
||||||
|
return None, (jsonify({"error": f"{key} required"}), 400)
|
||||||
|
raw = body[key]
|
||||||
|
if not isinstance(raw, list) or not raw:
|
||||||
|
return None, (jsonify({"error": f"{key} must be a non-empty list"}), 400)
|
||||||
|
if len(raw) > max_ids:
|
||||||
|
return None, (jsonify({"error": f"too many ids (max {max_ids})"}), 400)
|
||||||
|
try:
|
||||||
|
return [int(x) for x in raw], None
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None, (jsonify({"error": f"{key} must be integers"}), 400)
|
||||||
|
|
||||||
|
|
||||||
@tags_bp.route("/series/<int:tag_id>/pages", methods=["GET"])
|
@tags_bp.route("/series/<int:tag_id>/pages", methods=["GET"])
|
||||||
async def series_pages(tag_id: int):
|
async def series_pages(tag_id: int):
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
@@ -360,15 +435,26 @@ async def series_remove(tag_id: int):
|
|||||||
return jsonify({"removed_count": n})
|
return jsonify({"removed_count": n})
|
||||||
|
|
||||||
|
|
||||||
@tags_bp.route("/series/<int:tag_id>/reorder", methods=["POST"])
|
@tags_bp.route("/series/<int:tag_id>/pages/number", methods=["POST"])
|
||||||
async def series_reorder(tag_id: int):
|
async def series_set_page_number(tag_id: int):
|
||||||
body = await request.get_json()
|
"""Set one placed page's number — the operator's value (sparse, gaps
|
||||||
ids, err = _parse_bulk_ids(body, max_ids=500)
|
allowed); pass page_number: null to leave it unnumbered."""
|
||||||
if err:
|
body = await request.get_json() or {}
|
||||||
return err
|
image_id, ierr = _opt_int(body, "image_id")
|
||||||
|
if ierr:
|
||||||
|
return ierr
|
||||||
|
if image_id is None:
|
||||||
|
return jsonify({"error": "image_id required"}), 400
|
||||||
|
if "page_number" not in body:
|
||||||
|
return jsonify({"error": "page_number required (may be null)"}), 400
|
||||||
|
page_number, perr = _opt_int(body, "page_number")
|
||||||
|
if perr:
|
||||||
|
return perr
|
||||||
async with get_session() as session:
|
async with get_session() as session:
|
||||||
try:
|
try:
|
||||||
await SeriesService(session).reorder(tag_id, ids)
|
await SeriesService(session).set_page_number(
|
||||||
|
tag_id, image_id, page_number
|
||||||
|
)
|
||||||
except SeriesError as exc:
|
except SeriesError as exc:
|
||||||
return _series_err(exc)
|
return _series_err(exc)
|
||||||
await session.commit()
|
await session.commit()
|
||||||
@@ -391,3 +477,201 @@ async def series_cover(tag_id: int):
|
|||||||
return _series_err(exc)
|
return _series_err(exc)
|
||||||
await session.commit()
|
await session.commit()
|
||||||
return jsonify({"ok": True})
|
return jsonify({"ok": True})
|
||||||
|
|
||||||
|
|
||||||
|
# ---- chapter dividers (FC-6.x) -------------------------------------------
|
||||||
|
# A chapter is a cosmetic divider anchored to the page that begins it; it owns
|
||||||
|
# no pages. Page ordering follows each page's operator-set number (the
|
||||||
|
# /pages/number endpoint), so there is no per-chapter reorder/merge — those are
|
||||||
|
# gone.
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series/<int:tag_id>/chapters", methods=["POST"])
|
||||||
|
async def series_chapter_create(tag_id: int):
|
||||||
|
body = await request.get_json() or {}
|
||||||
|
anchor, aerr = _opt_int(body, "anchor_image_id")
|
||||||
|
if aerr:
|
||||||
|
return aerr
|
||||||
|
if anchor is None:
|
||||||
|
return jsonify({"error": "anchor_image_id required"}), 400
|
||||||
|
title = body.get("title")
|
||||||
|
if title is not None and not isinstance(title, str):
|
||||||
|
return jsonify({"error": "title must be a string"}), 400
|
||||||
|
part, perr = _opt_int(body, "stated_part")
|
||||||
|
if perr:
|
||||||
|
return perr
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
ch = await SeriesService(session).create_divider(
|
||||||
|
tag_id, anchor, title=title, stated_part=part,
|
||||||
|
)
|
||||||
|
except SeriesError as exc:
|
||||||
|
return _series_err(exc)
|
||||||
|
await session.commit()
|
||||||
|
return jsonify(ch)
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route(
|
||||||
|
"/series/<int:tag_id>/chapters/<int:chapter_id>", methods=["PATCH"]
|
||||||
|
)
|
||||||
|
async def series_chapter_update(tag_id: int, chapter_id: int):
|
||||||
|
body = await request.get_json() or {}
|
||||||
|
kwargs: dict = {}
|
||||||
|
if "title" in body:
|
||||||
|
if body["title"] is not None and not isinstance(body["title"], str):
|
||||||
|
return jsonify({"error": "title must be a string"}), 400
|
||||||
|
kwargs.update(set_title=True, title=body["title"])
|
||||||
|
if "stated_part" in body:
|
||||||
|
part, perr = _opt_int(body, "stated_part")
|
||||||
|
if perr:
|
||||||
|
return perr
|
||||||
|
kwargs.update(set_part=True, stated_part=part)
|
||||||
|
if "anchor_image_id" in body:
|
||||||
|
anchor, aerr = _opt_int(body, "anchor_image_id")
|
||||||
|
if aerr:
|
||||||
|
return aerr
|
||||||
|
if anchor is None:
|
||||||
|
return jsonify(
|
||||||
|
{"error": "anchor_image_id must be an integer"}
|
||||||
|
), 400
|
||||||
|
kwargs.update(set_anchor=True, anchor_image_id=anchor)
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
await SeriesService(session).update_divider(
|
||||||
|
tag_id, chapter_id, **kwargs
|
||||||
|
)
|
||||||
|
except SeriesError as exc:
|
||||||
|
return _series_err(exc)
|
||||||
|
await session.commit()
|
||||||
|
return jsonify({"ok": True})
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route(
|
||||||
|
"/series/<int:tag_id>/chapters/<int:chapter_id>", methods=["DELETE"]
|
||||||
|
)
|
||||||
|
async def series_chapter_delete(tag_id: int, chapter_id: int):
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
await SeriesService(session).delete_divider(tag_id, chapter_id)
|
||||||
|
except SeriesError as exc:
|
||||||
|
return _series_err(exc)
|
||||||
|
await session.commit()
|
||||||
|
return jsonify({"ok": True})
|
||||||
|
|
||||||
|
|
||||||
|
# ---- browse list + post→series flows (FC-6.2) -----------------------------
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series", methods=["GET"])
|
||||||
|
async def series_list():
|
||||||
|
args = request.args
|
||||||
|
sort = args.get("sort", "recent")
|
||||||
|
if sort not in ("recent", "name", "size"):
|
||||||
|
return jsonify({"error": "sort must be recent|name|size"}), 400
|
||||||
|
artist_id = None
|
||||||
|
if args.get("artist_id") is not None:
|
||||||
|
try:
|
||||||
|
artist_id = int(args["artist_id"])
|
||||||
|
except ValueError:
|
||||||
|
return jsonify({"error": "artist_id must be an integer"}), 400
|
||||||
|
async with get_session() as session:
|
||||||
|
rows = await SeriesService(session).list_series(
|
||||||
|
sort=sort, artist_id=artist_id
|
||||||
|
)
|
||||||
|
return jsonify({"series": rows})
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series/from-post", methods=["POST"])
|
||||||
|
async def series_from_post():
|
||||||
|
body = await request.get_json()
|
||||||
|
post_id, err = _opt_int(body, "post_id")
|
||||||
|
if err:
|
||||||
|
return err
|
||||||
|
if post_id is None:
|
||||||
|
return jsonify({"error": "post_id required"}), 400
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
out = await SeriesService(session).promote_post_to_series(post_id)
|
||||||
|
except SeriesError as exc:
|
||||||
|
return _series_err(exc)
|
||||||
|
await session.commit()
|
||||||
|
return jsonify(out)
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series/<int:tag_id>/add-post", methods=["POST"])
|
||||||
|
async def series_add_post(tag_id: int):
|
||||||
|
body = await request.get_json()
|
||||||
|
post_id, err = _opt_int(body, "post_id")
|
||||||
|
if err:
|
||||||
|
return err
|
||||||
|
if post_id is None:
|
||||||
|
return jsonify({"error": "post_id required"}), 400
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
out = await SeriesService(session).add_post(tag_id, post_id)
|
||||||
|
except SeriesError as exc:
|
||||||
|
return _series_err(exc)
|
||||||
|
await session.commit()
|
||||||
|
return jsonify(out)
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series/<int:tag_id>/pending/place", methods=["POST"])
|
||||||
|
async def series_place_pending(tag_id: int):
|
||||||
|
"""Place staged (pending) pages into the run, numbered sequentially from
|
||||||
|
`start_page` in the given order (#789). start_page null → unnumbered."""
|
||||||
|
body = await request.get_json()
|
||||||
|
ids, err = _parse_bulk_ids(body, max_ids=500)
|
||||||
|
if err:
|
||||||
|
return err
|
||||||
|
start, serr = _opt_int(body, "start_page")
|
||||||
|
if serr:
|
||||||
|
return serr
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
n = await SeriesService(session).place_pending(
|
||||||
|
tag_id, ids, start_page=start
|
||||||
|
)
|
||||||
|
except SeriesError as exc:
|
||||||
|
return _series_err(exc)
|
||||||
|
await session.commit()
|
||||||
|
return jsonify({"placed_count": n})
|
||||||
|
|
||||||
|
|
||||||
|
# ---- suggestion queue (FC-6.3) --------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series/suggestions", methods=["GET"])
|
||||||
|
async def series_suggestions_list():
|
||||||
|
async with get_session() as session:
|
||||||
|
rows = await SeriesMatchService(session).list_pending()
|
||||||
|
return jsonify({"suggestions": rows})
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series/suggestions/<int:sid>/accept", methods=["POST"])
|
||||||
|
async def series_suggestion_accept(sid: int):
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
out = await SeriesMatchService(session).accept(sid)
|
||||||
|
except SeriesError as exc:
|
||||||
|
return _series_err(exc)
|
||||||
|
await session.commit()
|
||||||
|
return jsonify(out)
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series/suggestions/<int:sid>/dismiss", methods=["POST"])
|
||||||
|
async def series_suggestion_dismiss(sid: int):
|
||||||
|
async with get_session() as session:
|
||||||
|
try:
|
||||||
|
await SeriesMatchService(session).dismiss(sid)
|
||||||
|
except SeriesError as exc:
|
||||||
|
return _series_err(exc)
|
||||||
|
await session.commit()
|
||||||
|
return jsonify({"ok": True})
|
||||||
|
|
||||||
|
|
||||||
|
@tags_bp.route("/series/suggestions/rescan", methods=["POST"])
|
||||||
|
async def series_suggestions_rescan():
|
||||||
|
from ..tasks.admin import rescan_series_suggestions_task
|
||||||
|
|
||||||
|
res = rescan_series_suggestions_task.delay()
|
||||||
|
return jsonify({"task_id": res.id})
|
||||||
|
|||||||
@@ -1,5 +1,7 @@
|
|||||||
"""Thumbnail admin API: backfill trigger."""
|
"""Thumbnail admin API: backfill trigger."""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
|
||||||
from quart import Blueprint, jsonify
|
from quart import Blueprint, jsonify
|
||||||
|
|
||||||
thumbnails_bp = Blueprint("thumbnails", __name__, url_prefix="/api/thumbnails")
|
thumbnails_bp = Blueprint("thumbnails", __name__, url_prefix="/api/thumbnails")
|
||||||
@@ -7,7 +9,20 @@ thumbnails_bp = Blueprint("thumbnails", __name__, url_prefix="/api/thumbnails")
|
|||||||
|
|
||||||
@thumbnails_bp.route("/backfill", methods=["POST"])
|
@thumbnails_bp.route("/backfill", methods=["POST"])
|
||||||
async def trigger_backfill():
|
async def trigger_backfill():
|
||||||
from ..tasks.thumbnail import backfill_thumbnails
|
"""Run the backfill scan synchronously, return the counts. The actual
|
||||||
|
thumbnail generation work is still off-loaded to the thumbnail Celery
|
||||||
|
queue via `generate_thumbnail.delay()` per missing row — so this
|
||||||
|
handler is fast even on a 100k-image library (a scan is just SELECT
|
||||||
|
id, thumbnail_path + a file.stat() per row, no heavy work).
|
||||||
|
|
||||||
r = backfill_thumbnails.delay()
|
Operator-flagged 2026-06-01: the previous fire-and-forget shape
|
||||||
return jsonify({"celery_task_id": r.id}), 202
|
returned `{celery_task_id}` only, so the admin UI had no idea whether
|
||||||
|
backfill found 0 or 5000 candidates — \"found nothing\" was
|
||||||
|
indistinguishable from \"the worker isn't picking up the task.\""""
|
||||||
|
from ..tasks.thumbnail import _run_backfill_scan
|
||||||
|
|
||||||
|
# Sync scan inside an executor so we don't block the event loop.
|
||||||
|
counts = await asyncio.get_running_loop().run_in_executor(
|
||||||
|
None, _run_backfill_scan,
|
||||||
|
)
|
||||||
|
return jsonify(counts), 200
|
||||||
|
|||||||
@@ -28,9 +28,9 @@ def make_celery() -> Celery:
|
|||||||
"backend.app.tasks.import_file",
|
"backend.app.tasks.import_file",
|
||||||
"backend.app.tasks.thumbnail",
|
"backend.app.tasks.thumbnail",
|
||||||
"backend.app.tasks.maintenance",
|
"backend.app.tasks.maintenance",
|
||||||
"backend.app.tasks.migration",
|
|
||||||
"backend.app.tasks.ml",
|
"backend.app.tasks.ml",
|
||||||
"backend.app.tasks.download",
|
"backend.app.tasks.download",
|
||||||
|
"backend.app.tasks.external",
|
||||||
"backend.app.tasks.backup",
|
"backend.app.tasks.backup",
|
||||||
"backend.app.tasks.admin",
|
"backend.app.tasks.admin",
|
||||||
"backend.app.tasks.library_audit",
|
"backend.app.tasks.library_audit",
|
||||||
@@ -43,12 +43,20 @@ def make_celery() -> Celery:
|
|||||||
"backend.app.tasks.ml.*": {"queue": "ml"},
|
"backend.app.tasks.ml.*": {"queue": "ml"},
|
||||||
"backend.app.tasks.thumbnail.*": {"queue": "thumbnail"},
|
"backend.app.tasks.thumbnail.*": {"queue": "thumbnail"},
|
||||||
"backend.app.tasks.download.*": {"queue": "download"},
|
"backend.app.tasks.download.*": {"queue": "download"},
|
||||||
|
# External file-host fetches are downloads — same lane (they can run
|
||||||
|
# long, but the download worker already tolerates long backfills).
|
||||||
|
"backend.app.tasks.external.*": {"queue": "download"},
|
||||||
"backend.app.tasks.scan.*": {"queue": "scan"},
|
"backend.app.tasks.scan.*": {"queue": "scan"},
|
||||||
|
# `maintenance` is the QUICK lane — recovery sweeps, vacuum, cleanup
|
||||||
|
# (concurrency-1 on the scheduler). The long one-shots (DB backups,
|
||||||
|
# library audits, admin maintenance: normalize/re-extract/cascade-
|
||||||
|
# delete) run on a SEPARATE `maintenance_long` lane + worker so they
|
||||||
|
# can never starve the quick self-healing sweeps (operator-flagged
|
||||||
|
# 2026-06-07: a 2h audit blocked vacuum/backup/normalize for hours).
|
||||||
"backend.app.tasks.maintenance.*": {"queue": "maintenance"},
|
"backend.app.tasks.maintenance.*": {"queue": "maintenance"},
|
||||||
"backend.app.tasks.migration.*": {"queue": "maintenance"},
|
"backend.app.tasks.backup.*": {"queue": "maintenance_long"},
|
||||||
"backend.app.tasks.backup.*": {"queue": "maintenance"},
|
"backend.app.tasks.admin.*": {"queue": "maintenance_long"},
|
||||||
"backend.app.tasks.admin.*": {"queue": "maintenance"},
|
"backend.app.tasks.library_audit.*": {"queue": "maintenance_long"},
|
||||||
"backend.app.tasks.library_audit.*": {"queue": "maintenance"},
|
|
||||||
},
|
},
|
||||||
# Heavy ML tasks need fair dispatch — see ImageRepo's precedent.
|
# Heavy ML tasks need fair dispatch — see ImageRepo's precedent.
|
||||||
task_acks_late=True,
|
task_acks_late=True,
|
||||||
@@ -87,6 +95,10 @@ def make_celery() -> Celery:
|
|||||||
"task": "backend.app.tasks.maintenance.cleanup_old_download_events",
|
"task": "backend.app.tasks.maintenance.cleanup_old_download_events",
|
||||||
"schedule": 86400.0, # daily
|
"schedule": 86400.0, # daily
|
||||||
},
|
},
|
||||||
|
"recover-stalled-download-events": {
|
||||||
|
"task": "backend.app.tasks.maintenance.recover_stalled_download_events",
|
||||||
|
"schedule": 300.0, # every 5 min, matches recover-interrupted-tasks
|
||||||
|
},
|
||||||
"recover-stalled-task-runs": {
|
"recover-stalled-task-runs": {
|
||||||
"task": "backend.app.tasks.maintenance.recover_stalled_task_runs",
|
"task": "backend.app.tasks.maintenance.recover_stalled_task_runs",
|
||||||
"schedule": 300.0, # every 5 min, matches recover-interrupted-tasks
|
"schedule": 300.0, # every 5 min, matches recover-interrupted-tasks
|
||||||
@@ -95,6 +107,10 @@ def make_celery() -> Celery:
|
|||||||
"task": "backend.app.tasks.maintenance.prune_task_runs",
|
"task": "backend.app.tasks.maintenance.prune_task_runs",
|
||||||
"schedule": 86400.0, # daily
|
"schedule": 86400.0, # daily
|
||||||
},
|
},
|
||||||
|
"vacuum-analyze": {
|
||||||
|
"task": "backend.app.tasks.maintenance.vacuum_analyze",
|
||||||
|
"schedule": 604800.0, # weekly — reclaim dead-tuple bloat + refresh stats
|
||||||
|
},
|
||||||
"fc3h-backup-db-nightly": {
|
"fc3h-backup-db-nightly": {
|
||||||
"task": "backend.app.tasks.backup.backup_db_nightly",
|
"task": "backend.app.tasks.backup.backup_db_nightly",
|
||||||
"schedule": 3600.0, # hourly tick; task self-gates on configured UTC hour
|
"schedule": 3600.0, # hourly tick; task self-gates on configured UTC hour
|
||||||
@@ -103,6 +119,56 @@ def make_celery() -> Celery:
|
|||||||
"task": "backend.app.tasks.backup.prune_backups",
|
"task": "backend.app.tasks.backup.prune_backups",
|
||||||
"schedule": 86400.0, # daily
|
"schedule": 86400.0, # daily
|
||||||
},
|
},
|
||||||
|
# Audit 2026-06-02 — three new per-entity recovery sweeps.
|
||||||
|
# Each runs every 5 min like the other recover_stalled_*
|
||||||
|
# sweeps; each is a no-op when nothing is stuck.
|
||||||
|
"recover-stalled-backup-runs": {
|
||||||
|
"task": "backend.app.tasks.maintenance.recover_stalled_backup_runs",
|
||||||
|
"schedule": 300.0,
|
||||||
|
},
|
||||||
|
"recover-stalled-library-audit-runs": {
|
||||||
|
"task": "backend.app.tasks.maintenance.recover_stalled_library_audit_runs",
|
||||||
|
"schedule": 300.0,
|
||||||
|
},
|
||||||
|
"recover-stalled-import-batches": {
|
||||||
|
"task": "backend.app.tasks.maintenance.recover_stalled_import_batches",
|
||||||
|
"schedule": 300.0,
|
||||||
|
},
|
||||||
|
# Audit 2026-06-02 — daily retention for two entities
|
||||||
|
# whose terminal rows otherwise accumulate forever.
|
||||||
|
"prune-library-audit-runs": {
|
||||||
|
"task": "backend.app.tasks.maintenance.prune_library_audit_runs",
|
||||||
|
"schedule": 86400.0,
|
||||||
|
},
|
||||||
|
"prune-import-batches": {
|
||||||
|
"task": "backend.app.tasks.maintenance.prune_import_batches",
|
||||||
|
"schedule": 86400.0,
|
||||||
|
},
|
||||||
|
# Audit 2026-06-02 — backfill_thumbnails's docstring claimed
|
||||||
|
# "periodic Beat" but the entry was never registered, so the
|
||||||
|
# library got no self-healing thumbnail repair; only the
|
||||||
|
# manual admin-UI button fired it. Daily cadence is gentle
|
||||||
|
# (the task is idempotent and only enqueues regen for rows
|
||||||
|
# whose stored thumbnails are missing or corrupt).
|
||||||
|
"backfill-thumbnails-daily": {
|
||||||
|
"task": "backend.app.tasks.thumbnail.backfill_thumbnails",
|
||||||
|
"schedule": 86400.0,
|
||||||
|
},
|
||||||
|
# External file-host downloads (#830): a steady sweep catches links
|
||||||
|
# the post-download hook missed (worker down, etc.); recovery re-tries
|
||||||
|
# dead links daily; retention prunes long-dead rows.
|
||||||
|
"extdl-sweep": {
|
||||||
|
"task": "backend.app.tasks.external.sweep_external_links",
|
||||||
|
"schedule": 600.0, # every 10 min
|
||||||
|
},
|
||||||
|
"extdl-recover-daily": {
|
||||||
|
"task": "backend.app.tasks.external.recover_external_links",
|
||||||
|
"schedule": 86400.0,
|
||||||
|
},
|
||||||
|
"extdl-prune-daily": {
|
||||||
|
"task": "backend.app.tasks.external.prune_external_links",
|
||||||
|
"schedule": 86400.0,
|
||||||
|
},
|
||||||
},
|
},
|
||||||
timezone="UTC",
|
timezone="UTC",
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -54,7 +54,14 @@ _INT32_MIN = -2_147_483_648
|
|||||||
|
|
||||||
def _queue_for(task) -> str:
|
def _queue_for(task) -> str:
|
||||||
"""Reverse the task→queue routing from celery_app.task_routes.
|
"""Reverse the task→queue routing from celery_app.task_routes.
|
||||||
Keep in sync if task_routes is reordered."""
|
Keep in sync if task_routes is reordered.
|
||||||
|
|
||||||
|
Audit 2026-06-02: backup/admin/library_audit prefixes were
|
||||||
|
missing here even though task_routes sent all three to
|
||||||
|
'maintenance'. The TaskRun.queue column then lied for those
|
||||||
|
rows (claimed 'default') so per-queue dashboard filters and
|
||||||
|
per-queue threshold overrides silently missed them.
|
||||||
|
"""
|
||||||
name = getattr(task, "name", "") or ""
|
name = getattr(task, "name", "") or ""
|
||||||
if name.startswith("backend.app.tasks.import_file."):
|
if name.startswith("backend.app.tasks.import_file."):
|
||||||
return "import"
|
return "import"
|
||||||
@@ -68,7 +75,9 @@ def _queue_for(task) -> str:
|
|||||||
return "scan"
|
return "scan"
|
||||||
if name.startswith((
|
if name.startswith((
|
||||||
"backend.app.tasks.maintenance.",
|
"backend.app.tasks.maintenance.",
|
||||||
"backend.app.tasks.migration.",
|
"backend.app.tasks.backup.",
|
||||||
|
"backend.app.tasks.admin.",
|
||||||
|
"backend.app.tasks.library_audit.",
|
||||||
)):
|
)):
|
||||||
return "maintenance"
|
return "maintenance"
|
||||||
return "default"
|
return "default"
|
||||||
|
|||||||
@@ -2,21 +2,27 @@
|
|||||||
|
|
||||||
from .app_setting import AppSetting
|
from .app_setting import AppSetting
|
||||||
from .artist import Artist
|
from .artist import Artist
|
||||||
|
from .artist_visit import ArtistVisit
|
||||||
from .backup_run import BackupRun
|
from .backup_run import BackupRun
|
||||||
from .base import Base
|
from .base import Base
|
||||||
from .credential import Credential
|
from .credential import Credential
|
||||||
from .download_event import DownloadEvent
|
from .download_event import DownloadEvent
|
||||||
|
from .external_link import ExternalLink
|
||||||
|
from .image_prediction import ImagePrediction
|
||||||
from .image_provenance import ImageProvenance
|
from .image_provenance import ImageProvenance
|
||||||
from .image_record import ImageRecord
|
from .image_record import ImageRecord
|
||||||
from .import_batch import ImportBatch
|
from .import_batch import ImportBatch
|
||||||
from .import_settings import ImportSettings
|
from .import_settings import ImportSettings
|
||||||
from .import_task import ImportTask
|
from .import_task import ImportTask
|
||||||
from .library_audit_run import LibraryAuditRun
|
from .library_audit_run import LibraryAuditRun
|
||||||
from .migration_run import MigrationRun
|
|
||||||
from .ml_settings import MLSettings
|
from .ml_settings import MLSettings
|
||||||
|
from .patreon_failed_media import PatreonFailedMedia
|
||||||
|
from .patreon_seen_media import PatreonSeenMedia
|
||||||
from .post import Post
|
from .post import Post
|
||||||
from .post_attachment import PostAttachment
|
from .post_attachment import PostAttachment
|
||||||
|
from .series_chapter import SeriesChapter
|
||||||
from .series_page import SeriesPage
|
from .series_page import SeriesPage
|
||||||
|
from .series_suggestion import SeriesSuggestion
|
||||||
from .source import Source
|
from .source import Source
|
||||||
from .tag import Tag, TagKind, image_tag
|
from .tag import Tag, TagKind, image_tag
|
||||||
from .tag_alias import TagAlias
|
from .tag_alias import TagAlias
|
||||||
@@ -29,24 +35,30 @@ __all__ = [
|
|||||||
"Base",
|
"Base",
|
||||||
"AppSetting",
|
"AppSetting",
|
||||||
"Artist",
|
"Artist",
|
||||||
|
"ArtistVisit",
|
||||||
"BackupRun",
|
"BackupRun",
|
||||||
"Source",
|
"Source",
|
||||||
"Credential",
|
"Credential",
|
||||||
|
"PatreonFailedMedia",
|
||||||
|
"PatreonSeenMedia",
|
||||||
"Post",
|
"Post",
|
||||||
"PostAttachment",
|
"PostAttachment",
|
||||||
|
"SeriesChapter",
|
||||||
"SeriesPage",
|
"SeriesPage",
|
||||||
|
"SeriesSuggestion",
|
||||||
"ImageRecord",
|
"ImageRecord",
|
||||||
|
"ImagePrediction",
|
||||||
"ImageProvenance",
|
"ImageProvenance",
|
||||||
"Tag",
|
"Tag",
|
||||||
"TagKind",
|
"TagKind",
|
||||||
"image_tag",
|
"image_tag",
|
||||||
"DownloadEvent",
|
"DownloadEvent",
|
||||||
|
"ExternalLink",
|
||||||
"ImportBatch",
|
"ImportBatch",
|
||||||
"ImportTask",
|
"ImportTask",
|
||||||
"ImportSettings",
|
"ImportSettings",
|
||||||
"LibraryAuditRun",
|
"LibraryAuditRun",
|
||||||
"MLSettings",
|
"MLSettings",
|
||||||
"MigrationRun",
|
|
||||||
"TagAlias",
|
"TagAlias",
|
||||||
"TagAllowlist",
|
"TagAllowlist",
|
||||||
"TagReferenceEmbedding",
|
"TagReferenceEmbedding",
|
||||||
|
|||||||
@@ -0,0 +1,36 @@
|
|||||||
|
"""ArtistVisit — per-artist 'last viewed' timestamp.
|
||||||
|
|
||||||
|
Powers the "+N new since last visit" badge on the artists directory and
|
||||||
|
the matching banner on `ArtistView`. One row per artist, single global
|
||||||
|
operator. When the multi-user model lands, the PK widens to
|
||||||
|
`(user_id, artist_id)` — currently aspirational only (no User model,
|
||||||
|
no services/access.py); operator approved skipping `user_id` for now
|
||||||
|
under rule #22 (breaking changes welcome).
|
||||||
|
|
||||||
|
Seed at migration time: every existing artist gets `last_viewed_at = NOW()`
|
||||||
|
so the badge starts at 0 across the board (no noisy "5000 unseen" on
|
||||||
|
first deploy). New artists also auto-get a row via
|
||||||
|
`ArtistService.find_or_create`.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import DateTime, ForeignKey, Integer, func
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from .base import Base
|
||||||
|
|
||||||
|
|
||||||
|
class ArtistVisit(Base):
|
||||||
|
__tablename__ = "artist_visit"
|
||||||
|
|
||||||
|
artist_id: Mapped[int] = mapped_column(
|
||||||
|
Integer,
|
||||||
|
ForeignKey("artist.id", ondelete="CASCADE"),
|
||||||
|
primary_key=True,
|
||||||
|
)
|
||||||
|
last_viewed_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=func.now(),
|
||||||
|
)
|
||||||
@@ -0,0 +1,73 @@
|
|||||||
|
"""ExternalLink — an off-platform file-host link found in a post body.
|
||||||
|
|
||||||
|
Creators host the actual files (films, packs) on mega.nz / Google Drive /
|
||||||
|
MediaFire / Dropbox / Pixeldrain and drop the link in the post text. This row
|
||||||
|
is the record that the link existed (so nothing is silently dropped), the
|
||||||
|
dedup + dead-letter ledger for fetching it, and the driver the download worker
|
||||||
|
walks. `url` keeps the FULL link including the `#fragment` (mega's decryption
|
||||||
|
key) — truncating it makes the file undownloadable.
|
||||||
|
|
||||||
|
status lifecycle: pending → downloading → downloaded | failed | dead
|
||||||
|
(too many attempts) | skipped (host disabled). `attachment_id` links the
|
||||||
|
captured file once a download lands (SET NULL so deleting the attachment
|
||||||
|
doesn't delete the link record).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import (
|
||||||
|
DateTime,
|
||||||
|
Float,
|
||||||
|
ForeignKey,
|
||||||
|
Index,
|
||||||
|
Integer,
|
||||||
|
String,
|
||||||
|
Text,
|
||||||
|
func,
|
||||||
|
text,
|
||||||
|
)
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from .base import Base
|
||||||
|
|
||||||
|
# Kept in sync with link_extract.SUPPORTED_HOSTS and the CHECK in migration 0049.
|
||||||
|
HOSTS = ("mega", "gdrive", "mediafire", "dropbox", "pixeldrain")
|
||||||
|
STATUSES = ("pending", "downloading", "downloaded", "failed", "skipped", "dead")
|
||||||
|
|
||||||
|
|
||||||
|
class ExternalLink(Base):
|
||||||
|
__tablename__ = "external_link"
|
||||||
|
__table_args__ = (
|
||||||
|
# One row per (post, url). The full url (incl. #fragment) is the identity
|
||||||
|
# — the same file linked twice in a post collapses to one row.
|
||||||
|
Index("uq_external_link_post_url", "post_id", "url", unique=True),
|
||||||
|
Index("ix_external_link_status", "status"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
|
post_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("post.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
|
)
|
||||||
|
artist_id: Mapped[int | None] = mapped_column(
|
||||||
|
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
||||||
|
)
|
||||||
|
host: Mapped[str] = mapped_column(String(16), nullable=False)
|
||||||
|
url: Mapped[str] = mapped_column(Text, nullable=False)
|
||||||
|
label: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
status: Mapped[str] = mapped_column(
|
||||||
|
String(16), nullable=False, server_default="pending"
|
||||||
|
)
|
||||||
|
attempts: Mapped[int] = mapped_column(
|
||||||
|
Integer, nullable=False, server_default=text("0")
|
||||||
|
)
|
||||||
|
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
attachment_id: Mapped[int | None] = mapped_column(
|
||||||
|
ForeignKey("post_attachment.id", ondelete="SET NULL"), nullable=True
|
||||||
|
)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
)
|
||||||
|
completed_at: Mapped[datetime | None] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=True
|
||||||
|
)
|
||||||
|
duration_seconds: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
"""ImagePrediction — one row per (image, tagger vocab prediction).
|
||||||
|
|
||||||
|
Replaces the image_record.tagger_predictions JSON blob (#768). Storing the
|
||||||
|
raw Camie/booru vocab name (not a tag_id) preserves the suggestion read
|
||||||
|
path's semantics: raw_name → canonical Tag resolution happens at read time
|
||||||
|
via the alias map, and accepting a prediction can CREATE the Tag. The store
|
||||||
|
floor (ml_settings.tagger_store_floor) is applied at WRITE time, so only
|
||||||
|
predictions >= the floor land here.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from sqlalchemy import Float, ForeignKey, Index, String, UniqueConstraint
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from .base import Base
|
||||||
|
|
||||||
|
|
||||||
|
class ImagePrediction(Base):
|
||||||
|
__tablename__ = "image_prediction"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint(
|
||||||
|
"image_record_id", "raw_name", name="image_raw_name",
|
||||||
|
),
|
||||||
|
# Per-image read (suggestion build) and the "images with tag X above
|
||||||
|
# Y" query the JSON blob never allowed.
|
||||||
|
Index("ix_image_prediction_image", "image_record_id"),
|
||||||
|
Index("ix_image_prediction_name_score", "raw_name", "score"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
image_record_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("image_record.id", ondelete="CASCADE"), nullable=False,
|
||||||
|
)
|
||||||
|
# The raw tagger vocab key (booru form) — NOT a tag_id. Resolved to a
|
||||||
|
# canonical Tag at read time, exactly as the old JSON keys were.
|
||||||
|
raw_name: Mapped[str] = mapped_column(String(255), nullable=False)
|
||||||
|
category: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||||
|
score: Mapped[float] = mapped_column(Float, nullable=False)
|
||||||
@@ -34,8 +34,12 @@ class ImageProvenance(Base):
|
|||||||
post_id: Mapped[int] = mapped_column(
|
post_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("post.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("post.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
)
|
)
|
||||||
source_id: Mapped[int] = mapped_column(
|
# Nullable since alembic 0030 — provenance rows for filesystem-imported
|
||||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
# content with no subscription have NULL source_id. FK ondelete SET
|
||||||
|
# NULL so deleting a Source detaches its provenance rows instead of
|
||||||
|
# destroying the linkage between image and post.
|
||||||
|
source_id: Mapped[int | None] = mapped_column(
|
||||||
|
ForeignKey("source.id", ondelete="SET NULL"), nullable=True, index=True
|
||||||
)
|
)
|
||||||
captured_metadata: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
captured_metadata: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
||||||
captured_at: Mapped[datetime] = mapped_column(
|
captured_at: Mapped[datetime] = mapped_column(
|
||||||
|
|||||||
@@ -49,6 +49,18 @@ class ImageRecord(Base):
|
|||||||
# Thumbnail (populated by FC-2)
|
# Thumbnail (populated by FC-2)
|
||||||
thumbnail_path: Mapped[str | None] = mapped_column(Text, nullable=True)
|
thumbnail_path: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
|
||||||
|
# Source provenance for downloaded media (#830 Phase 2). `source_url` is the
|
||||||
|
# CDN/origin URL the file was fetched from (debugging + future re-fetch).
|
||||||
|
# `source_filehash` is the URL's 32-hex CDN identity segment
|
||||||
|
# (utils.paths.filehash_from_url) — the JOIN KEY that maps a post body's
|
||||||
|
# inline `<img src=CDN>` back to this local copy so the rendered body serves
|
||||||
|
# our stored image instead of hotlinking the public source. Indexed for the
|
||||||
|
# render-time lookup. NULL for filesystem-imported / pre-Phase-2 rows.
|
||||||
|
source_url: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
source_filehash: Mapped[str | None] = mapped_column(
|
||||||
|
String(32), nullable=True, index=True
|
||||||
|
)
|
||||||
|
|
||||||
# Origin / provenance pointers
|
# Origin / provenance pointers
|
||||||
origin: Mapped[str] = mapped_column(Enum(*ORIGIN_CHOICES, name="origin_enum"), nullable=False)
|
origin: Mapped[str] = mapped_column(Enum(*ORIGIN_CHOICES, name="origin_enum"), nullable=False)
|
||||||
primary_post_id: Mapped[int | None] = mapped_column(
|
primary_post_id: Mapped[int | None] = mapped_column(
|
||||||
@@ -60,8 +72,10 @@ class ImageRecord(Base):
|
|||||||
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
||||||
)
|
)
|
||||||
|
|
||||||
# ML fields (populated by FC-2's ml-worker)
|
# ML fields (populated by FC-2's ml-worker). Per-tag predictions live in the
|
||||||
tagger_predictions: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
# normalized image_prediction table (#768) — the tagger_predictions JSON
|
||||||
|
# column was dropped in migration 0046. tagger_model_version stays as the
|
||||||
|
# "has this been tagged / is it current?" signal the backfill sweep reads.
|
||||||
tagger_model_version: Mapped[str | None] = mapped_column(String(128), nullable=True)
|
tagger_model_version: Mapped[str | None] = mapped_column(String(128), nullable=True)
|
||||||
# 1152 = SigLIP-so400m embedding dim. Swapping models in FC-2 may require
|
# 1152 = SigLIP-so400m embedding dim. Swapping models in FC-2 may require
|
||||||
# a column-width migration.
|
# a column-width migration.
|
||||||
@@ -74,6 +88,17 @@ class ImageRecord(Base):
|
|||||||
created_at: Mapped[datetime] = mapped_column(
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
)
|
)
|
||||||
|
# Denormalized gallery sort key = COALESCE(primary post's post_date,
|
||||||
|
# created_at) (alembic 0035). The gallery used to compute this as a
|
||||||
|
# COALESCE across the Post outer join on every /scroll, which can't use
|
||||||
|
# an index and re-sorted a large slice of the library per page (×10 with
|
||||||
|
# the old serial batching). Materializing it lets the cursor scroll read
|
||||||
|
# ix_image_record_effective_date directly. Maintained by the importer
|
||||||
|
# (services/importer.py _apply_sidecar) when a primary post with a date
|
||||||
|
# is linked; plain inserts keep the created_at-equivalent server default.
|
||||||
|
effective_date: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
)
|
||||||
updated_at: Mapped[datetime] = mapped_column(
|
updated_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True),
|
DateTime(timezone=True),
|
||||||
nullable=False,
|
nullable=False,
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ Enforced as a single row via a CHECK (id = 1) constraint. The application
|
|||||||
always SELECTs id=1 and never inserts/deletes after the initial migration.
|
always SELECTs id=1 and never inserts/deletes after the initial migration.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from sqlalchemy import Boolean, CheckConstraint, Float, Integer, Text
|
from sqlalchemy import Boolean, CheckConstraint, Float, Integer, Text, select
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -63,3 +63,41 @@ class ImportSettings(Base):
|
|||||||
backup_images_keep_last_n: Mapped[int] = mapped_column(
|
backup_images_keep_last_n: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=3,
|
Integer, nullable=False, default=3,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# FC-6.3 series continuation matcher. enabled gates the rescan; threshold is
|
||||||
|
# the weighted-score cut-off (0..1) above which a pending suggestion is made.
|
||||||
|
series_suggest_enabled: Mapped[bool] = mapped_column(
|
||||||
|
Boolean, nullable=False, default=True,
|
||||||
|
)
|
||||||
|
series_suggest_threshold: Mapped[float] = mapped_column(
|
||||||
|
Float, nullable=False, default=0.5,
|
||||||
|
)
|
||||||
|
|
||||||
|
# #830 off-platform file-host downloads — per-host enable lever (default on,
|
||||||
|
# rule #26). Column names are extdl_<host>_enabled so the worker reads them
|
||||||
|
# via getattr(settings, f"extdl_{host}_enabled", True).
|
||||||
|
extdl_mega_enabled: Mapped[bool] = mapped_column(
|
||||||
|
Boolean, nullable=False, default=True, server_default="true",
|
||||||
|
)
|
||||||
|
extdl_gdrive_enabled: Mapped[bool] = mapped_column(
|
||||||
|
Boolean, nullable=False, default=True, server_default="true",
|
||||||
|
)
|
||||||
|
extdl_mediafire_enabled: Mapped[bool] = mapped_column(
|
||||||
|
Boolean, nullable=False, default=True, server_default="true",
|
||||||
|
)
|
||||||
|
extdl_dropbox_enabled: Mapped[bool] = mapped_column(
|
||||||
|
Boolean, nullable=False, default=True, server_default="true",
|
||||||
|
)
|
||||||
|
extdl_pixeldrain_enabled: Mapped[bool] = mapped_column(
|
||||||
|
Boolean, nullable=False, default=True, server_default="true",
|
||||||
|
)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
async def load(cls, session) -> ImportSettings:
|
||||||
|
"""The singleton settings row (id=1), via an async session."""
|
||||||
|
return (await session.execute(select(cls).where(cls.id == 1))).scalar_one()
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def load_sync(cls, session) -> ImportSettings:
|
||||||
|
"""The singleton settings row (id=1), via a sync session."""
|
||||||
|
return session.execute(select(cls).where(cls.id == 1)).scalar_one()
|
||||||
|
|||||||
@@ -35,3 +35,10 @@ class LibraryAuditRun(Base):
|
|||||||
matched_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
matched_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||||
matched_ids: Mapped[list[int]] = mapped_column(JSONB, nullable=False, default=list)
|
matched_ids: Mapped[list[int]] = mapped_column(JSONB, nullable=False, default=list)
|
||||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
# Chunked-scan state (alembic 0039): keyset cursor the next chunk resumes
|
||||||
|
# from, and the last time a chunk made progress (so the recovery sweep can
|
||||||
|
# tell a progressing multi-chunk audit from a stuck one).
|
||||||
|
resume_after_id: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||||
|
last_progress_at: Mapped[datetime | None] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=True,
|
||||||
|
)
|
||||||
|
|||||||
@@ -1,37 +0,0 @@
|
|||||||
"""MigrationRun — tracks each FC-5 migration invocation (backup/gs/ir/etc).
|
|
||||||
|
|
||||||
kind/status are String(32) not Postgres ENUM so adding kinds later
|
|
||||||
doesn't need a schema migration. The API layer validates values.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from datetime import datetime
|
|
||||||
|
|
||||||
import sqlalchemy as sa
|
|
||||||
from sqlalchemy import Boolean, DateTime, Integer, String, Text, func
|
|
||||||
from sqlalchemy.dialects.postgresql import JSONB
|
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
|
||||||
|
|
||||||
from .base import Base
|
|
||||||
|
|
||||||
|
|
||||||
class MigrationRun(Base):
|
|
||||||
__tablename__ = "migration_run"
|
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
|
||||||
kind: Mapped[str] = mapped_column(String(32), nullable=False, index=True)
|
|
||||||
status: Mapped[str] = mapped_column(String(32), nullable=False, index=True)
|
|
||||||
dry_run: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
|
||||||
started_at: Mapped[datetime] = mapped_column(
|
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now(),
|
|
||||||
)
|
|
||||||
finished_at: Mapped[datetime | None] = mapped_column(
|
|
||||||
DateTime(timezone=True), nullable=True,
|
|
||||||
)
|
|
||||||
counts: Mapped[dict] = mapped_column(
|
|
||||||
JSONB, nullable=False, default=dict, server_default=sa.text("'{}'::jsonb"),
|
|
||||||
)
|
|
||||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
|
||||||
metadata_: Mapped[dict] = mapped_column(
|
|
||||||
"metadata", JSONB, nullable=False, default=dict,
|
|
||||||
server_default=sa.text("'{}'::jsonb"),
|
|
||||||
)
|
|
||||||
@@ -15,21 +15,28 @@ class MLSettings(Base):
|
|||||||
__table_args__ = (CheckConstraint("id = 1", name="singleton"),)
|
__table_args__ = (CheckConstraint("id = 1", name="singleton"),)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
suggestion_threshold_artist: Mapped[float] = mapped_column(
|
|
||||||
Float, nullable=False, default=0.30
|
|
||||||
)
|
|
||||||
suggestion_threshold_character: Mapped[float] = mapped_column(
|
suggestion_threshold_character: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.50
|
Float, nullable=False, default=0.70
|
||||||
)
|
|
||||||
suggestion_threshold_copyright: Mapped[float] = mapped_column(
|
|
||||||
Float, nullable=False, default=0.50
|
|
||||||
)
|
)
|
||||||
|
# Default raised 0.50 → 0.70 on 2026-06-02 — operator-flagged 0.50
|
||||||
|
# surfaced too many low-confidence picks; 0.70 keeps the rail
|
||||||
|
# signal-rich while still surfacing more than the original 0.95
|
||||||
|
# which hid almost everything. Operator-tunable via Settings → ML.
|
||||||
suggestion_threshold_general: Mapped[float] = mapped_column(
|
suggestion_threshold_general: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.95
|
Float, nullable=False, default=0.70
|
||||||
)
|
)
|
||||||
centroid_similarity_threshold: Mapped[float] = mapped_column(
|
centroid_similarity_threshold: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.55
|
Float, nullable=False, default=0.55
|
||||||
)
|
)
|
||||||
|
# Ingest floor: tagger predictions below this confidence are not stored
|
||||||
|
# (tagger.Tagger.infer). Default 0.70 — the suggestion path already
|
||||||
|
# filters at 0.70 and the centroid/learned path covers low-confidence
|
||||||
|
# preferred tags, so the sub-0.70 tail is redundant weight (it had
|
||||||
|
# bloated image_record's TOAST to ~100 GB; plan-task #764). Operator-
|
||||||
|
# tunable via Settings → ML; must stay ≤ the suggestion thresholds.
|
||||||
|
tagger_store_floor: Mapped[float] = mapped_column(
|
||||||
|
Float, nullable=False, default=0.70
|
||||||
|
)
|
||||||
min_reference_images: Mapped[int] = mapped_column(
|
min_reference_images: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=5
|
Integer, nullable=False, default=5
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -0,0 +1,45 @@
|
|||||||
|
"""PatreonFailedMedia — per-source dead-letter ledger of Patreon media that
|
||||||
|
keeps failing to download/validate.
|
||||||
|
|
||||||
|
Plan #705 (#7). A media that fails every walk (404'd CDN URL, deleted post,
|
||||||
|
geo-blocked Mux stream, persistently-corrupt bytes) would otherwise re-error
|
||||||
|
forever and re-burn backfill chunks. After ``attempts`` reaches the dead-letter
|
||||||
|
threshold the ingester skips it on routine tick/backfill walks (recovery still
|
||||||
|
re-attempts it — the operator's "try everything again"). A later clean download
|
||||||
|
clears the row (the media recovered).
|
||||||
|
|
||||||
|
`filehash` is the same per-media key the seen-ledger uses (32-hex CDN MD5, or a
|
||||||
|
``video:`` / ``post:filename`` synthesized key) — hence String(128). UNIQUE
|
||||||
|
(source_id, filehash) is the upsert key.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import ForeignKey, Integer, String, Text, UniqueConstraint, func
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
from sqlalchemy.types import DateTime
|
||||||
|
|
||||||
|
from .base import Base
|
||||||
|
|
||||||
|
|
||||||
|
class PatreonFailedMedia(Base):
|
||||||
|
__tablename__ = "patreon_failed_media"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint(
|
||||||
|
"source_id", "filehash", name="uq_patreon_failed_media_source_id"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
|
source_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
|
)
|
||||||
|
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||||
|
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1)
|
||||||
|
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
first_failed_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
)
|
||||||
|
last_failed_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
)
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
"""PatreonSeenMedia — per-source ledger of Patreon media already
|
||||||
|
downloaded+processed.
|
||||||
|
|
||||||
|
Replaces gallery-dl's archive.sqlite3 with our own queryable table so
|
||||||
|
routine walks can skip media we've already ingested (and a future
|
||||||
|
"recovery" mode can deliberately bypass the ledger to re-walk).
|
||||||
|
|
||||||
|
`filehash` is normally a Patreon CDN MD5 (32 hex chars), but videos —
|
||||||
|
which have no stable content hash at discovery time — use a sentinel of
|
||||||
|
the form ``video:<post_id>:<media_id>``, hence String(128) rather than 32.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import ForeignKey, Integer, String, UniqueConstraint, func
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
from sqlalchemy.types import DateTime
|
||||||
|
|
||||||
|
from .base import Base
|
||||||
|
|
||||||
|
|
||||||
|
class PatreonSeenMedia(Base):
|
||||||
|
__tablename__ = "patreon_seen_media"
|
||||||
|
__table_args__ = (
|
||||||
|
# Dedup key the downloader upserts against: one ledger row per
|
||||||
|
# (source, media). A second sighting of the same media is a no-op.
|
||||||
|
UniqueConstraint("source_id", "filehash", name="uq_patreon_seen_media_source_id"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
|
source_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
|
)
|
||||||
|
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||||
|
post_id: Mapped[str | None] = mapped_column(String(64), nullable=True)
|
||||||
|
seen_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
)
|
||||||
@@ -1,6 +1,9 @@
|
|||||||
"""Post — provenance anchor for content downloaded from a Source.
|
"""Post — provenance anchor for one creator post (may contain many images).
|
||||||
|
|
||||||
A Post is one creator post; it may contain many images/videos.
|
`source_id` is nullable since alembic 0030 — filesystem-imported posts
|
||||||
|
with no live subscription have NULL source_id. `artist_id` is the
|
||||||
|
denormalized always-present link to the creator (added in 0030 so
|
||||||
|
artist-filter queries don't depend on the Source detour).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
@@ -14,12 +17,25 @@ from .base import Base
|
|||||||
class Post(Base):
|
class Post(Base):
|
||||||
__tablename__ = "post"
|
__tablename__ = "post"
|
||||||
__table_args__ = (
|
__table_args__ = (
|
||||||
|
# Source-bound dedup. Postgres treats NULL != NULL so rows
|
||||||
|
# with source_id IS NULL aren't deduped by this constraint;
|
||||||
|
# the partial unique index `uq_post_artist_external_id_null_source`
|
||||||
|
# (created in alembic 0030) covers that case via
|
||||||
|
# (artist_id, external_post_id).
|
||||||
UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"),
|
UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"),
|
||||||
)
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
source_id: Mapped[int] = mapped_column(
|
source_id: Mapped[int | None] = mapped_column(
|
||||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("source.id", ondelete="SET NULL"), nullable=True, index=True
|
||||||
|
)
|
||||||
|
# Denormalized; always equals source.artist_id when source_id is set
|
||||||
|
# (the importer is responsible for keeping them consistent on insert).
|
||||||
|
# Filter queries (artist detail, artist-scoped posts feed) use this
|
||||||
|
# directly instead of joining through Source.
|
||||||
|
artist_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("artist.id", ondelete="CASCADE"),
|
||||||
|
nullable=False, index=True,
|
||||||
)
|
)
|
||||||
external_post_id: Mapped[str] = mapped_column(String(128), nullable=False)
|
external_post_id: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||||
post_url: Mapped[str | None] = mapped_column(Text, nullable=True)
|
post_url: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
|||||||
@@ -14,10 +14,12 @@ from sqlalchemy import (
|
|||||||
BigInteger,
|
BigInteger,
|
||||||
DateTime,
|
DateTime,
|
||||||
ForeignKey,
|
ForeignKey,
|
||||||
|
Index,
|
||||||
Integer,
|
Integer,
|
||||||
String,
|
String,
|
||||||
Text,
|
Text,
|
||||||
func,
|
func,
|
||||||
|
text,
|
||||||
)
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
@@ -26,6 +28,24 @@ from .base import Base
|
|||||||
|
|
||||||
class PostAttachment(Base):
|
class PostAttachment(Base):
|
||||||
__tablename__ = "post_attachment"
|
__tablename__ = "post_attachment"
|
||||||
|
# Dedup is PER-POST, not global (2026-06-08): the same non-art file attached
|
||||||
|
# to many posts gets one row per post over a single sha-addressed blob, so no
|
||||||
|
# post is left a bare shell. Partial uniques: (post_id, sha256) for real posts;
|
||||||
|
# (sha256) alone for the NULL-post filesystem case (one row per file there).
|
||||||
|
__table_args__ = (
|
||||||
|
Index(
|
||||||
|
"uq_post_attachment_post_sha",
|
||||||
|
"post_id", "sha256",
|
||||||
|
unique=True,
|
||||||
|
postgresql_where=text("post_id IS NOT NULL"),
|
||||||
|
),
|
||||||
|
Index(
|
||||||
|
"uq_post_attachment_null_post_sha",
|
||||||
|
"sha256",
|
||||||
|
unique=True,
|
||||||
|
postgresql_where=text("post_id IS NULL"),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
post_id: Mapped[int | None] = mapped_column(
|
post_id: Mapped[int | None] = mapped_column(
|
||||||
@@ -35,7 +55,7 @@ class PostAttachment(Base):
|
|||||||
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
||||||
)
|
)
|
||||||
sha256: Mapped[str] = mapped_column(
|
sha256: Mapped[str] = mapped_column(
|
||||||
String(64), nullable=False, unique=True, index=True
|
String(64), nullable=False, index=True
|
||||||
)
|
)
|
||||||
path: Mapped[str] = mapped_column(Text, nullable=False)
|
path: Mapped[str] = mapped_column(Text, nullable=False)
|
||||||
original_filename: Mapped[str] = mapped_column(Text, nullable=False)
|
original_filename: Mapped[str] = mapped_column(Text, nullable=False)
|
||||||
|
|||||||
@@ -0,0 +1,47 @@
|
|||||||
|
"""SeriesChapter — a cosmetic chapter DIVIDER within a series (FC-6.x reframe).
|
||||||
|
|
||||||
|
A series is ONE flat, series-global ordered run of SeriesPages. A chapter is NOT
|
||||||
|
a container — it owns no pages. It is a labeled divider anchored to the page that
|
||||||
|
BEGINS the chapter (anchor_page_id → series_page): "a new chapter starts here."
|
||||||
|
A page's chapter is derived at read time as the nearest preceding divider.
|
||||||
|
|
||||||
|
Dividers never affect page ordering or the series-global page numbers; they stay
|
||||||
|
pinned to their anchor page across reorders. anchor_page_id is UNIQUE — at most
|
||||||
|
one chapter begins at a given page — and FK-cascades, so removing the anchor page
|
||||||
|
from the series drops the divider (the chapter merges into the preceding run).
|
||||||
|
|
||||||
|
title is the optional chapter name; stated_part is the optional operator-facing
|
||||||
|
"Part N" label (shown instead of a derived ordinal when set).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import DateTime, ForeignKey, Integer, Text, func
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from .base import Base
|
||||||
|
|
||||||
|
|
||||||
|
class SeriesChapter(Base):
|
||||||
|
__tablename__ = "series_chapter"
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
|
series_tag_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
|
)
|
||||||
|
anchor_page_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("series_page.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
unique=True,
|
||||||
|
)
|
||||||
|
title: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
stated_part: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
)
|
||||||
|
updated_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=func.now(),
|
||||||
|
onupdate=func.now(),
|
||||||
|
)
|
||||||
@@ -1,14 +1,20 @@
|
|||||||
"""SeriesPage — ordered image membership for a series-kind Tag.
|
"""SeriesPage — ordered image membership for a series-kind Tag.
|
||||||
|
|
||||||
A series IS a Tag with kind='series'; series_page gives it ordered pages.
|
A series IS a Tag with kind='series'; series_page gives it a SINGLE flat,
|
||||||
An image belongs to at most one series (UNIQUE image_id). Cover = the
|
series-global ordered run of pages (FC-6.x divider reframe). An image belongs to
|
||||||
lowest page_number. page_number is an ordering key only (not unique) —
|
at most one series (UNIQUE image_id). Reading order is `page_number` alone — a
|
||||||
reorder rewrites 1..N wholesale.
|
series-wide ordering key (not unique), rewritten 1..N wholesale on reorder so a
|
||||||
|
reorder can't transiently collide on an index.
|
||||||
|
|
||||||
|
Chapters are cosmetic DIVIDERS anchored to a page (see SeriesChapter); they do
|
||||||
|
NOT own pages, so there is no chapter_id here — a page's chapter is derived at
|
||||||
|
read time as the nearest preceding divider. stated_page carries the printed page
|
||||||
|
number parsed from the source post, nullable when unknown.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import DateTime, ForeignKey, Integer, func
|
from sqlalchemy import DateTime, ForeignKey, Integer, String, func
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -26,7 +32,13 @@ class SeriesPage(Base):
|
|||||||
nullable=False,
|
nullable=False,
|
||||||
unique=True,
|
unique=True,
|
||||||
)
|
)
|
||||||
page_number: Mapped[int] = mapped_column(Integer, nullable=False)
|
# 'placed' = in the series-global run (page_number set); 'pending' = staged
|
||||||
|
# from a post awaiting the operator's sort (page_number NULL). (#789 P2)
|
||||||
|
status: Mapped[str] = mapped_column(
|
||||||
|
String(16), nullable=False, server_default="placed"
|
||||||
|
)
|
||||||
|
page_number: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
|
stated_page: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
created_at: Mapped[datetime] = mapped_column(
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -0,0 +1,55 @@
|
|||||||
|
"""SeriesSuggestion — a confirm-only "this post may continue this series" hint.
|
||||||
|
|
||||||
|
The matcher (FC-6.3) scores a (post, candidate series) pair from several weighted
|
||||||
|
signals and, above the configured threshold, records a pending suggestion. The
|
||||||
|
operator confirms (→ the post is added as a chapter) or dismisses it; FC never
|
||||||
|
files a post into a series on its own. status is a plain string (no Postgres
|
||||||
|
ENUM — see the check-existing-enums lesson): pending | added | dismissed.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import (
|
||||||
|
JSON,
|
||||||
|
DateTime,
|
||||||
|
Float,
|
||||||
|
ForeignKey,
|
||||||
|
Integer,
|
||||||
|
String,
|
||||||
|
UniqueConstraint,
|
||||||
|
func,
|
||||||
|
)
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from .base import Base
|
||||||
|
|
||||||
|
|
||||||
|
class SeriesSuggestion(Base):
|
||||||
|
__tablename__ = "series_suggestion"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint(
|
||||||
|
"post_id", "series_tag_id", name="uq_series_suggestion_post_series"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
|
post_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("post.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
|
)
|
||||||
|
series_tag_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
|
)
|
||||||
|
score: Mapped[float] = mapped_column(Float, nullable=False)
|
||||||
|
signals: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
||||||
|
status: Mapped[str] = mapped_column(
|
||||||
|
String(16), nullable=False, server_default="pending", index=True
|
||||||
|
)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
)
|
||||||
|
updated_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True),
|
||||||
|
nullable=False,
|
||||||
|
server_default=func.now(),
|
||||||
|
onupdate=func.now(),
|
||||||
|
)
|
||||||
@@ -26,7 +26,21 @@ class Source(Base):
|
|||||||
|
|
||||||
last_checked_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
last_checked_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
||||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
# alembic 0032: last ErrorType category (auth_error, rate_limited,
|
||||||
|
# not_found, ...). Lets FailingSourcesCard surface the taxonomy as
|
||||||
|
# a colored chip so operators can bulk-triage by error class. Set
|
||||||
|
# by _update_source_health alongside last_error; cleared on 'ok'.
|
||||||
|
error_type: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True)
|
||||||
check_interval_override: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
check_interval_override: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
consecutive_failures: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
consecutive_failures: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||||
|
|
||||||
|
# alembic 0031: sticky deep-scan budget. When > 0, the next N download
|
||||||
|
# runs use gallery-dl's full-walk config (skip: True + 1800s timeout);
|
||||||
|
# when 0, runs use tick mode (skip: "exit:20" + 870s, exits early once
|
||||||
|
# 20 contiguous archived items are seen). Auto-decrements per run, with
|
||||||
|
# an auto-reset to 0 on clean exit + zero downloads (queue drained).
|
||||||
|
backfill_runs_remaining: Mapped[int] = mapped_column(
|
||||||
|
Integer, nullable=False, default=0, server_default="0",
|
||||||
|
)
|
||||||
|
|
||||||
artist = relationship("Artist", back_populates="sources")
|
artist = relationship("Artist", back_populates="sources")
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
"""TagAlias — maps a model's (name, category) prediction to the operator's
|
"""TagAlias — maps a model's (name, category) prediction to the operator's
|
||||||
canonical tag. Resolved at suggestion-read time so raw predictions stay
|
canonical tag. Resolved at suggestion-read time so the raw predictions stored
|
||||||
unmolested in image_record.tagger_predictions.
|
in image_prediction stay unmolested.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|||||||
@@ -22,7 +22,11 @@ class TagAllowlist(Base):
|
|||||||
tag_id: Mapped[int] = mapped_column(
|
tag_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True
|
ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True
|
||||||
)
|
)
|
||||||
min_confidence: Mapped[float] = mapped_column(Float, nullable=False, default=0.95)
|
# Default auto-apply threshold for a newly-accepted tag. 0.90 (lowered from
|
||||||
|
# 0.95 on operator evidence 2026-06-07: 0.95 was too strict and skipped
|
||||||
|
# confident-enough applications). Per-tag value is still tunable in the
|
||||||
|
# allowlist table; existing rows keep whatever they were stored with.
|
||||||
|
min_confidence: Mapped[float] = mapped_column(Float, nullable=False, default=0.90)
|
||||||
added_at: Mapped[datetime] = mapped_column(
|
added_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -16,9 +16,45 @@ log = logging.getLogger(__name__)
|
|||||||
|
|
||||||
ARCHIVE_EXTS = {".zip", ".cbz", ".rar", ".7z"}
|
ARCHIVE_EXTS = {".zip", ".cbz", ".rar", ".7z"}
|
||||||
|
|
||||||
|
# Magic-byte signatures, so an archive with a mangled / extension-less filename
|
||||||
|
# is still recognised. Patreon attachment download URLs sanitize to names like
|
||||||
|
# `01_https___www.patreon.com_media-u_v3_131083093`, whose `Path.suffix` is junk
|
||||||
|
# (`.com_media-u_v3_131083093`), never `.zip` — an extension-only gate filed
|
||||||
|
# those as opaque PostAttachments and NEVER extracted them (operator-flagged
|
||||||
|
# 2026-06-06). Detection is by extension first (cheap), then header sniff.
|
||||||
|
_RAR_MAGIC = b"Rar!\x1a\x07"
|
||||||
|
_7Z_MAGIC = b"7z\xbc\xaf\x27\x1c"
|
||||||
|
|
||||||
|
|
||||||
|
def detect_archive_format(path: Path) -> str | None:
|
||||||
|
"""Return "zip" | "rar" | "7z" for an archive, else None.
|
||||||
|
|
||||||
|
Trusts a known extension first, then falls back to magic-byte sniffing so a
|
||||||
|
mis-named or extension-less archive is still handled. (zip covers .cbz too.)
|
||||||
|
"""
|
||||||
|
ext = Path(path).suffix.lower()
|
||||||
|
if ext in (".zip", ".cbz"):
|
||||||
|
return "zip"
|
||||||
|
if ext == ".rar":
|
||||||
|
return "rar"
|
||||||
|
if ext == ".7z":
|
||||||
|
return "7z"
|
||||||
|
try:
|
||||||
|
if zipfile.is_zipfile(path):
|
||||||
|
return "zip"
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
head = fh.read(8)
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
if head.startswith(_RAR_MAGIC):
|
||||||
|
return "rar"
|
||||||
|
if head.startswith(_7Z_MAGIC):
|
||||||
|
return "7z"
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def is_archive(path: Path) -> bool:
|
def is_archive(path: Path) -> bool:
|
||||||
return Path(path).suffix.lower() in ARCHIVE_EXTS
|
return detect_archive_format(path) is not None
|
||||||
|
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
@@ -32,16 +68,16 @@ def extract_archive(path: Path):
|
|||||||
members: list[tuple[str, Path]] = []
|
members: list[tuple[str, Path]] = []
|
||||||
try:
|
try:
|
||||||
try:
|
try:
|
||||||
ext = Path(path).suffix.lower()
|
fmt = detect_archive_format(path)
|
||||||
if ext in (".zip", ".cbz"):
|
if fmt == "zip":
|
||||||
with zipfile.ZipFile(path) as zf:
|
with zipfile.ZipFile(path) as zf:
|
||||||
zf.extractall(base)
|
zf.extractall(base)
|
||||||
elif ext == ".rar":
|
elif fmt == "rar":
|
||||||
import rarfile
|
import rarfile
|
||||||
|
|
||||||
with rarfile.RarFile(path) as rf:
|
with rarfile.RarFile(path) as rf:
|
||||||
rf.extractall(base)
|
rf.extractall(base)
|
||||||
elif ext == ".7z":
|
elif fmt == "7z":
|
||||||
import py7zr
|
import py7zr
|
||||||
|
|
||||||
with py7zr.SevenZipFile(path, "r") as zf:
|
with py7zr.SevenZipFile(path, "r") as zf:
|
||||||
|
|||||||
@@ -13,10 +13,10 @@ from __future__ import annotations
|
|||||||
import base64
|
import base64
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
|
|
||||||
from sqlalchemy import and_, exists, func, or_, select
|
from sqlalchemy import and_, case, exists, func, or_, select
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from ..models import Artist, ImageRecord, Source
|
from ..models import Artist, ArtistVisit, ImageRecord, Source
|
||||||
from .gallery_service import thumbnail_url
|
from .gallery_service import thumbnail_url
|
||||||
|
|
||||||
_SEP = "|"
|
_SEP = "|"
|
||||||
@@ -58,9 +58,27 @@ class ArtistDirectoryService:
|
|||||||
raise ValueError("limit must be between 1 and 200")
|
raise ValueError("limit must be between 1 and 200")
|
||||||
|
|
||||||
count_col = func.count(ImageRecord.id).label("image_count")
|
count_col = func.count(ImageRecord.id).label("image_count")
|
||||||
|
# Unseen = images imported since the artist's last_viewed_at.
|
||||||
|
# NULL last_viewed_at (artist created before alembic 0034 seed
|
||||||
|
# or before find_or_create autoseed) defensively counts as
|
||||||
|
# "never visited" → all images unseen. Single grouped query, no
|
||||||
|
# N+1.
|
||||||
|
unseen_col = func.count(
|
||||||
|
case(
|
||||||
|
(
|
||||||
|
or_(
|
||||||
|
ArtistVisit.last_viewed_at.is_(None),
|
||||||
|
ImageRecord.created_at > ArtistVisit.last_viewed_at,
|
||||||
|
),
|
||||||
|
ImageRecord.id,
|
||||||
|
),
|
||||||
|
else_=None,
|
||||||
|
)
|
||||||
|
).label("unseen_count")
|
||||||
stmt = (
|
stmt = (
|
||||||
select(Artist, count_col)
|
select(Artist, count_col, unseen_col)
|
||||||
.outerjoin(ImageRecord, ImageRecord.artist_id == Artist.id)
|
.outerjoin(ImageRecord, ImageRecord.artist_id == Artist.id)
|
||||||
|
.outerjoin(ArtistVisit, ArtistVisit.artist_id == Artist.id)
|
||||||
.group_by(Artist.id)
|
.group_by(Artist.id)
|
||||||
)
|
)
|
||||||
if q:
|
if q:
|
||||||
@@ -94,7 +112,7 @@ class ArtistDirectoryService:
|
|||||||
next_cursor = _encode(last_artist.name, last_artist.id)
|
next_cursor = _encode(last_artist.name, last_artist.id)
|
||||||
rows = rows[:limit]
|
rows = rows[:limit]
|
||||||
|
|
||||||
artist_ids = [a.id for a, _ in rows]
|
artist_ids = [a.id for a, _, _ in rows]
|
||||||
previews = await self._previews(artist_ids)
|
previews = await self._previews(artist_ids)
|
||||||
|
|
||||||
cards = [
|
cards = [
|
||||||
@@ -104,9 +122,10 @@ class ArtistDirectoryService:
|
|||||||
"slug": artist.slug,
|
"slug": artist.slug,
|
||||||
"is_subscription": bool(artist.is_subscription),
|
"is_subscription": bool(artist.is_subscription),
|
||||||
"image_count": int(image_count),
|
"image_count": int(image_count),
|
||||||
|
"unseen_count": int(unseen_count),
|
||||||
"preview_thumbnails": previews.get(artist.id, []),
|
"preview_thumbnails": previews.get(artist.id, []),
|
||||||
}
|
}
|
||||||
for artist, image_count in rows
|
for artist, image_count, unseen_count in rows
|
||||||
]
|
]
|
||||||
return DirectoryPage(cards=cards, next_cursor=next_cursor)
|
return DirectoryPage(cards=cards, next_cursor=next_cursor)
|
||||||
|
|
||||||
@@ -127,17 +146,20 @@ class ArtistDirectoryService:
|
|||||||
ImageRecord.artist_id.label("artist_id"),
|
ImageRecord.artist_id.label("artist_id"),
|
||||||
ImageRecord.sha256.label("sha256"),
|
ImageRecord.sha256.label("sha256"),
|
||||||
ImageRecord.mime.label("mime"),
|
ImageRecord.mime.label("mime"),
|
||||||
|
ImageRecord.thumbnail_path.label("thumbnail_path"),
|
||||||
rn,
|
rn,
|
||||||
)
|
)
|
||||||
.where(ImageRecord.artist_id.in_(artist_ids))
|
.where(ImageRecord.artist_id.in_(artist_ids))
|
||||||
.subquery()
|
.subquery()
|
||||||
)
|
)
|
||||||
stmt = (
|
stmt = (
|
||||||
select(sub.c.artist_id, sub.c.sha256, sub.c.mime)
|
select(
|
||||||
|
sub.c.artist_id, sub.c.sha256, sub.c.mime, sub.c.thumbnail_path,
|
||||||
|
)
|
||||||
.where(sub.c.rn <= _PREVIEW_COUNT)
|
.where(sub.c.rn <= _PREVIEW_COUNT)
|
||||||
.order_by(sub.c.artist_id, sub.c.rn)
|
.order_by(sub.c.artist_id, sub.c.rn)
|
||||||
)
|
)
|
||||||
out: dict[int, list[str]] = {}
|
out: dict[int, list[str]] = {}
|
||||||
for aid, sha, mime in (await self.session.execute(stmt)).all():
|
for aid, sha, mime, tp in (await self.session.execute(stmt)).all():
|
||||||
out.setdefault(aid, []).append(thumbnail_url(sha, mime))
|
out.setdefault(aid, []).append(thumbnail_url(tp, sha, mime))
|
||||||
return out
|
return out
|
||||||
|
|||||||
@@ -9,11 +9,12 @@ Dates come from Post.post_date via ImageProvenance.post_id.
|
|||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
|
|
||||||
from sqlalchemy import and_, case, func, or_, select
|
from sqlalchemy import and_, case, func, or_, select
|
||||||
from sqlalchemy.exc import IntegrityError
|
from sqlalchemy.dialects.postgresql import insert as pg_insert
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from ..models import (
|
from ..models import (
|
||||||
Artist,
|
Artist,
|
||||||
|
ArtistVisit,
|
||||||
ImageProvenance,
|
ImageProvenance,
|
||||||
ImageRecord,
|
ImageRecord,
|
||||||
Post,
|
Post,
|
||||||
@@ -22,7 +23,9 @@ from ..models import (
|
|||||||
)
|
)
|
||||||
from ..models.tag import image_tag
|
from ..models.tag import image_tag
|
||||||
from ..utils.slug import slugify
|
from ..utils.slug import slugify
|
||||||
from .gallery_service import decode_cursor, encode_cursor, thumbnail_url
|
from .db_helpers import get_or_create
|
||||||
|
from .gallery_service import thumbnail_url
|
||||||
|
from .pagination import decode_cursor, encode_cursor
|
||||||
|
|
||||||
|
|
||||||
@dataclass(frozen=True)
|
@dataclass(frozen=True)
|
||||||
@@ -58,13 +61,17 @@ class ArtistService:
|
|||||||
)
|
)
|
||||||
).scalar_one()
|
).scalar_one()
|
||||||
|
|
||||||
|
# Posts under this artist that have at least one image attached.
|
||||||
|
# Use Post.artist_id (alembic 0030) for the artist filter; keep
|
||||||
|
# the ImageProvenance JOIN so date bounds reflect only image-
|
||||||
|
# bearing posts (matches the original semantic). NULL-source
|
||||||
|
# posts now surface too.
|
||||||
date_row = (
|
date_row = (
|
||||||
await self.session.execute(
|
await self.session.execute(
|
||||||
select(func.min(Post.post_date), func.max(Post.post_date))
|
select(func.min(Post.post_date), func.max(Post.post_date))
|
||||||
.select_from(Post)
|
.select_from(Post)
|
||||||
.join(ImageProvenance, ImageProvenance.post_id == Post.id)
|
.join(ImageProvenance, ImageProvenance.post_id == Post.id)
|
||||||
.join(Source, Source.id == ImageProvenance.source_id)
|
.where(Post.artist_id == aid)
|
||||||
.where(Source.artist_id == aid)
|
|
||||||
)
|
)
|
||||||
).first()
|
).first()
|
||||||
dmin, dmax = date_row if date_row else (None, None)
|
dmin, dmax = date_row if date_row else (None, None)
|
||||||
@@ -98,14 +105,14 @@ class ArtistService:
|
|||||||
)
|
)
|
||||||
).all()
|
).all()
|
||||||
|
|
||||||
|
# Same Post.artist_id direct filter — counts NULL-source posts too.
|
||||||
month = func.date_trunc("month", Post.post_date).label("m")
|
month = func.date_trunc("month", Post.post_date).label("m")
|
||||||
activity = (
|
activity = (
|
||||||
await self.session.execute(
|
await self.session.execute(
|
||||||
select(month, func.count(func.distinct(ImageProvenance.image_record_id)))
|
select(month, func.count(func.distinct(ImageProvenance.image_record_id)))
|
||||||
.select_from(Post)
|
.select_from(Post)
|
||||||
.join(ImageProvenance, ImageProvenance.post_id == Post.id)
|
.join(ImageProvenance, ImageProvenance.post_id == Post.id)
|
||||||
.join(Source, Source.id == ImageProvenance.source_id)
|
.where(and_(Post.artist_id == aid, Post.post_date.isnot(None)))
|
||||||
.where(and_(Source.artist_id == aid, Post.post_date.isnot(None)))
|
|
||||||
.group_by(month)
|
.group_by(month)
|
||||||
.order_by(month)
|
.order_by(month)
|
||||||
)
|
)
|
||||||
@@ -114,12 +121,16 @@ class ArtistService:
|
|||||||
post_count = (
|
post_count = (
|
||||||
await self.session.execute(
|
await self.session.execute(
|
||||||
select(func.count(func.distinct(Post.id)))
|
select(func.count(func.distinct(Post.id)))
|
||||||
.select_from(Post)
|
.where(Post.artist_id == aid)
|
||||||
.join(Source, Source.id == Post.source_id)
|
|
||||||
.where(Source.artist_id == aid)
|
|
||||||
)
|
)
|
||||||
).scalar_one()
|
).scalar_one()
|
||||||
|
|
||||||
|
# Mark this artist as "visited now"; the returned count is what
|
||||||
|
# the operator should see in the banner ("N new since last
|
||||||
|
# visit"). Done LAST so the read aggregates above all see the
|
||||||
|
# pre-visit state (cosmetic — none depend on visit data).
|
||||||
|
unseen_at_visit = await self._mark_visited_returning_unseen(aid)
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"id": artist.id,
|
"id": artist.id,
|
||||||
"name": artist.name,
|
"name": artist.name,
|
||||||
@@ -127,6 +138,7 @@ class ArtistService:
|
|||||||
"is_subscription": bool(artist.is_subscription),
|
"is_subscription": bool(artist.is_subscription),
|
||||||
"image_count": int(image_count),
|
"image_count": int(image_count),
|
||||||
"post_count": int(post_count),
|
"post_count": int(post_count),
|
||||||
|
"unseen_count_at_visit": unseen_at_visit,
|
||||||
"date_range": {
|
"date_range": {
|
||||||
"min": dmin.isoformat() if dmin else None,
|
"min": dmin.isoformat() if dmin else None,
|
||||||
"max": dmax.isoformat() if dmax else None,
|
"max": dmax.isoformat() if dmax else None,
|
||||||
@@ -155,6 +167,39 @@ class ArtistService:
|
|||||||
],
|
],
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async def _mark_visited_returning_unseen(self, artist_id: int) -> int:
|
||||||
|
"""Read pre-visit `last_viewed_at`, count images added since,
|
||||||
|
then upsert `last_viewed_at = NOW()`. Returns the count BEFORE
|
||||||
|
the upsert so the banner has data to render.
|
||||||
|
|
||||||
|
Postgres UPSERT (`ON CONFLICT DO UPDATE`) keeps the write
|
||||||
|
atomic — no SELECT-then-INSERT race per
|
||||||
|
`reference_scalar_one_or_none_duplicates`.
|
||||||
|
"""
|
||||||
|
prev = (
|
||||||
|
await self.session.execute(
|
||||||
|
select(ArtistVisit.last_viewed_at).where(
|
||||||
|
ArtistVisit.artist_id == artist_id
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).scalar_one_or_none()
|
||||||
|
|
||||||
|
count_stmt = select(func.count(ImageRecord.id)).where(
|
||||||
|
ImageRecord.artist_id == artist_id
|
||||||
|
)
|
||||||
|
if prev is not None:
|
||||||
|
count_stmt = count_stmt.where(ImageRecord.created_at > prev)
|
||||||
|
unseen = (await self.session.execute(count_stmt)).scalar_one()
|
||||||
|
|
||||||
|
upsert = pg_insert(ArtistVisit.__table__).values(artist_id=artist_id)
|
||||||
|
upsert = upsert.on_conflict_do_update(
|
||||||
|
index_elements=["artist_id"],
|
||||||
|
set_={"last_viewed_at": func.now()},
|
||||||
|
)
|
||||||
|
await self.session.execute(upsert)
|
||||||
|
await self.session.commit()
|
||||||
|
return int(unseen)
|
||||||
|
|
||||||
async def images(
|
async def images(
|
||||||
self, slug: str, cursor: str | None, limit: int = 60
|
self, slug: str, cursor: str | None, limit: int = 60
|
||||||
) -> ArtistImagesPage | None:
|
) -> ArtistImagesPage | None:
|
||||||
@@ -198,7 +243,7 @@ class ArtistService:
|
|||||||
"mime": r.mime,
|
"mime": r.mime,
|
||||||
"width": r.width,
|
"width": r.width,
|
||||||
"height": r.height,
|
"height": r.height,
|
||||||
"thumbnail_url": thumbnail_url(r.sha256, r.mime),
|
"thumbnail_url": thumbnail_url(r.thumbnail_path, r.sha256, r.mime),
|
||||||
}
|
}
|
||||||
for r in rows
|
for r in rows
|
||||||
],
|
],
|
||||||
@@ -206,30 +251,37 @@ class ArtistService:
|
|||||||
)
|
)
|
||||||
|
|
||||||
async def find_or_create(self, name: str) -> tuple[Artist, bool]:
|
async def find_or_create(self, name: str) -> tuple[Artist, bool]:
|
||||||
"""Return (artist, created). Slug-keyed; idempotent under races."""
|
"""Return (artist, created). Slug-keyed; idempotent under races via the
|
||||||
|
shared race-safe db_helpers.get_or_create (savepoint + IntegrityError
|
||||||
|
recovery). A new artist also seeds an ArtistVisit so the directory's
|
||||||
|
`+N new` badge starts at 0.
|
||||||
|
"""
|
||||||
cleaned = (name or "").strip()
|
cleaned = (name or "").strip()
|
||||||
if not cleaned:
|
if not cleaned:
|
||||||
raise ValueError("artist name must not be empty")
|
raise ValueError("artist name must not be empty")
|
||||||
slug = slugify(cleaned)
|
slug = slugify(cleaned)
|
||||||
|
|
||||||
existing = (await self.session.execute(
|
select_existing = select(Artist).where(Artist.slug == slug)
|
||||||
select(Artist).where(Artist.slug == slug)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
return existing, False
|
|
||||||
|
|
||||||
artist = Artist(name=cleaned, slug=slug)
|
async def _create() -> Artist:
|
||||||
self.session.add(artist)
|
artist = Artist(name=cleaned, slug=slug)
|
||||||
try:
|
self.session.add(artist)
|
||||||
await self.session.flush()
|
await self.session.flush()
|
||||||
except IntegrityError:
|
# New artist starts "caught up" — seed ArtistVisit so the
|
||||||
await self.session.rollback()
|
# directory's `+N new` badge stays at 0 until real new
|
||||||
existing = (await self.session.execute(
|
# content arrives. Without this, the unseen-count query
|
||||||
select(Artist).where(Artist.slug == slug)
|
# treats NULL last_viewed_at as "never visited" and would
|
||||||
)).scalar_one()
|
# count every image imported in the same session.
|
||||||
return existing, False
|
self.session.add(ArtistVisit(artist_id=artist.id))
|
||||||
await self.session.commit()
|
await self.session.flush()
|
||||||
return artist, True
|
return artist
|
||||||
|
|
||||||
|
artist, created = await get_or_create(
|
||||||
|
self.session, select_existing, _create
|
||||||
|
)
|
||||||
|
if created:
|
||||||
|
await self.session.commit()
|
||||||
|
return artist, created
|
||||||
|
|
||||||
async def autocomplete(self, prefix: str, limit: int = 20) -> list[Artist]:
|
async def autocomplete(self, prefix: str, limit: int = 20) -> list[Artist]:
|
||||||
cleaned = (prefix or "").strip()
|
cleaned = (prefix or "").strip()
|
||||||
|
|||||||
@@ -15,17 +15,55 @@ lifecycle + soft/hard time limits + retention bookkeeping.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
import subprocess
|
import subprocess
|
||||||
|
import tempfile
|
||||||
from datetime import UTC, datetime
|
from datetime import UTC, datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
_BACKUPS_DIRNAME = "_backups"
|
_BACKUPS_DIRNAME = "_backups"
|
||||||
|
|
||||||
# Subprocess-level guardrails BEYOND the Celery soft_time_limit. The
|
# Subprocess-level guardrails BEYOND the Celery soft_time_limit. The Celery
|
||||||
# Celery soft limit signals the Python process; subprocess.Popen in a
|
# soft limit signals the Python process; subprocess.Popen in a blocking syscall
|
||||||
# blocking syscall ignores that signal. These bound the worst case.
|
# ignores that signal, so these bound the worst case directly. Each sits just
|
||||||
_DB_SUBPROCESS_TIMEOUT_S = 12 * 60 # 12 min (Celery soft is 10 min)
|
# UNDER its task's Celery soft_time_limit so the bounded-kill (_run_bounded) is
|
||||||
_IMAGES_SUBPROCESS_TIMEOUT_S = 7 * 60 * 60 # 7 hr (Celery soft is 6 hr)
|
# the primary guard and fires cleanly before Celery's soft/hard limits — which
|
||||||
|
# matters because an NFS D-state hang defeats even Celery's SIGKILL (the failure
|
||||||
|
# that wedged the maintenance lane for hours, #739).
|
||||||
|
# backup_db_task: soft=1800s / hard=2100s → 1700s
|
||||||
|
# backup_images_task: soft=21600s / hard=23400s → 21000s
|
||||||
|
_DB_SUBPROCESS_TIMEOUT_S = 1700 # ~28 min, under the 30-min DB soft limit
|
||||||
|
_IMAGES_SUBPROCESS_TIMEOUT_S = 21000 # ~5.8 hr, under the 6-hr images soft limit
|
||||||
|
# Grace after SIGKILL to reap the child. If it can't be reaped in this window
|
||||||
|
# (an uninterruptible NFS D-state — the failure mode that wedged the
|
||||||
|
# concurrency-1 maintenance lane for hours, operator-flagged 2026-06-07), we
|
||||||
|
# STOP waiting and fail fast, freeing the worker slot. The orphan is reaped by
|
||||||
|
# the OS once its blocking syscall clears.
|
||||||
|
_KILL_REAP_GRACE_S = 10
|
||||||
|
|
||||||
|
|
||||||
|
def _run_bounded(cmd: list[str], timeout: int) -> None:
|
||||||
|
"""subprocess.run(check=True, timeout) whose reaper can't itself hang.
|
||||||
|
|
||||||
|
subprocess.run's timeout path SIGKILLs the child then blocks in wait() to
|
||||||
|
reap it — but a process stuck in uninterruptible I/O (NFS) can't be reaped,
|
||||||
|
so wait() blocks for hours. Here we bound the post-kill reap and re-raise
|
||||||
|
TimeoutExpired regardless, so the caller fails fast instead of wedging."""
|
||||||
|
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||||
|
try:
|
||||||
|
out, err = proc.communicate(timeout=timeout)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
proc.kill()
|
||||||
|
try:
|
||||||
|
proc.communicate(timeout=_KILL_REAP_GRACE_S)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
pass # unkillable (D-state) — abandon the reap, fail fast
|
||||||
|
raise
|
||||||
|
if proc.returncode != 0:
|
||||||
|
raise subprocess.CalledProcessError(
|
||||||
|
proc.returncode, cmd, output=out, stderr=err
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _libpq_url(sa_url: str) -> str:
|
def _libpq_url(sa_url: str) -> str:
|
||||||
@@ -83,15 +121,29 @@ def backup_db(
|
|||||||
to persist into BackupRun. Raises on subprocess failure."""
|
to persist into BackupRun. Raises on subprocess failure."""
|
||||||
ts = _now_ts()
|
ts = _now_ts()
|
||||||
out_dir = _backups_dir(images_root)
|
out_dir = _backups_dir(images_root)
|
||||||
sql_path = out_dir / f"fc_db_{ts}.sql"
|
# Custom format (-Fc): compressed (much smaller on NFS) and restored with
|
||||||
subprocess.run(
|
# pg_restore. The .dump extension marks it as non-SQL. The BackupRun field
|
||||||
[
|
# is still named sql_path — it's just "the db artifact path".
|
||||||
"pg_dump", "--no-owner", "--no-acl",
|
sql_path = out_dir / f"fc_db_{ts}.dump"
|
||||||
"-f", str(sql_path), _libpq_url(db_url),
|
# Dump to LOCAL disk first, then move the finished file to the (NFS) backups
|
||||||
],
|
# dir. pg_dump's long phase is then a DB-socket wait + local writes — both
|
||||||
capture_output=True, check=True,
|
# killable — instead of an NFS write that can hang uninterruptibly. Only the
|
||||||
timeout=_DB_SUBPROCESS_TIMEOUT_S,
|
# final move touches NFS, and it's a bounded single-file step.
|
||||||
)
|
fd, tmp_name = tempfile.mkstemp(prefix="fc_db_", suffix=".dump")
|
||||||
|
os.close(fd)
|
||||||
|
tmp_path = Path(tmp_name)
|
||||||
|
try:
|
||||||
|
_run_bounded(
|
||||||
|
[
|
||||||
|
"pg_dump", "--no-owner", "--no-acl", "-Fc",
|
||||||
|
"-f", str(tmp_path), _libpq_url(db_url),
|
||||||
|
],
|
||||||
|
_DB_SUBPROCESS_TIMEOUT_S,
|
||||||
|
)
|
||||||
|
shutil.move(str(tmp_path), str(sql_path))
|
||||||
|
finally:
|
||||||
|
if tmp_path.exists():
|
||||||
|
tmp_path.unlink(missing_ok=True)
|
||||||
manifest_path = _write_manifest(
|
manifest_path = _write_manifest(
|
||||||
out_dir, kind="db", ts=ts, tag=tag, triggered_by=triggered_by,
|
out_dir, kind="db", ts=ts, tag=tag, triggered_by=triggered_by,
|
||||||
artifact_path=sql_path,
|
artifact_path=sql_path,
|
||||||
@@ -114,15 +166,17 @@ def backup_images(
|
|||||||
ts = _now_ts()
|
ts = _now_ts()
|
||||||
out_dir = _backups_dir(images_root)
|
out_dir = _backups_dir(images_root)
|
||||||
tar_path = out_dir / f"fc_images_{ts}.tar.zst"
|
tar_path = out_dir / f"fc_images_{ts}.tar.zst"
|
||||||
subprocess.run(
|
# No local-temp here (the archive is hundreds of GB — it can't stage in
|
||||||
|
# /tmp), but bounded-kill still applies so a tar wedged on NFS fails fast
|
||||||
|
# rather than holding the lane for hours.
|
||||||
|
_run_bounded(
|
||||||
[
|
[
|
||||||
"tar", "--zstd", "-cf", str(tar_path),
|
"tar", "--zstd", "-cf", str(tar_path),
|
||||||
"-C", str(images_root.parent), images_root.name,
|
"-C", str(images_root.parent), images_root.name,
|
||||||
f"--exclude={images_root.name}/_backups",
|
f"--exclude={images_root.name}/_backups",
|
||||||
f"--exclude={images_root.name}/_quarantine",
|
f"--exclude={images_root.name}/_quarantine",
|
||||||
],
|
],
|
||||||
capture_output=True, check=True,
|
_IMAGES_SUBPROCESS_TIMEOUT_S,
|
||||||
timeout=_IMAGES_SUBPROCESS_TIMEOUT_S,
|
|
||||||
)
|
)
|
||||||
manifest_path = _write_manifest(
|
manifest_path = _write_manifest(
|
||||||
out_dir, kind="images", ts=ts, tag=tag, triggered_by=triggered_by,
|
out_dir, kind="images", ts=ts, tag=tag, triggered_by=triggered_by,
|
||||||
@@ -139,8 +193,8 @@ def backup_images(
|
|||||||
|
|
||||||
|
|
||||||
def restore_db(*, db_url: str, sql_path: Path) -> None:
|
def restore_db(*, db_url: str, sql_path: Path) -> None:
|
||||||
"""Wipe public schema, then load from .sql. Raises on subprocess
|
"""Wipe public schema, then load from the custom-format dump. Raises on
|
||||||
failure; partial-restore state is the caller's concern."""
|
subprocess failure; partial-restore state is the caller's concern."""
|
||||||
libpq = _libpq_url(db_url)
|
libpq = _libpq_url(db_url)
|
||||||
subprocess.run(
|
subprocess.run(
|
||||||
[
|
[
|
||||||
@@ -149,8 +203,9 @@ def restore_db(*, db_url: str, sql_path: Path) -> None:
|
|||||||
],
|
],
|
||||||
capture_output=True, check=True, timeout=120,
|
capture_output=True, check=True, timeout=120,
|
||||||
)
|
)
|
||||||
|
# Custom-format (-Fc) dumps are restored with pg_restore, not psql.
|
||||||
subprocess.run(
|
subprocess.run(
|
||||||
["psql", libpq, "-f", str(sql_path)],
|
["pg_restore", "--no-owner", "--no-acl", "-d", libpq, str(sql_path)],
|
||||||
capture_output=True, check=True,
|
capture_output=True, check=True,
|
||||||
timeout=_DB_SUBPROCESS_TIMEOUT_S,
|
timeout=_DB_SUBPROCESS_TIMEOUT_S,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -6,23 +6,35 @@ HTTP handlers (small ops) and from Celery tasks in
|
|||||||
backend.app.tasks.admin (long ops).
|
backend.app.tasks.admin (long ops).
|
||||||
|
|
||||||
This module is the PERMANENT home of artist-cascade + image-unlink
|
This module is the PERMANENT home of artist-cascade + image-unlink
|
||||||
logic. The legacy copy at backend/app/services/migrators/cleanup.py
|
logic. (The legacy migrators/cleanup.py copy was removed with the rest of
|
||||||
stays in place until FC-3j; FC-3j will replace its body with thin
|
the one-and-done GS/IR migration tooling.)
|
||||||
re-exports from this module and then delete the wrapper.
|
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from datetime import UTC, datetime
|
import logging
|
||||||
|
import time
|
||||||
|
from datetime import UTC, datetime, timedelta
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
from sqlalchemy import func, select, update
|
from sqlalchemy import func, or_, select, update
|
||||||
from sqlalchemy.orm import Session
|
from sqlalchemy.orm import Session
|
||||||
|
|
||||||
from ..models import Artist, ImageRecord, LibraryAuditRun, Tag
|
from ..models import (
|
||||||
|
Artist,
|
||||||
|
ImageProvenance,
|
||||||
|
ImageRecord,
|
||||||
|
LibraryAuditRun,
|
||||||
|
Post,
|
||||||
|
PostAttachment,
|
||||||
|
Tag,
|
||||||
|
)
|
||||||
|
from ..models.series_chapter import SeriesChapter
|
||||||
from ..models.series_page import SeriesPage
|
from ..models.series_page import SeriesPage
|
||||||
from ..models.tag import image_tag
|
from ..models.tag import image_tag
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def project_artist_cascade(session: Session, *, slug: str) -> dict:
|
def project_artist_cascade(session: Session, *, slug: str) -> dict:
|
||||||
"""Read-only projection of what delete_artist_cascade would touch.
|
"""Read-only projection of what delete_artist_cascade would touch.
|
||||||
@@ -135,26 +147,48 @@ def count_tag_associations(session: Session, *, tag_id: int) -> int:
|
|||||||
).scalar_one()
|
).scalar_one()
|
||||||
|
|
||||||
|
|
||||||
def find_unused_tags(
|
def _unused_tag_conditions() -> list:
|
||||||
session: Session, *, limit: int | None = None,
|
"""The WHERE conditions that define an 'unused' tag — the SINGLE source of
|
||||||
) -> list[Tag]:
|
truth shared by find_unused_tags (preview sample), the dry-run count, AND the
|
||||||
"""Tags with no image_tag rows AND no series_page rows.
|
live delete, so the preview can NEVER diverge from what the delete removes.
|
||||||
|
|
||||||
Sorted by name. Used by both dry-run preview and the live prune.
|
A tag is "unused" iff it has zero references across ALL the ways a tag can be
|
||||||
A tag is "unused" iff it has zero rows in image_tag AND zero rows
|
in use:
|
||||||
in series_page (so we don't accidentally prune a series tag that
|
- image_tag (applied to an image)
|
||||||
happens to have no images yet).
|
- series_page (a series tag with ordered pages)
|
||||||
|
- series_chapter (a series tag with chapters but no pages yet)
|
||||||
|
- tag.fandom_id (a fandom referenced by a character)
|
||||||
|
|
||||||
|
The fandom check is essential: fandom tags are NEVER applied to images — a
|
||||||
|
character carries its fandom via fandom_id — so without it every assigned
|
||||||
|
fandom looks "unused", and the FK is ondelete=SET NULL, so deleting one
|
||||||
|
silently strips the fandom off all its characters. The delete had only the
|
||||||
|
first two checks while the preview sample had all four, so the preview showed
|
||||||
|
a safe list but the delete removed every fandom anyway (operator-flagged
|
||||||
|
2026-06-08). Defining the predicate once makes that impossible.
|
||||||
"""
|
"""
|
||||||
used_via_image_tag = select(image_tag.c.tag_id).distinct()
|
used_via_image_tag = select(image_tag.c.tag_id).distinct()
|
||||||
used_via_series = select(SeriesPage.series_tag_id).where(
|
used_via_series = select(SeriesPage.series_tag_id).where(
|
||||||
SeriesPage.series_tag_id.is_not(None)
|
SeriesPage.series_tag_id.is_not(None)
|
||||||
).distinct()
|
).distinct()
|
||||||
stmt = (
|
used_via_chapter = select(SeriesChapter.series_tag_id).distinct()
|
||||||
select(Tag)
|
used_via_fandom = select(Tag.fandom_id).where(
|
||||||
.where(Tag.id.not_in(used_via_image_tag))
|
Tag.fandom_id.is_not(None)
|
||||||
.where(Tag.id.not_in(used_via_series))
|
).distinct()
|
||||||
.order_by(Tag.name)
|
return [
|
||||||
)
|
Tag.id.not_in(used_via_image_tag),
|
||||||
|
Tag.id.not_in(used_via_series),
|
||||||
|
Tag.id.not_in(used_via_chapter),
|
||||||
|
Tag.id.not_in(used_via_fandom),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def find_unused_tags(
|
||||||
|
session: Session, *, limit: int | None = None,
|
||||||
|
) -> list[Tag]:
|
||||||
|
"""Tags genuinely referenced by nothing — safe to sweep. Sorted by name.
|
||||||
|
Shares its predicate with the live prune via _unused_tag_conditions()."""
|
||||||
|
stmt = select(Tag).where(*_unused_tag_conditions()).order_by(Tag.name)
|
||||||
if limit is not None:
|
if limit is not None:
|
||||||
stmt = stmt.limit(limit)
|
stmt = stmt.limit(limit)
|
||||||
return list(session.execute(stmt).scalars().all())
|
return list(session.execute(stmt).scalars().all())
|
||||||
@@ -188,10 +222,14 @@ def unlink_image_files(
|
|||||||
out["thumbnail"] = True
|
out["thumbnail"] = True
|
||||||
except OSError:
|
except OSError:
|
||||||
out["thumbnail"] = False
|
out["thumbnail"] = False
|
||||||
# Convention thumbs dir — try all extensions; missing OK.
|
# Convention thumbs dir — try both extensions thumbnailer writes
|
||||||
|
# (.jpg for opaque, .png for alpha). `.webp` used to be in this
|
||||||
|
# tuple but the thumbnailer never writes it (operator-flagged in
|
||||||
|
# the 2026-06-02 audit) — keep the tuple aligned with what
|
||||||
|
# actually lands on disk.
|
||||||
if image.sha256:
|
if image.sha256:
|
||||||
bucket = image.sha256[:3]
|
bucket = image.sha256[:3]
|
||||||
for ext in ("jpg", "png", "webp"):
|
for ext in ("jpg", "png"):
|
||||||
try:
|
try:
|
||||||
(images_root / "thumbs" / bucket / f"{image.sha256}.{ext}").unlink(
|
(images_root / "thumbs" / bucket / f"{image.sha256}.{ext}").unlink(
|
||||||
missing_ok=True,
|
missing_ok=True,
|
||||||
@@ -355,18 +393,230 @@ def prune_unused_tags(session: Session, *, dry_run: bool = False) -> dict:
|
|||||||
Returns:
|
Returns:
|
||||||
dry_run=True: {"count": N, "sample_names": [first 50]}
|
dry_run=True: {"count": N, "sample_names": [first 50]}
|
||||||
dry_run=False: {"deleted": N, "sample_names": [first 50]}
|
dry_run=False: {"deleted": N, "sample_names": [first 50]}
|
||||||
|
|
||||||
|
Implementation note: the previous SELECT-ids → DELETE-WHERE-IN
|
||||||
|
pattern was vulnerable to the psycopg 65535-parameter ceiling on
|
||||||
|
libraries with tag explosions. The live delete runs a single DELETE
|
||||||
|
with the SAME predicate (_unused_tag_conditions) the preview uses, so
|
||||||
|
the row count scales without binding every id as a parameter AND the
|
||||||
|
delete can never remove a tag the preview deemed safe. Audit 2026-06-02;
|
||||||
|
predicate unified 2026-06-08.
|
||||||
"""
|
"""
|
||||||
unused = find_unused_tags(session)
|
conditions = _unused_tag_conditions()
|
||||||
sample = [t.name for t in unused[:50]]
|
sample_rows = find_unused_tags(session, limit=50)
|
||||||
|
sample = [t.name for t in sample_rows]
|
||||||
if dry_run:
|
if dry_run:
|
||||||
return {"count": len(unused), "sample_names": sample}
|
count = session.execute(
|
||||||
ids = [t.id for t in unused]
|
select(func.count()).select_from(Tag).where(*conditions)
|
||||||
if ids:
|
).scalar_one()
|
||||||
session.execute(
|
return {"count": count, "sample_names": sample}
|
||||||
Tag.__table__.delete().where(Tag.id.in_(ids))
|
result = session.execute(Tag.__table__.delete().where(*conditions))
|
||||||
|
session.commit()
|
||||||
|
return {"deleted": result.rowcount or 0, "sample_names": sample}
|
||||||
|
|
||||||
|
|
||||||
|
def _bare_post_conditions() -> list:
|
||||||
|
"""The WHERE conditions that define a 'bare' post — the SINGLE source of truth
|
||||||
|
shared by find_bare_posts (preview sample), the dry-run count, AND the live
|
||||||
|
delete, so the preview can NEVER diverge from what the delete removes
|
||||||
|
([[feedback_preview_apply_parity]]).
|
||||||
|
|
||||||
|
A post is "bare" iff NOTHING is attached to it across every way content links
|
||||||
|
to a post:
|
||||||
|
- image_record.primary_post_id (a post's own canonical images)
|
||||||
|
- image_provenance.post_id (cross-posted/duplicate images linked here)
|
||||||
|
- post_attachment.post_id (preserved non-art files)
|
||||||
|
|
||||||
|
These are exactly the shells the empty-post flood produced: the native Patreon
|
||||||
|
ingester synthesized a Post per walked post, but when its only content was a
|
||||||
|
duplicate image/attachment that linked to an EARLIER post, the new post was
|
||||||
|
left with none of the three (operator-flagged 2026-06-08). Every FK to post.id
|
||||||
|
is SET NULL or CASCADE, so deleting a bare post is non-destructive by
|
||||||
|
construction — there is nothing pointing at it to orphan. Must run AFTER the
|
||||||
|
provenance-render fix so a post that DOES have a hidden provenance link is
|
||||||
|
spared, not deleted.
|
||||||
|
"""
|
||||||
|
has_primary = select(ImageRecord.id).where(
|
||||||
|
ImageRecord.primary_post_id == Post.id
|
||||||
|
)
|
||||||
|
has_provenance = select(ImageProvenance.id).where(
|
||||||
|
ImageProvenance.post_id == Post.id
|
||||||
|
)
|
||||||
|
has_attachment = select(PostAttachment.id).where(
|
||||||
|
PostAttachment.post_id == Post.id
|
||||||
|
)
|
||||||
|
return [
|
||||||
|
~has_primary.exists(),
|
||||||
|
~has_provenance.exists(),
|
||||||
|
~has_attachment.exists(),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def find_bare_posts(
|
||||||
|
session: Session, *, limit: int | None = None,
|
||||||
|
) -> list[Post]:
|
||||||
|
"""Posts with zero linked images (primary OR provenance) AND zero
|
||||||
|
attachments — safe to sweep. Sorted by id. Shares its predicate with the
|
||||||
|
live prune via _bare_post_conditions()."""
|
||||||
|
stmt = select(Post).where(*_bare_post_conditions()).order_by(Post.id)
|
||||||
|
if limit is not None:
|
||||||
|
stmt = stmt.limit(limit)
|
||||||
|
return list(session.execute(stmt).scalars().all())
|
||||||
|
|
||||||
|
|
||||||
|
def _bare_post_label(post: Post) -> str:
|
||||||
|
"""Human label for the preview sample — mirrors the feed's fallback title."""
|
||||||
|
if post.post_title:
|
||||||
|
return post.post_title
|
||||||
|
return f"Post {post.external_post_id or post.id}"
|
||||||
|
|
||||||
|
|
||||||
|
def prune_bare_posts(session: Session, *, dry_run: bool = False) -> dict:
|
||||||
|
"""Find posts with no images and no attachments and (unless dry_run) delete
|
||||||
|
them.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dry_run=True: {"count": N, "sample_names": [first 50]}
|
||||||
|
dry_run=False: {"deleted": N, "sample_names": [first 50]}
|
||||||
|
|
||||||
|
The live delete runs a single DELETE with the SAME predicate
|
||||||
|
(_bare_post_conditions) the preview uses, so the row count scales without
|
||||||
|
binding every id as a parameter ([[reference_psycopg_65535_param_ceiling]])
|
||||||
|
AND the delete can never remove a post the preview deemed kept.
|
||||||
|
"""
|
||||||
|
conditions = _bare_post_conditions()
|
||||||
|
sample_rows = find_bare_posts(session, limit=50)
|
||||||
|
sample = [_bare_post_label(p) for p in sample_rows]
|
||||||
|
if dry_run:
|
||||||
|
count = session.execute(
|
||||||
|
select(func.count()).select_from(Post).where(*conditions)
|
||||||
|
).scalar_one()
|
||||||
|
return {"count": count, "sample_names": sample}
|
||||||
|
result = session.execute(Post.__table__.delete().where(*conditions))
|
||||||
|
session.commit()
|
||||||
|
return {"deleted": result.rowcount or 0, "sample_names": sample}
|
||||||
|
|
||||||
|
|
||||||
|
# Legacy tags FC no longer uses, in two shapes:
|
||||||
|
# (1) kinds the tag input never produces — archive/post/artist.
|
||||||
|
# provenance (post grouping) + archive membership are their own
|
||||||
|
# systems now, and artists are first-class Artist/Source rows.
|
||||||
|
# meta/rating were already hard-deleted by alembic 0023.
|
||||||
|
# (2) name prefixes from IR kinds FC never adopted — `source:*`.
|
||||||
|
# ImageRepo had a `source` kind; FC's enum doesn't, so ir_ingest
|
||||||
|
# fell those back to `general` (kind=general, name="source:patreon"
|
||||||
|
# etc.). They can't be caught by kind, so we match the name prefix.
|
||||||
|
PURGEABLE_TAG_KINDS = ("archive", "post", "artist")
|
||||||
|
LEGACY_NAME_PREFIXES = ("source:",)
|
||||||
|
|
||||||
|
|
||||||
|
def _legacy_tag_predicate():
|
||||||
|
name_clauses = [Tag.name.like(f"{p}%") for p in LEGACY_NAME_PREFIXES]
|
||||||
|
return or_(Tag.kind.in_(PURGEABLE_TAG_KINDS), *name_clauses)
|
||||||
|
|
||||||
|
|
||||||
|
def purge_legacy_tags(session: Session, *, dry_run: bool = False) -> dict:
|
||||||
|
"""Count (dry_run) or delete legacy IR-migration tags: archive/post/
|
||||||
|
artist-kind tags PLUS general tags whose name matches a legacy
|
||||||
|
prefix (source:*).
|
||||||
|
|
||||||
|
CASCADE on image_tag / tag_alias / tag_allowlist /
|
||||||
|
tag_reference_embedding / tag_suggestion_rejection / series_page
|
||||||
|
clears the related rows on the parent DELETE.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
{"by_kind": {kind: count, ...}, # kind-matched rows
|
||||||
|
"by_prefix": {"source:*": count}, # name-prefix-matched rows
|
||||||
|
"count": total, "sample_names": [first 50],
|
||||||
|
and on live runs "deleted": total}
|
||||||
|
"""
|
||||||
|
predicate = _legacy_tag_predicate()
|
||||||
|
rows = session.execute(
|
||||||
|
select(Tag.id, Tag.name, Tag.kind).where(predicate)
|
||||||
|
).all()
|
||||||
|
by_kind: dict[str, int] = {}
|
||||||
|
by_prefix: dict[str, int] = {}
|
||||||
|
for _id, name, kind in rows:
|
||||||
|
# Classify by name-prefix first so a source:* row counts once,
|
||||||
|
# under the prefix bucket, regardless of its (general) kind.
|
||||||
|
matched_prefix = next(
|
||||||
|
(p for p in LEGACY_NAME_PREFIXES if name.startswith(p)), None,
|
||||||
)
|
)
|
||||||
|
if matched_prefix is not None:
|
||||||
|
label = f"{matched_prefix}*"
|
||||||
|
by_prefix[label] = by_prefix.get(label, 0) + 1
|
||||||
|
else:
|
||||||
|
key = kind.value if hasattr(kind, "value") else str(kind)
|
||||||
|
by_kind[key] = by_kind.get(key, 0) + 1
|
||||||
|
sample = [name for _id, name, _kind in rows[:50]]
|
||||||
|
total = len(rows)
|
||||||
|
result = {
|
||||||
|
"by_kind": by_kind, "by_prefix": by_prefix,
|
||||||
|
"count": total, "sample_names": sample,
|
||||||
|
}
|
||||||
|
if dry_run:
|
||||||
|
return result
|
||||||
|
if total:
|
||||||
|
session.execute(Tag.__table__.delete().where(predicate))
|
||||||
session.commit()
|
session.commit()
|
||||||
return {"deleted": len(ids), "sample_names": sample}
|
result["deleted"] = total
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
# The Camie-suggestable CONTENT vocabulary. "Reset content tagging" wipes
|
||||||
|
# these so the operator can re-tag from scratch via auto-suggest. fandom +
|
||||||
|
# series (and series_page ordering) are deliberately NOT here — they're kept.
|
||||||
|
RESETTABLE_TAG_KINDS = ("general", "character")
|
||||||
|
|
||||||
|
|
||||||
|
def reset_content_tagging(session: Session, *, dry_run: bool = False) -> dict:
|
||||||
|
"""Count (dry_run) or DELETE every general + character tag so the operator
|
||||||
|
can re-tag from scratch via the Camie auto-suggest.
|
||||||
|
|
||||||
|
PRESERVED: fandom + series tags and their series_page ordering, plus every
|
||||||
|
image's image_prediction rows (untouched) so suggestions
|
||||||
|
repopulate immediately. CASCADE on image_tag / tag_alias / tag_allowlist /
|
||||||
|
tag_reference_embedding / tag_suggestion_rejection clears each deleted
|
||||||
|
tag's applications + metadata. Tag.fandom_id is SET NULL, so deleting
|
||||||
|
character tags never touches the fandom rows. Irreversible except via DB
|
||||||
|
backup restore.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
{"by_kind": {"general": N, "character": M},
|
||||||
|
"count": total tags,
|
||||||
|
"applications": image_tag rows that will be / were removed,
|
||||||
|
"sample_names": [first 50],
|
||||||
|
and on live runs "deleted": total}
|
||||||
|
"""
|
||||||
|
predicate = Tag.kind.in_(RESETTABLE_TAG_KINDS)
|
||||||
|
rows = session.execute(
|
||||||
|
select(Tag.id, Tag.name, Tag.kind).where(predicate)
|
||||||
|
).all()
|
||||||
|
by_kind: dict[str, int] = {}
|
||||||
|
for _id, _name, kind in rows:
|
||||||
|
key = kind.value if hasattr(kind, "value") else str(kind)
|
||||||
|
by_kind[key] = by_kind.get(key, 0) + 1
|
||||||
|
# Headline impact: applications (image_tag rows) that vanish via cascade.
|
||||||
|
applications = session.execute(
|
||||||
|
select(func.count())
|
||||||
|
.select_from(image_tag)
|
||||||
|
.where(image_tag.c.tag_id.in_(select(Tag.id).where(predicate)))
|
||||||
|
).scalar_one()
|
||||||
|
sample = [name for _id, name, _kind in rows[:50]]
|
||||||
|
total = len(rows)
|
||||||
|
result = {
|
||||||
|
"by_kind": by_kind,
|
||||||
|
"count": total,
|
||||||
|
"applications": applications,
|
||||||
|
"sample_names": sample,
|
||||||
|
}
|
||||||
|
if dry_run:
|
||||||
|
return result
|
||||||
|
if total:
|
||||||
|
session.execute(Tag.__table__.delete().where(predicate))
|
||||||
|
session.commit()
|
||||||
|
result["deleted"] = total
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -435,6 +685,9 @@ class ConfirmTokenMismatch(Exception):
|
|||||||
_VALID_RULES = ("transparency", "single_color")
|
_VALID_RULES = ("transparency", "single_color")
|
||||||
|
|
||||||
|
|
||||||
|
_AUDIT_GUARD_THRESHOLD_MINUTES = 135 # matches LIBRARY_AUDIT_STALL_THRESHOLD_MINUTES
|
||||||
|
|
||||||
|
|
||||||
def start_audit_run(
|
def start_audit_run(
|
||||||
session: Session, *, rule: str, params: dict[str, Any],
|
session: Session, *, rule: str, params: dict[str, Any],
|
||||||
) -> int:
|
) -> int:
|
||||||
@@ -442,11 +695,21 @@ def start_audit_run(
|
|||||||
scan_library_for_rule Celery task. Returns the new audit_id.
|
scan_library_for_rule Celery task. Returns the new audit_id.
|
||||||
|
|
||||||
Concurrent-runs guard: raises AuditAlreadyRunning if any audit_run
|
Concurrent-runs guard: raises AuditAlreadyRunning if any audit_run
|
||||||
has status='running'. Operator must cancel or wait."""
|
has status='running' AND started recently. Audit 2026-06-02 made
|
||||||
|
the guard age-aware: a SIGKILL'd run leaves a row in 'running'
|
||||||
|
that the recovery sweep flips on its next pass (~5 min), but a
|
||||||
|
fresh start_audit_run between the SIGKILL and the sweep would
|
||||||
|
previously block forever. Past the threshold, treat the running
|
||||||
|
row as stale and let the sweep clean it up — the new run still
|
||||||
|
gets to start.
|
||||||
|
"""
|
||||||
if rule not in _VALID_RULES:
|
if rule not in _VALID_RULES:
|
||||||
raise ValueError(f"unknown rule {rule!r}; expected one of {_VALID_RULES}")
|
raise ValueError(f"unknown rule {rule!r}; expected one of {_VALID_RULES}")
|
||||||
|
cutoff = datetime.now(UTC) - timedelta(minutes=_AUDIT_GUARD_THRESHOLD_MINUTES)
|
||||||
existing = session.execute(
|
existing = session.execute(
|
||||||
select(LibraryAuditRun.id).where(LibraryAuditRun.status == "running")
|
select(LibraryAuditRun.id)
|
||||||
|
.where(LibraryAuditRun.status == "running")
|
||||||
|
.where(LibraryAuditRun.started_at >= cutoff)
|
||||||
).scalar_one_or_none()
|
).scalar_one_or_none()
|
||||||
if existing is not None:
|
if existing is not None:
|
||||||
raise AuditAlreadyRunning(existing)
|
raise AuditAlreadyRunning(existing)
|
||||||
@@ -510,3 +773,170 @@ def cancel_audit_run(session: Session, *, audit_id: int) -> None:
|
|||||||
.where(LibraryAuditRun.status == "running")
|
.where(LibraryAuditRun.status == "running")
|
||||||
.values(status="cancelled", finished_at=datetime.now(UTC))
|
.values(status="cancelled", finished_at=datetime.now(UTC))
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# -- archive-attachment re-extraction (#713 part 2) ------------------------
|
||||||
|
|
||||||
|
_ARCHIVE_EXT_FOR_FORMAT = {"zip": ".zip", "rar": ".rar", "7z": ".7z"}
|
||||||
|
|
||||||
|
|
||||||
|
def _reextract_archive_to_post(
|
||||||
|
importer, archive_path: Path, post, source_row, artist, images_root: Path,
|
||||||
|
) -> list[int]:
|
||||||
|
"""Extract one stored archive and link its members to `post`.
|
||||||
|
|
||||||
|
The stored attachment has no adjacent sidecar (it lives in the sha-addressed
|
||||||
|
attachment store). Stage a copy + a reconstructed sidecar UNDER the artist's
|
||||||
|
library dir (`images_root/<slug>/<platform>/<post>/`) — the importer
|
||||||
|
re-derives the artist from the path AND copies members relative to it, so the
|
||||||
|
members land in the real library and resolve to the right artist — then re-run
|
||||||
|
`attach_in_place`: the archive extracts and `find_or_create_post` re-attaches
|
||||||
|
the members to the SAME Post (source_id + external_post_id). Removes only the
|
||||||
|
staged archive + sidecar afterward; the imported member files stay. Returns
|
||||||
|
the new member image ids.
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import shutil
|
||||||
|
|
||||||
|
from .archive_extractor import detect_archive_format
|
||||||
|
|
||||||
|
fmt = detect_archive_format(archive_path)
|
||||||
|
ext = _ARCHIVE_EXT_FOR_FORMAT.get(fmt or "", ".zip")
|
||||||
|
platform = source_row.platform if source_row is not None else "imported"
|
||||||
|
sidecar = {
|
||||||
|
"category": source_row.platform if source_row is not None
|
||||||
|
else (post.raw_metadata or {}).get("category"),
|
||||||
|
"id": post.external_post_id,
|
||||||
|
"title": post.post_title or "",
|
||||||
|
"content": post.description or "",
|
||||||
|
"published_at": post.post_date.isoformat() if post.post_date else None,
|
||||||
|
"url": post.post_url,
|
||||||
|
}
|
||||||
|
work = images_root / artist.slug / platform / str(post.external_post_id)
|
||||||
|
work.mkdir(parents=True, exist_ok=True)
|
||||||
|
staged = work / f"archive{ext}" # clean ext → is_archive + find_sidecar
|
||||||
|
sidecar_path = staged.with_suffix(".json")
|
||||||
|
try:
|
||||||
|
shutil.copy2(archive_path, staged)
|
||||||
|
sidecar_path.write_text(json.dumps(sidecar))
|
||||||
|
res = importer.attach_in_place(staged, artist=artist, source=source_row)
|
||||||
|
return list(res.member_image_ids or [])
|
||||||
|
finally:
|
||||||
|
# Drop only the staged archive + sidecar; the extracted member files
|
||||||
|
# were copied into the library alongside them and must stay.
|
||||||
|
staged.unlink(missing_ok=True)
|
||||||
|
sidecar_path.unlink(missing_ok=True)
|
||||||
|
|
||||||
|
|
||||||
|
def reextract_archive_attachments(
|
||||||
|
session: Session,
|
||||||
|
*,
|
||||||
|
images_root: Path,
|
||||||
|
time_budget_seconds: float | None = None,
|
||||||
|
after_id: int = 0,
|
||||||
|
) -> dict:
|
||||||
|
"""Re-process existing PostAttachments that are ACTUALLY archives but were
|
||||||
|
filed opaquely before #713 part 1 (extension-only is_archive missed mangled /
|
||||||
|
extension-less Patreon attachment names). For each: extract the members,
|
||||||
|
import them, and link them to the attachment's post.
|
||||||
|
|
||||||
|
Idempotent — members dedupe by sha256, the archive dedupes by sha — so it's
|
||||||
|
safe to run repeatedly. Returns a summary dict for task_run.metadata.
|
||||||
|
|
||||||
|
Time-boxed + resumable: scans PostAttachments in ascending id order starting
|
||||||
|
after ``after_id``. When ``time_budget_seconds`` elapses, stops and reports
|
||||||
|
``partial=True`` + ``resume_after_id`` (the last scanned id) so the task can
|
||||||
|
re-enqueue itself and continue — a large archive back-catalog can't run the
|
||||||
|
task into the Celery time limit or hog the maintenance lane. A bare re-run
|
||||||
|
(after_id=0) would never advance because an already-extracted archive is
|
||||||
|
still an archive on disk, so the cursor is what guarantees forward progress.
|
||||||
|
"""
|
||||||
|
from ..models import ImportSettings, Post, PostAttachment, Source
|
||||||
|
from ..tasks.ml import tag_and_embed
|
||||||
|
from ..tasks.thumbnail import generate_thumbnail
|
||||||
|
from .archive_extractor import is_archive
|
||||||
|
from .importer import Importer
|
||||||
|
from .thumbnailer import Thumbnailer
|
||||||
|
|
||||||
|
summary = {
|
||||||
|
"scanned": 0, "archives": 0, "members_imported": 0,
|
||||||
|
"posts_touched": 0, "skipped_no_post": 0, "skipped_no_artist": 0,
|
||||||
|
"errors": 0, "partial": False, "resume_after_id": after_id,
|
||||||
|
}
|
||||||
|
settings = ImportSettings.load_sync(session)
|
||||||
|
importer = Importer(
|
||||||
|
session=session, images_root=images_root, import_root=images_root,
|
||||||
|
thumbnailer=Thumbnailer(images_root=images_root), settings=settings,
|
||||||
|
)
|
||||||
|
|
||||||
|
attachments = session.execute(
|
||||||
|
select(PostAttachment)
|
||||||
|
.where(PostAttachment.id > after_id)
|
||||||
|
.order_by(PostAttachment.id)
|
||||||
|
).scalars().all()
|
||||||
|
enqueue_ids: list[int] = []
|
||||||
|
start = time.monotonic()
|
||||||
|
for att in attachments:
|
||||||
|
summary["scanned"] += 1
|
||||||
|
summary["resume_after_id"] = att.id
|
||||||
|
stored = Path(att.path)
|
||||||
|
try:
|
||||||
|
if not stored.is_file() or not is_archive(stored):
|
||||||
|
continue
|
||||||
|
except OSError:
|
||||||
|
continue
|
||||||
|
summary["archives"] += 1
|
||||||
|
if att.post_id is None:
|
||||||
|
summary["skipped_no_post"] += 1
|
||||||
|
continue
|
||||||
|
post = session.get(Post, att.post_id)
|
||||||
|
if post is None:
|
||||||
|
summary["skipped_no_post"] += 1
|
||||||
|
continue
|
||||||
|
artist = session.get(Artist, att.artist_id) if att.artist_id else None
|
||||||
|
if artist is None and post.artist_id:
|
||||||
|
artist = session.get(Artist, post.artist_id)
|
||||||
|
if artist is None or not artist.slug:
|
||||||
|
# The importer re-derives the artist from the staged path, so we need
|
||||||
|
# a real artist+slug to anchor under. (Shouldn't happen for
|
||||||
|
# subscription posts; skip rather than orphan the members.)
|
||||||
|
summary["skipped_no_artist"] += 1
|
||||||
|
continue
|
||||||
|
source_row = session.get(Source, post.source_id) if post.source_id else None
|
||||||
|
try:
|
||||||
|
ids = _reextract_archive_to_post(
|
||||||
|
importer, stored, post, source_row, artist, images_root,
|
||||||
|
)
|
||||||
|
session.commit()
|
||||||
|
except Exception as exc: # one bad archive must not strand the rest
|
||||||
|
session.rollback()
|
||||||
|
summary["errors"] += 1
|
||||||
|
log.warning("re-extract failed for attachment %s: %s", att.id, exc)
|
||||||
|
continue
|
||||||
|
if ids:
|
||||||
|
summary["members_imported"] += len(ids)
|
||||||
|
summary["posts_touched"] += 1
|
||||||
|
enqueue_ids.extend(ids)
|
||||||
|
|
||||||
|
# Time-box the chunk. resume_after_id already points at this attachment,
|
||||||
|
# so the next run starts strictly after it. Checked after the commit so a
|
||||||
|
# half-extracted archive never straddles the boundary.
|
||||||
|
if (
|
||||||
|
time_budget_seconds is not None
|
||||||
|
and time.monotonic() - start >= time_budget_seconds
|
||||||
|
):
|
||||||
|
summary["partial"] = True
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
# Loop ran to exhaustion — nothing left to resume.
|
||||||
|
summary["partial"] = False
|
||||||
|
|
||||||
|
# Thumbnails + ML for the newly-imported members (best-effort; off the
|
||||||
|
# critical path — a Redis hiccup must not fail the whole re-extract).
|
||||||
|
for img_id in enqueue_ids:
|
||||||
|
try:
|
||||||
|
generate_thumbnail.delay(img_id)
|
||||||
|
tag_and_embed.delay(img_id)
|
||||||
|
except Exception as exc:
|
||||||
|
log.warning("re-extract enqueue failed for image %s: %s", img_id, exc)
|
||||||
|
return summary
|
||||||
|
|||||||
@@ -1,40 +1,82 @@
|
|||||||
"""Fernet-based encryption for credential blobs.
|
"""Fernet-based encryption for credential blobs.
|
||||||
|
|
||||||
The key is a single 32-byte value (urlsafe-base64-encoded; what
|
The key is a single 32-byte value (urlsafe-base64-encoded; what
|
||||||
Fernet.generate_key produces) stored at a fixed path inside the
|
Fernet.generate_key produces) stored at /images/secrets/credential_key.b64
|
||||||
images/data root. Created on first boot if absent; mode 0600. No KDF
|
(mode 0600, parent dir 0700). The 2026-06-02 audit caught a silent
|
||||||
needed — the file contents are already maximum-entropy random bytes.
|
key-regeneration path: on a partial disaster restore where the DB was
|
||||||
|
restored but the secrets dir was lost, the old `_load_or_create_key`
|
||||||
|
would mint a fresh key with no log, producing a working-looking system
|
||||||
|
where every authenticated download failed AUTH_ERROR until the operator
|
||||||
|
re-uploaded every credential by hand. Now the constructor refuses to
|
||||||
|
auto-generate unless either:
|
||||||
|
|
||||||
Operator backup procedure must include this file alongside the rest
|
* the caller explicitly passes `bootstrap_ok=True` (tests, scripts), or
|
||||||
of /images/ — losing it makes existing encrypted_blob rows
|
* the env var `CURATOR_BOOTSTRAP_NEW_KEY=1` is set (operator opt-in
|
||||||
undecryptable (recovery = delete the rows and re-upload).
|
during first-time setup).
|
||||||
|
|
||||||
|
Otherwise it raises `MissingCredentialKey` so the app fails fast at
|
||||||
|
startup and the operator can restore the key file from backup.
|
||||||
|
|
||||||
|
Operator backup procedure must include /images/secrets/ alongside the
|
||||||
|
rest of /images/ — losing the key file makes existing encrypted_blob
|
||||||
|
rows undecryptable (recovery = delete the rows and re-upload).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import logging
|
||||||
import os
|
import os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from cryptography.fernet import Fernet, InvalidToken
|
from cryptography.fernet import Fernet, InvalidToken
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
_BOOTSTRAP_ENV_VAR = "CURATOR_BOOTSTRAP_NEW_KEY"
|
||||||
|
|
||||||
|
|
||||||
class InvalidCredentialBlob(Exception):
|
class InvalidCredentialBlob(Exception):
|
||||||
"""Raised when decryption fails (wrong key, tampered blob, …)."""
|
"""Raised when decryption fails (wrong key, tampered blob, …)."""
|
||||||
|
|
||||||
|
|
||||||
|
class MissingCredentialKey(Exception):
|
||||||
|
"""The Fernet key file is missing AND the caller hasn't opted in to
|
||||||
|
generating a new one. Audit 2026-06-02: prevents silent key
|
||||||
|
regeneration on partial DB-restored / secrets-lost deployments.
|
||||||
|
Set CURATOR_BOOTSTRAP_NEW_KEY=1 for first-time setup, or restore the
|
||||||
|
key file from backup."""
|
||||||
|
|
||||||
|
|
||||||
class CredentialCrypto:
|
class CredentialCrypto:
|
||||||
"""Fernet encrypt/decrypt with an on-disk key file.
|
"""Fernet encrypt/decrypt with an on-disk key file.
|
||||||
|
|
||||||
Instantiate with a path; the file is created on first access and
|
Instantiate with a path; the file is loaded if present, or created
|
||||||
reused thereafter. Tests pass a tmp_path; production calls with
|
if absent AND the caller has opted in (bootstrap_ok=True or
|
||||||
|
CURATOR_BOOTSTRAP_NEW_KEY=1 env var). Production sites:
|
||||||
`IMAGES_ROOT / "secrets" / "credential_key.b64"`.
|
`IMAGES_ROOT / "secrets" / "credential_key.b64"`.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, key_path: Path):
|
def __init__(self, key_path: Path, *, bootstrap_ok: bool | None = None):
|
||||||
self._key_path = Path(key_path)
|
self._key_path = Path(key_path)
|
||||||
self._fernet = Fernet(self._load_or_create_key())
|
if bootstrap_ok is None:
|
||||||
|
bootstrap_ok = os.environ.get(_BOOTSTRAP_ENV_VAR) == "1"
|
||||||
|
self._fernet = Fernet(self._load_or_create_key(bootstrap_ok))
|
||||||
|
|
||||||
def _load_or_create_key(self) -> bytes:
|
def _load_or_create_key(self, bootstrap_ok: bool) -> bytes:
|
||||||
if self._key_path.exists():
|
if self._key_path.exists():
|
||||||
return self._key_path.read_bytes()
|
return self._key_path.read_bytes()
|
||||||
|
if not bootstrap_ok:
|
||||||
|
raise MissingCredentialKey(
|
||||||
|
f"Fernet key file not found at {self._key_path}. "
|
||||||
|
f"For first-time setup, set {_BOOTSTRAP_ENV_VAR}=1. "
|
||||||
|
f"If this is a restored instance, restore the key file "
|
||||||
|
f"from backup — generating a new one would make every "
|
||||||
|
f"existing Credential row undecryptable."
|
||||||
|
)
|
||||||
|
log.warning(
|
||||||
|
"Generating NEW Fernet credential key at %s. Any existing "
|
||||||
|
"encrypted_blob rows in the DB will be undecryptable — "
|
||||||
|
"re-upload each credential after this completes.",
|
||||||
|
self._key_path,
|
||||||
|
)
|
||||||
parent = self._key_path.parent
|
parent = self._key_path.parent
|
||||||
parent.mkdir(parents=True, exist_ok=True)
|
parent.mkdir(parents=True, exist_ok=True)
|
||||||
os.chmod(parent, 0o700)
|
os.chmod(parent, 0o700)
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ from __future__ import annotations
|
|||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from datetime import datetime
|
from datetime import UTC, datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
@@ -163,6 +163,19 @@ class CredentialService:
|
|||||||
return None
|
return None
|
||||||
return self.crypto.decrypt(row.encrypted_blob)
|
return self.crypto.decrypt(row.encrypted_blob)
|
||||||
|
|
||||||
|
async def mark_verified(self, platform: str) -> datetime | None:
|
||||||
|
"""Stamp last_verified=now after a successful verify. Returns the
|
||||||
|
timestamp, or None if the credential is gone."""
|
||||||
|
row = (await self.session.execute(
|
||||||
|
select(Credential).where(Credential.platform == platform)
|
||||||
|
)).scalar_one_or_none()
|
||||||
|
if row is None:
|
||||||
|
return None
|
||||||
|
ts = datetime.now(UTC)
|
||||||
|
row.last_verified = ts
|
||||||
|
await self.session.commit()
|
||||||
|
return ts
|
||||||
|
|
||||||
|
|
||||||
def _augment_cookies(platform: str, netscape: str) -> str:
|
def _augment_cookies(platform: str, netscape: str) -> str:
|
||||||
"""Delegate to the platform's `augment_cookies` hook if one is
|
"""Delegate to the platform's `augment_cookies` hook if one is
|
||||||
|
|||||||
@@ -0,0 +1,52 @@
|
|||||||
|
"""Shared DB-access helpers for the async services.
|
||||||
|
|
||||||
|
`get_or_create` centralizes the race-safe find-or-create dance — SELECT, then on
|
||||||
|
a miss a savepoint INSERT that recovers (NOT a full rollback) when a concurrent
|
||||||
|
worker inserted the same row first. It was hand-rolled identically in
|
||||||
|
ArtistService, TagService and ExtensionService; divergent copies are exactly how
|
||||||
|
the duplicate-row / race bugs in [[reference_scalar_one_or_none_duplicates]] crept
|
||||||
|
in, so it lives in one place now (DRY pattern sweep 2026-06-09).
|
||||||
|
|
||||||
|
Note: this is the ASYNC sibling of `Importer._get_or_create` (sync, used by the
|
||||||
|
filesystem-import path). The two can't share an implementation across the
|
||||||
|
sync/async boundary; the importer one stays as the lone sync consumer.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from collections.abc import Awaitable, Callable
|
||||||
|
|
||||||
|
from sqlalchemy import Select
|
||||||
|
from sqlalchemy.exc import IntegrityError
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
|
||||||
|
async def get_or_create[T](
|
||||||
|
session: AsyncSession,
|
||||||
|
select_stmt: Select,
|
||||||
|
factory: Callable[[], Awaitable[T]],
|
||||||
|
) -> tuple[T, bool]:
|
||||||
|
"""Race-safe find-or-create. Returns ``(row, created)``.
|
||||||
|
|
||||||
|
Run ``select_stmt`` (scalar_one_or_none); if a row exists, return it with
|
||||||
|
``created=False``. Otherwise open a SAVEPOINT and ``await factory()`` — which
|
||||||
|
must add its row(s), flush, and return the primary row. On ``IntegrityError``
|
||||||
|
(a concurrent worker inserted the same row first) roll back the SAVEPOINT —
|
||||||
|
NOT the outer transaction, which would lose the caller's surrounding work —
|
||||||
|
and re-run ``select_stmt`` (scalar_one) to return the row the other worker
|
||||||
|
created. The caller owns the outer commit.
|
||||||
|
|
||||||
|
A UNIQUE/partial-unique constraint matching ``select_stmt``'s predicate is
|
||||||
|
required for the recovery to trip; without it a duplicate slips through.
|
||||||
|
"""
|
||||||
|
existing = (await session.execute(select_stmt)).scalar_one_or_none()
|
||||||
|
if existing is not None:
|
||||||
|
return existing, False
|
||||||
|
sp = await session.begin_nested()
|
||||||
|
try:
|
||||||
|
row = await factory()
|
||||||
|
await sp.commit()
|
||||||
|
return row, True
|
||||||
|
except IntegrityError:
|
||||||
|
await sp.rollback()
|
||||||
|
return (await session.execute(select_stmt)).scalar_one(), False
|
||||||
@@ -0,0 +1,241 @@
|
|||||||
|
"""Platform → download-backend dispatch (one place that knows which platforms
|
||||||
|
are served by the native FC ingester vs. the gallery-dl subprocess).
|
||||||
|
|
||||||
|
gallery-dl wasn't built to be driven by an automated scheduler — no native
|
||||||
|
checkpoint/resume, no structured logs, per-file HEADs that dominate wall-clock.
|
||||||
|
The native ingester (services/patreon_ingester.py, plan #697) replaces it for
|
||||||
|
Patreon and is the path we grow as more platforms migrate. To keep that
|
||||||
|
migration DRY, every caller that has to behave differently per backend —
|
||||||
|
download routing, the credential-verify probe, cursor handling — asks THIS
|
||||||
|
module instead of testing ``platform == "patreon"`` inline. When a platform gets
|
||||||
|
a native ingester, it moves into ``NATIVE_INGESTER_PLATFORMS`` here and both the
|
||||||
|
download path and verify switch over together.
|
||||||
|
|
||||||
|
The backend surfaces share a UNIFORM signature so a caller invokes the same
|
||||||
|
function regardless of platform:
|
||||||
|
- verify_credential(...) → (ok: bool|None, message: str)
|
||||||
|
- (download stays in download_service for now; uses_native_ingester() is the
|
||||||
|
shared predicate it routes on, so the decision lives here too.)
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from .gallery_dl import DownloadResult, ErrorType
|
||||||
|
from .patreon_ingester import PatreonIngester
|
||||||
|
from .patreon_resolver import extract_vanity, resolve_campaign_id_for_source
|
||||||
|
|
||||||
|
# Platforms whose download + verify go through the native ingester rather than
|
||||||
|
# gallery-dl. gallery-dl still serves every other platform (subscribestar,
|
||||||
|
# hentaifoundry, discord, pixiv, deviantart) unchanged.
|
||||||
|
NATIVE_INGESTER_PLATFORMS = frozenset({"patreon"})
|
||||||
|
|
||||||
|
# Mirrors patreon_resolver._CAMPAIGNS_URL — surfaced in resolution-failure
|
||||||
|
# messages so the operator sees the exact lookup endpoint that was hit.
|
||||||
|
_CAMPAIGNS_API = "https://www.patreon.com/api/campaigns"
|
||||||
|
|
||||||
|
|
||||||
|
def uses_native_ingester(platform: str) -> bool:
|
||||||
|
"""True when `platform` is served by the native ingester (not gallery-dl).
|
||||||
|
The single predicate the download path and verify both route on."""
|
||||||
|
return platform in NATIVE_INGESTER_PLATFORMS
|
||||||
|
|
||||||
|
|
||||||
|
async def run_download(
|
||||||
|
*,
|
||||||
|
ctx: dict,
|
||||||
|
source_config,
|
||||||
|
skip_value: bool | str,
|
||||||
|
mode: str | None,
|
||||||
|
gdl,
|
||||||
|
sync_session_factory,
|
||||||
|
) -> tuple[DownloadResult, str | None]:
|
||||||
|
"""Uniform download across backends — the download counterpart to
|
||||||
|
`verify_source_credential`, so this module is the ONE place that knows how
|
||||||
|
each platform both downloads AND verifies (the seam that makes adding a
|
||||||
|
platform a bounded job).
|
||||||
|
|
||||||
|
Returns `(DownloadResult, resolved_campaign_id)`; `resolved_campaign_id` is
|
||||||
|
non-None only when a native vanity lookup ran this call (so phase 3 caches
|
||||||
|
it). Native platforms route through their ingester in `mode`
|
||||||
|
(tick/backfill/recovery); gallery-dl platforms run the subprocess. The caller
|
||||||
|
(download_service) prepares `source_config`/`skip_value`/`mode` from the
|
||||||
|
backfill state machine and owns phase 3.
|
||||||
|
"""
|
||||||
|
platform = ctx["platform"]
|
||||||
|
if uses_native_ingester(platform):
|
||||||
|
return await _run_native_ingester(
|
||||||
|
ctx, source_config, mode, gdl, sync_session_factory
|
||||||
|
)
|
||||||
|
result = await gdl.download(
|
||||||
|
url=ctx["url"],
|
||||||
|
artist_slug=ctx["artist_slug"],
|
||||||
|
platform=platform,
|
||||||
|
source_config=source_config,
|
||||||
|
cookies_path=ctx["cookies_path"],
|
||||||
|
auth_token=ctx["auth_token"],
|
||||||
|
skip_value=skip_value,
|
||||||
|
)
|
||||||
|
return result, None
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_native_ingester(
|
||||||
|
ctx: dict, source_config, mode: str | None, gdl, sync_session_factory,
|
||||||
|
) -> tuple[DownloadResult, str | None]:
|
||||||
|
"""Patreon (today the only native platform): resolve the campaign id, then run
|
||||||
|
the native ingester in a worker thread (it is sync requests/subprocess).
|
||||||
|
|
||||||
|
`resolved_campaign_id` is non-None only when we had to look it up from the
|
||||||
|
vanity URL this run, so phase 3 caches it the way the old gallery-dl retry
|
||||||
|
did. A campaign id we cannot resolve is a loud NOT_FOUND — never a silent
|
||||||
|
empty success.
|
||||||
|
"""
|
||||||
|
overrides = ctx["config_overrides"] or {}
|
||||||
|
campaign_id, resolved_campaign_id = await resolve_campaign_id_for_source(
|
||||||
|
ctx["url"], ctx["cookies_path"], overrides
|
||||||
|
)
|
||||||
|
if not campaign_id:
|
||||||
|
url = ctx["url"]
|
||||||
|
vanity = extract_vanity(url)
|
||||||
|
return (
|
||||||
|
DownloadResult(
|
||||||
|
success=False,
|
||||||
|
url=url,
|
||||||
|
artist_slug=ctx["artist_slug"],
|
||||||
|
platform="patreon",
|
||||||
|
error_type=ErrorType.NOT_FOUND,
|
||||||
|
error_message=(
|
||||||
|
f"Could not resolve Patreon campaign id. source_url={url!r}; "
|
||||||
|
f"vanity={vanity!r}; "
|
||||||
|
f"lookup=GET {_CAMPAIGNS_API}?filter[vanity]={vanity or ''} "
|
||||||
|
"(vanity lookup failed — cookies expired or creator moved?)"
|
||||||
|
),
|
||||||
|
),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Honor the operator's existing rate-limit knobs on the native path (plan
|
||||||
|
# #703): the global download_rate_limit_seconds (gallery-dl's `rate_limit`,
|
||||||
|
# here on the gdl service) paces media downloads; page fetches use the
|
||||||
|
# per-source sleep_request override, else `max(0.5, rate_limit/4)` — the same
|
||||||
|
# API-pacing default gallery-dl applied as its `sleep-request`.
|
||||||
|
rate_limit = gdl._rate_limit
|
||||||
|
request_sleep = (
|
||||||
|
source_config.sleep_request
|
||||||
|
if source_config.sleep_request is not None
|
||||||
|
else max(0.5, rate_limit / 4)
|
||||||
|
)
|
||||||
|
ingester = PatreonIngester(
|
||||||
|
images_root=gdl.images_root,
|
||||||
|
cookies_path=ctx["cookies_path"],
|
||||||
|
session_factory=sync_session_factory,
|
||||||
|
validate=gdl._validate_files,
|
||||||
|
rate_limit=rate_limit,
|
||||||
|
request_sleep=request_sleep,
|
||||||
|
)
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
dl_result = await loop.run_in_executor(
|
||||||
|
None,
|
||||||
|
lambda: ingester.run(
|
||||||
|
source_id=ctx["source_id"],
|
||||||
|
campaign_id=campaign_id,
|
||||||
|
artist_slug=ctx["artist_slug"],
|
||||||
|
url=ctx["url"],
|
||||||
|
mode=mode,
|
||||||
|
resume_cursor=source_config.resume_cursor,
|
||||||
|
time_budget_seconds=source_config.timeout,
|
||||||
|
posts_base=int(overrides.get("_backfill_posts", 0)),
|
||||||
|
# plan #709: live progress writes to this running event mid-walk.
|
||||||
|
event_id=ctx.get("event_id"),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
return dl_result, resolved_campaign_id
|
||||||
|
|
||||||
|
|
||||||
|
async def preview_source(
|
||||||
|
*,
|
||||||
|
platform: str,
|
||||||
|
url: str,
|
||||||
|
source_id: int,
|
||||||
|
config_overrides: dict | None,
|
||||||
|
cookies_path: str | None,
|
||||||
|
images_root: Path,
|
||||||
|
sync_session_factory,
|
||||||
|
page_limit: int = 3,
|
||||||
|
) -> dict:
|
||||||
|
"""Dry-run preview for a native platform (plan #708 B4): resolve the campaign
|
||||||
|
id, then walk a few pages counting media not already seen/dead — no download.
|
||||||
|
|
||||||
|
Returns the preview dict (total_new / posts_scanned / pages_scanned /
|
||||||
|
has_more / sample), or `{"error": msg}` on a resolve / auth / drift failure.
|
||||||
|
Native-only — the caller gates on `uses_native_ingester`.
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
from .patreon_client import PatreonAPIError
|
||||||
|
|
||||||
|
campaign_id, _ = await resolve_campaign_id_for_source(
|
||||||
|
url, cookies_path, config_overrides or {}
|
||||||
|
)
|
||||||
|
if not campaign_id:
|
||||||
|
vanity = extract_vanity(url)
|
||||||
|
return {
|
||||||
|
"error": (
|
||||||
|
f"Couldn't resolve the campaign id. source_url={url!r}; "
|
||||||
|
f"vanity={vanity!r}; lookup=GET {_CAMPAIGNS_API}?filter[vanity]={vanity or ''} "
|
||||||
|
"(cookies expired, or the creator moved/renamed?)."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
ingester = PatreonIngester(
|
||||||
|
images_root=images_root,
|
||||||
|
cookies_path=cookies_path,
|
||||||
|
session_factory=sync_session_factory,
|
||||||
|
)
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
try:
|
||||||
|
result = await loop.run_in_executor(
|
||||||
|
None,
|
||||||
|
lambda: ingester.preview(source_id, campaign_id, page_limit=page_limit),
|
||||||
|
)
|
||||||
|
except PatreonAPIError as exc:
|
||||||
|
return {"error": f"Couldn't preview: {exc}"}
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
async def verify_source_credential(
|
||||||
|
*,
|
||||||
|
platform: str,
|
||||||
|
url: str,
|
||||||
|
artist_slug: str,
|
||||||
|
config_overrides: dict | None,
|
||||||
|
cookies_path: str | None,
|
||||||
|
auth_token: str | None,
|
||||||
|
images_root: Path,
|
||||||
|
) -> tuple[bool | None, str]:
|
||||||
|
"""Uniform credential probe across backends. Returns `(ok, message)`:
|
||||||
|
True = authenticated, False = rejected, None = inconclusive (drift /
|
||||||
|
network / nothing to test). Callers don't branch on platform — they call
|
||||||
|
this and render the result.
|
||||||
|
"""
|
||||||
|
if uses_native_ingester(platform):
|
||||||
|
# Native ingester platforms verify via their own lightweight auth probe
|
||||||
|
# (resolve campaign id + one authenticated API page). Patreon today.
|
||||||
|
from .patreon_ingester import verify_patreon_credential
|
||||||
|
|
||||||
|
return await verify_patreon_credential(url, cookies_path, config_overrides)
|
||||||
|
|
||||||
|
# gallery-dl platforms: --simulate one item; the extractor errors before it
|
||||||
|
# can list if auth is bad.
|
||||||
|
from .gallery_dl import GalleryDLService, SourceConfig
|
||||||
|
|
||||||
|
gdl = GalleryDLService(images_root=images_root)
|
||||||
|
return await gdl.verify(
|
||||||
|
url=url,
|
||||||
|
artist_slug=artist_slug,
|
||||||
|
platform=platform,
|
||||||
|
source_config=SourceConfig.from_dict(config_overrides or {}),
|
||||||
|
cookies_path=cookies_path,
|
||||||
|
auth_token=auth_token,
|
||||||
|
)
|
||||||
@@ -13,7 +13,6 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
import logging
|
import logging
|
||||||
import re
|
|
||||||
from datetime import UTC, datetime
|
from datetime import UTC, datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
@@ -25,35 +24,25 @@ from sqlalchemy.orm import joinedload
|
|||||||
|
|
||||||
from ..models import Artist, DownloadEvent, Source
|
from ..models import Artist, DownloadEvent, Source
|
||||||
from .credential_service import CredentialService
|
from .credential_service import CredentialService
|
||||||
from .gallery_dl import GalleryDLService, SourceConfig
|
from .download_backends import run_download, uses_native_ingester
|
||||||
|
from .gallery_dl import (
|
||||||
|
BACKFILL_CHUNK_SECONDS,
|
||||||
|
BACKFILL_SKIP_VALUE,
|
||||||
|
TICK_SKIP_VALUE,
|
||||||
|
DownloadResult,
|
||||||
|
ErrorType,
|
||||||
|
GalleryDLService,
|
||||||
|
SourceConfig,
|
||||||
|
extract_errors_warnings,
|
||||||
|
truncate_log,
|
||||||
|
)
|
||||||
from .importer import Importer
|
from .importer import Importer
|
||||||
from .patreon_resolver import resolve_campaign_id
|
from .platforms import auth_type_for
|
||||||
|
from .scheduler_service import set_platform_cooldown
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
_PATREON_VANITY_RE = re.compile(
|
|
||||||
r"^https?://(?:www\.)?patreon\.com/(?:c/)?(?!id:)([^/?#]+)",
|
|
||||||
re.IGNORECASE,
|
|
||||||
)
|
|
||||||
_CAMPAIGN_ID_FAILURE_PATTERN = "failed to extract campaign id"
|
|
||||||
|
|
||||||
|
|
||||||
def _extract_patreon_vanity(url: str) -> str | None:
|
|
||||||
m = _PATREON_VANITY_RE.match(url)
|
|
||||||
return m.group(1) if m else None
|
|
||||||
|
|
||||||
|
|
||||||
def _looks_like_campaign_id_failure(stdout: str, stderr: str) -> bool:
|
|
||||||
return _CAMPAIGN_ID_FAILURE_PATTERN in f"{stdout}\n{stderr}".lower()
|
|
||||||
|
|
||||||
|
|
||||||
def _effective_url(platform: str, source_url: str, overrides: dict) -> str:
|
|
||||||
if platform == "patreon" and overrides.get("patreon_campaign_id"):
|
|
||||||
return f"https://www.patreon.com/id:{overrides['patreon_campaign_id']}"
|
|
||||||
return source_url
|
|
||||||
|
|
||||||
|
|
||||||
class DownloadService:
|
class DownloadService:
|
||||||
"""Async orchestrator. The Celery task runs `asyncio.run(svc.download_source(N))`.
|
"""Async orchestrator. The Celery task runs `asyncio.run(svc.download_source(N))`.
|
||||||
|
|
||||||
@@ -68,12 +57,18 @@ class DownloadService:
|
|||||||
gdl: GalleryDLService,
|
gdl: GalleryDLService,
|
||||||
importer: Importer,
|
importer: Importer,
|
||||||
cred_service: CredentialService,
|
cred_service: CredentialService,
|
||||||
|
sync_session_factory=None,
|
||||||
):
|
):
|
||||||
self.async_session = async_session
|
self.async_session = async_session
|
||||||
self.sync_session = sync_session
|
self.sync_session = sync_session
|
||||||
self.gdl = gdl
|
self.gdl = gdl
|
||||||
self.importer = importer
|
self.importer = importer
|
||||||
self.cred_service = cred_service
|
self.cred_service = cred_service
|
||||||
|
# Sync sessionmaker the native Patreon ingester opens SHORT-LIVED
|
||||||
|
# sessions from for its seen-ledger reads/writes (never one held across
|
||||||
|
# the multi-minute walk — see PatreonIngester). Only the patreon branch
|
||||||
|
# of phase 2 uses it; gallery-dl sources leave it None.
|
||||||
|
self.sync_session_factory = sync_session_factory
|
||||||
|
|
||||||
async def download_source(self, source_id: int) -> int:
|
async def download_source(self, source_id: int) -> int:
|
||||||
"""Returns DownloadEvent.id. Idempotent: in-flight events are returned as-is."""
|
"""Returns DownloadEvent.id. Idempotent: in-flight events are returned as-is."""
|
||||||
@@ -82,50 +77,101 @@ class DownloadService:
|
|||||||
return setup["event_id"]
|
return setup["event_id"]
|
||||||
ctx = setup
|
ctx = setup
|
||||||
|
|
||||||
|
# Release the phase-1 DB connections before the (up to ~19.5-min in
|
||||||
|
# backfill) gallery-dl subprocess. Held checked-out across that idle
|
||||||
|
# window, the asyncpg/psycopg connections get reaped by the server,
|
||||||
|
# and phase 3's first query then hits a dead socket
|
||||||
|
# (asyncpg ConnectionDoesNotExistError) → download_source autoretry →
|
||||||
|
# _phase1_setup's in-flight guard no-ops the retry → the event
|
||||||
|
# strands empty for the recovery sweep (Anduo #40014, 2026-06-04).
|
||||||
|
# pool_pre_ping can't help a *held* connection — it only validates on
|
||||||
|
# pool checkout. Closing returns them to the pool so phase 3 re-
|
||||||
|
# acquires a live one (the async task engine uses NullPool, the sync
|
||||||
|
# engine pre_ping + pool_recycle=300). This is what makes the
|
||||||
|
# "Phase 2 — no DB connection" contract in the class docstring true.
|
||||||
|
await self.async_session.close()
|
||||||
|
self.sync_session.close()
|
||||||
|
|
||||||
source_config = SourceConfig.from_dict(ctx["config_overrides"] or {})
|
source_config = SourceConfig.from_dict(ctx["config_overrides"] or {})
|
||||||
effective_url = _effective_url(
|
# Backfill mode (plan #693): the source's `_backfill_state == "running"`
|
||||||
ctx["platform"], ctx["url"], ctx["config_overrides"] or {}
|
# selects a time-boxed deep-walk chunk — skip: True (walk full history)
|
||||||
|
# + the BACKFILL_CHUNK_SECONDS budget, resuming from the cursor
|
||||||
|
# checkpoint (plan #689). State is the single source of truth; it stays
|
||||||
|
# "running" across ticks until the walk reaches the bottom (phase 3
|
||||||
|
# flips it to "complete"). Otherwise tick mode: exit gallery-dl after
|
||||||
|
# 20 contiguous archived items (skip: "exit:20" + the default 870s).
|
||||||
|
# Operator drives this via POST /api/sources/{id}/backfill {action}.
|
||||||
|
overrides = ctx["config_overrides"] or {}
|
||||||
|
in_backfill = overrides.get("_backfill_state") == "running"
|
||||||
|
# Recovery (plan #697) reuses the entire #693 backfill state machine —
|
||||||
|
# cursor checkpoint, time-boxed chunks, complete/stall lifecycle — and
|
||||||
|
# differs only in bypassing the tier-1 seen-ledger so dropped-and-deleted
|
||||||
|
# near-dups get re-fetched and re-evaluated under the CURRENT pHash
|
||||||
|
# threshold (tier-2 disk still spares files we kept). The
|
||||||
|
# `_backfill_bypass_seen` flag rides alongside the running backfill state;
|
||||||
|
# download mode is "recovery" when both are set.
|
||||||
|
bypass_seen = bool(overrides.get("_backfill_bypass_seen"))
|
||||||
|
if in_backfill:
|
||||||
|
skip_value: bool | str = BACKFILL_SKIP_VALUE
|
||||||
|
source_config.timeout = BACKFILL_CHUNK_SECONDS
|
||||||
|
pending_cursor = overrides.get("_backfill_cursor")
|
||||||
|
if uses_native_ingester(ctx["platform"]) and pending_cursor:
|
||||||
|
source_config.resume_cursor = pending_cursor
|
||||||
|
else:
|
||||||
|
skip_value = TICK_SKIP_VALUE
|
||||||
|
|
||||||
|
# Phase 2 dispatch is uniform across backends (download_backends.
|
||||||
|
# run_download — the download counterpart to verify_source_credential):
|
||||||
|
# native platforms run their ingester in `mode` (zero per-file HEADs,
|
||||||
|
# native cursor/resume, loud drift detection); gallery-dl platforms run
|
||||||
|
# the subprocess. Either returns a DownloadResult-shaped object so phase 3
|
||||||
|
# is untouched. `mode` is None for gallery-dl (run_download ignores it).
|
||||||
|
mode: str | None = None
|
||||||
|
if uses_native_ingester(ctx["platform"]):
|
||||||
|
if in_backfill and bypass_seen:
|
||||||
|
mode = "recovery"
|
||||||
|
elif in_backfill:
|
||||||
|
mode = "backfill"
|
||||||
|
else:
|
||||||
|
mode = "tick"
|
||||||
|
dl_result, resolved_campaign_id = await self._run_download(
|
||||||
|
ctx=ctx, source_config=source_config, skip_value=skip_value, mode=mode,
|
||||||
)
|
)
|
||||||
|
|
||||||
dl_result = await self.gdl.download(
|
# A backfill chunk that hit its time-box but made forward progress is
|
||||||
url=effective_url,
|
# NORMAL, not a failure — reclassify TIMEOUT → PARTIAL so it reads as
|
||||||
artist_slug=ctx["artist_slug"],
|
# "ok/progress" (PARTIAL maps to status "ok"), not a red error, and the
|
||||||
platform=ctx["platform"],
|
# next chunk just resumes from the new cursor (plan #693). A chunk that
|
||||||
source_config=source_config,
|
# timed out with NO progress stays TIMEOUT and feeds phase 3's
|
||||||
cookies_path=ctx["cookies_path"],
|
# stall-guard. Leave RATE_LIMITED alone so the platform-cooldown fires.
|
||||||
auth_token=ctx["auth_token"],
|
if in_backfill and dl_result.error_type == ErrorType.TIMEOUT:
|
||||||
)
|
new_cursor = dl_result.cursor # plan #704: structured, not scraped
|
||||||
|
advanced = bool(
|
||||||
resolved_campaign_id: str | None = None
|
(new_cursor and new_cursor != overrides.get("_backfill_cursor"))
|
||||||
if (
|
or dl_result.files_downloaded > 0
|
||||||
ctx["platform"] == "patreon"
|
)
|
||||||
and not dl_result.success
|
if advanced:
|
||||||
and "patreon_campaign_id" not in (ctx["config_overrides"] or {})
|
dl_result.error_type = ErrorType.PARTIAL
|
||||||
and _looks_like_campaign_id_failure(dl_result.stdout, dl_result.stderr)
|
dl_result.error_message = (
|
||||||
):
|
f"Backfill chunk: {dl_result.files_downloaded} file(s) — continuing"
|
||||||
vanity = _extract_patreon_vanity(ctx["url"])
|
|
||||||
if vanity:
|
|
||||||
log.info(
|
|
||||||
"Attempting campaign-ID resolution for %s (%s)",
|
|
||||||
ctx["artist_slug"], vanity,
|
|
||||||
)
|
)
|
||||||
resolved_campaign_id = await resolve_campaign_id(
|
|
||||||
vanity, ctx["cookies_path"]
|
|
||||||
)
|
|
||||||
if resolved_campaign_id:
|
|
||||||
dl_result = await self.gdl.download(
|
|
||||||
url=f"https://www.patreon.com/id:{resolved_campaign_id}",
|
|
||||||
artist_slug=ctx["artist_slug"],
|
|
||||||
platform=ctx["platform"],
|
|
||||||
source_config=source_config,
|
|
||||||
cookies_path=ctx["cookies_path"],
|
|
||||||
auth_token=ctx["auth_token"],
|
|
||||||
)
|
|
||||||
|
|
||||||
return await self._phase3_persist(
|
return await self._phase3_persist(
|
||||||
ctx["event_id"], ctx, dl_result, resolved_campaign_id,
|
ctx["event_id"], ctx, dl_result, resolved_campaign_id,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
async def _run_download(
|
||||||
|
self, *, ctx: dict, source_config, skip_value, mode: str | None,
|
||||||
|
) -> tuple[DownloadResult, str | None]:
|
||||||
|
"""Phase-2 dispatch → `download_backends.run_download`, passing this
|
||||||
|
service's gdl + sync sessionmaker. Kept as a thin instance method so tests
|
||||||
|
can stub the whole phase-2 dispatch on the service, and so the per-backend
|
||||||
|
construction lives in download_backends (the backend registry)."""
|
||||||
|
return await run_download(
|
||||||
|
ctx=ctx, source_config=source_config, skip_value=skip_value, mode=mode,
|
||||||
|
gdl=self.gdl, sync_session_factory=self.sync_session_factory,
|
||||||
|
)
|
||||||
|
|
||||||
async def _phase1_setup(self, source_id: int) -> dict[str, Any]:
|
async def _phase1_setup(self, source_id: int) -> dict[str, Any]:
|
||||||
source = (await self.async_session.execute(
|
source = (await self.async_session.execute(
|
||||||
select(Source).options(joinedload(Source.artist)).where(Source.id == source_id)
|
select(Source).options(joinedload(Source.artist)).where(Source.id == source_id)
|
||||||
@@ -155,6 +201,13 @@ class DownloadService:
|
|||||||
return {"status": "in_flight", "event_id": existing.id}
|
return {"status": "in_flight", "event_id": existing.id}
|
||||||
if existing and existing.status == "pending":
|
if existing and existing.status == "pending":
|
||||||
existing.status = "running"
|
existing.status = "running"
|
||||||
|
# Reset started_at on the pending→running transition so the
|
||||||
|
# recovery sweep (DOWNLOAD_STALL_THRESHOLD_MINUTES, 30 min)
|
||||||
|
# measures from real start, not from enqueue. On heavy-queue
|
||||||
|
# days a freshly-promoted event whose original started_at
|
||||||
|
# predated the cutoff would otherwise get swept mid-flight,
|
||||||
|
# racing phase3's commit. Audit 2026-06-02.
|
||||||
|
existing.started_at = datetime.now(UTC)
|
||||||
await self.async_session.commit()
|
await self.async_session.commit()
|
||||||
event_id = existing.id
|
event_id = existing.id
|
||||||
else:
|
else:
|
||||||
@@ -165,7 +218,12 @@ class DownloadService:
|
|||||||
event_id = ev.id
|
event_id = ev.id
|
||||||
|
|
||||||
artist = source.artist
|
artist = source.artist
|
||||||
if source.platform in ("discord", "pixiv"):
|
# Drive cookies-vs-token selection from the platform registry's
|
||||||
|
# auth_type so a new 7th token-platform automatically picks the
|
||||||
|
# right credential path. The hardcoded tuple here used to drift
|
||||||
|
# out of sync with credential_service's auth_type_for(). Audit
|
||||||
|
# 2026-06-02.
|
||||||
|
if auth_type_for(source.platform) == "token":
|
||||||
cookies_path = None
|
cookies_path = None
|
||||||
auth_token = await self.cred_service.get_token(source.platform)
|
auth_token = await self.cred_service.get_token(source.platform)
|
||||||
else:
|
else:
|
||||||
@@ -184,6 +242,7 @@ class DownloadService:
|
|||||||
"config_overrides": dict(source.config_overrides or {}),
|
"config_overrides": dict(source.config_overrides or {}),
|
||||||
"cookies_path": cookies_path,
|
"cookies_path": cookies_path,
|
||||||
"auth_token": auth_token,
|
"auth_token": auth_token,
|
||||||
|
"backfill_runs_remaining": source.backfill_runs_remaining or 0,
|
||||||
}
|
}
|
||||||
|
|
||||||
async def _phase3_persist(
|
async def _phase3_persist(
|
||||||
@@ -210,6 +269,11 @@ class DownloadService:
|
|||||||
source_row = self.sync_session.get(Source, ctx["source_id"])
|
source_row = self.sync_session.get(Source, ctx["source_id"])
|
||||||
|
|
||||||
import_summary = {"attached": 0, "skipped": 0, "errors": 0}
|
import_summary = {"attached": 0, "skipped": 0, "errors": 0}
|
||||||
|
# Archives detected but captured WITHOUT extracting any image (probe
|
||||||
|
# rejected / corrupt / missing extractor backend). Surfaced on the event
|
||||||
|
# so a post showing "no images" beside a zip is diagnosable (plan
|
||||||
|
# follow-up 2026-06-06 — the recurring archive-association report).
|
||||||
|
unextracted_archives: list[dict] = []
|
||||||
bytes_downloaded = 0
|
bytes_downloaded = 0
|
||||||
|
|
||||||
loop = asyncio.get_running_loop()
|
loop = asyncio.get_running_loop()
|
||||||
@@ -237,6 +301,51 @@ class DownloadService:
|
|||||||
bytes_downloaded += path.stat().st_size # noqa: ASYNC240
|
bytes_downloaded += path.stat().st_size # noqa: ASYNC240
|
||||||
except OSError:
|
except OSError:
|
||||||
pass
|
pass
|
||||||
|
# Enqueue thumbnail + ML for newly-attached images, matching
|
||||||
|
# the filesystem-import path (tasks/import_file.py:228-239).
|
||||||
|
# Importer.attach_in_place deliberately skips inline thumb
|
||||||
|
# generation to keep the import queue moving; the calling
|
||||||
|
# task is responsible for the enqueue. Operator-flagged
|
||||||
|
# 2026-06-01: without this, every downloaded image stayed
|
||||||
|
# at thumbnail_path=NULL until a periodic backfill swept
|
||||||
|
# it up, surfacing as broken-thumbnail tiles in the gallery
|
||||||
|
# for hours after a download landed. Lazy import to avoid
|
||||||
|
# circular-import risk between this service and the
|
||||||
|
# tasks/* modules that import it.
|
||||||
|
from ..tasks.ml import tag_and_embed
|
||||||
|
from ..tasks.thumbnail import generate_thumbnail
|
||||||
|
ids = list(result.member_image_ids)
|
||||||
|
if result.image_id is not None and result.image_id not in ids:
|
||||||
|
ids.append(result.image_id)
|
||||||
|
for img_id in ids:
|
||||||
|
generate_thumbnail.delay(img_id)
|
||||||
|
tag_and_embed.delay(img_id)
|
||||||
|
elif result.status == "attached":
|
||||||
|
# Non-media or extracted archive captured as PostAttachment
|
||||||
|
# (FC-2d-iii). The canonical copy lives in the attachments
|
||||||
|
# store; the original download path is now redundant —
|
||||||
|
# mirror duplicate_hash cleanup so we don't keep two copies.
|
||||||
|
# Operator-flagged 2026-06-02 (Lustria OST zip).
|
||||||
|
import_summary["attached"] += 1
|
||||||
|
# An archive captured WITHOUT extracting any image carries a
|
||||||
|
# reason on result.error — record it so the event explains the
|
||||||
|
# "no images beside a zip" symptom instead of staying silent.
|
||||||
|
if result.error:
|
||||||
|
unextracted_archives.append(
|
||||||
|
{"file": path.name, "reason": result.error}
|
||||||
|
)
|
||||||
|
log.warning(
|
||||||
|
"archive captured unextracted (%s): %s",
|
||||||
|
path.name, result.error,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
bytes_downloaded += path.stat().st_size # noqa: ASYNC240
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
path.unlink(missing_ok=True) # noqa: ASYNC240
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
elif result.status == "skipped" and result.skip_reason and result.skip_reason.value in (
|
elif result.status == "skipped" and result.skip_reason and result.skip_reason.value in (
|
||||||
"duplicate_hash", "duplicate_phash",
|
"duplicate_hash", "duplicate_phash",
|
||||||
):
|
):
|
||||||
@@ -245,49 +354,211 @@ class DownloadService:
|
|||||||
path.unlink(missing_ok=True) # noqa: ASYNC240
|
path.unlink(missing_ok=True) # noqa: ASYNC240
|
||||||
except OSError:
|
except OSError:
|
||||||
pass
|
pass
|
||||||
|
elif result.status == "skipped":
|
||||||
|
# Soft skip (too_small, too_transparent, invalid_image) —
|
||||||
|
# the file just didn't qualify, not a download/ingest
|
||||||
|
# failure. Don't flag the run as error; the file stays
|
||||||
|
# on disk for operator inspection.
|
||||||
|
import_summary["skipped"] += 1
|
||||||
|
elif result.status == "failed":
|
||||||
|
# Hard failure (today only: archive probe crash/timeout).
|
||||||
|
# The original archive sits in /images/ as an orphan; the
|
||||||
|
# filesystem scanner would re-import and re-crash on the
|
||||||
|
# same file, so delete the source file and surface the
|
||||||
|
# error in import_summary. Audit 2026-06-02.
|
||||||
|
import_summary["errors"] += 1
|
||||||
|
try:
|
||||||
|
path.unlink(missing_ok=True) # noqa: ASYNC240
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
elif result.status == "refreshed":
|
||||||
|
# Currently unreachable from attach_in_place (the download
|
||||||
|
# path never runs in deep=True mode), but the importer's
|
||||||
|
# ImportResult contract enumerates it. Treat the same as
|
||||||
|
# 'attached' — work happened, no error. Audit 2026-06-02.
|
||||||
|
import_summary["attached"] += 1
|
||||||
else:
|
else:
|
||||||
import_summary["errors"] += 1
|
import_summary["errors"] += 1
|
||||||
|
|
||||||
|
# Post-only records: media-less posts (pure-text) the native ingester
|
||||||
|
# captured so the artist archive is complete. Upsert each (keyed on
|
||||||
|
# external_post_id → updates the same Post a media import would create,
|
||||||
|
# never doubles). No file to clean up; the sidecar stays on disk.
|
||||||
|
for rec_str in getattr(dl_result, "post_record_paths", None) or []:
|
||||||
|
rec_path = Path(rec_str)
|
||||||
|
if not rec_path.exists(): # noqa: ASYNC240
|
||||||
|
continue
|
||||||
|
|
||||||
|
def _upsert(p=rec_path):
|
||||||
|
return self.importer.upsert_post_record(
|
||||||
|
p, artist=artist, source=source_row,
|
||||||
|
)
|
||||||
|
|
||||||
|
await loop.run_in_executor(None, _upsert)
|
||||||
|
|
||||||
|
# Kick the off-platform file-host downloader for any links this run
|
||||||
|
# recorded (mega/gdrive/…). Global + idempotent (only claims pending/
|
||||||
|
# retryable rows); the beat sweep is the backstop. Lazy import dodges a
|
||||||
|
# task-module import cycle.
|
||||||
|
from ..tasks.external import sweep_external_links
|
||||||
|
sweep_external_links.delay()
|
||||||
|
|
||||||
ev = (await self.async_session.execute(
|
ev = (await self.async_session.execute(
|
||||||
select(DownloadEvent).where(DownloadEvent.id == event_id)
|
select(DownloadEvent).where(DownloadEvent.id == event_id)
|
||||||
)).scalar_one()
|
)).scalar_one()
|
||||||
|
|
||||||
run_stats = self.gdl._compute_run_stats(
|
# plan #704: the native ingester returns structured run_stats; only the
|
||||||
dl_result.return_code, dl_result.stdout, dl_result.stderr
|
# gallery-dl path needs the regex-over-stdout reconstruction.
|
||||||
)
|
if dl_result.run_stats is not None:
|
||||||
|
run_stats = dict(dl_result.run_stats)
|
||||||
|
else:
|
||||||
|
run_stats = self.gdl._compute_run_stats(
|
||||||
|
dl_result.return_code, dl_result.stdout, dl_result.stderr
|
||||||
|
)
|
||||||
run_stats["quarantined_count"] = dl_result.files_quarantined
|
run_stats["quarantined_count"] = dl_result.files_quarantined
|
||||||
stderr_summary = self.gdl._extract_errors_warnings(dl_result.stderr)
|
stderr_summary = extract_errors_warnings(dl_result.stderr)
|
||||||
|
|
||||||
status = "ok" if (dl_result.success and import_summary["errors"] == 0) else "error"
|
# Plan #544: PARTIAL means the run downloaded ≥1 file but the
|
||||||
|
# subprocess didn't finish in budget (typically wall-clock timeout
|
||||||
|
# mid-walk). Real work happened; the next tick continues via
|
||||||
|
# gallery-dl's archive. NOT a failure for status purposes.
|
||||||
|
if dl_result.success and import_summary["errors"] == 0:
|
||||||
|
status = "ok"
|
||||||
|
elif dl_result.error_type == ErrorType.PARTIAL and import_summary["errors"] == 0:
|
||||||
|
status = "ok"
|
||||||
|
else:
|
||||||
|
status = "error"
|
||||||
ev.status = status
|
ev.status = status
|
||||||
ev.finished_at = datetime.now(UTC)
|
ev.finished_at = datetime.now(UTC)
|
||||||
ev.files_count = import_summary["attached"]
|
ev.files_count = import_summary["attached"]
|
||||||
ev.bytes_downloaded = bytes_downloaded
|
ev.bytes_downloaded = bytes_downloaded
|
||||||
ev.error = dl_result.error_message if not dl_result.success else None
|
ev.error = dl_result.error_message if status == "error" else None
|
||||||
ev.metadata_ = {
|
ev.metadata_ = {
|
||||||
"run_stats": run_stats,
|
"run_stats": run_stats,
|
||||||
"error_type": dl_result.error_type.value if dl_result.error_type else None,
|
"error_type": dl_result.error_type.value if dl_result.error_type else None,
|
||||||
"stdout": self.gdl._truncate_log(dl_result.stdout) or None,
|
"stdout": truncate_log(dl_result.stdout) or None,
|
||||||
"stderr": self.gdl._truncate_log(dl_result.stderr) or None,
|
"stderr": truncate_log(dl_result.stderr) or None,
|
||||||
"stderr_errors_warnings": stderr_summary or None,
|
"stderr_errors_warnings": stderr_summary or None,
|
||||||
"duration_seconds": dl_result.duration_seconds,
|
"duration_seconds": dl_result.duration_seconds,
|
||||||
"quarantined_paths": dl_result.quarantined_paths or None,
|
"quarantined_paths": dl_result.quarantined_paths or None,
|
||||||
"import_summary": import_summary,
|
"import_summary": import_summary,
|
||||||
|
# Archives detected but captured without extracting an image — the
|
||||||
|
# recurring "post shows a zip but no images" report. Each entry is
|
||||||
|
# {file, reason}; None when every archive extracted cleanly.
|
||||||
|
"unextracted_archives": unextracted_archives or None,
|
||||||
}
|
}
|
||||||
await self._update_source_health(
|
await self._update_source_health(
|
||||||
source_id=ctx["source_id"], status=status, error_message=ev.error,
|
source_id=ctx["source_id"], status=status, error_message=ev.error,
|
||||||
|
error_type=dl_result.error_type.value if dl_result.error_type else None,
|
||||||
|
retry_after_seconds=getattr(dl_result, "retry_after_seconds", None),
|
||||||
)
|
)
|
||||||
|
await self._apply_backfill_lifecycle(ctx, dl_result)
|
||||||
await self.async_session.commit()
|
await self.async_session.commit()
|
||||||
return event_id
|
return event_id
|
||||||
|
|
||||||
|
async def _apply_backfill_lifecycle(self, ctx: dict, dl_result) -> None:
|
||||||
|
"""Backfill state machine (plan #693, building on the cursor of #689).
|
||||||
|
|
||||||
|
A backfill runs in time-boxed chunks while
|
||||||
|
`config_overrides["_backfill_state"] == "running"`. Each chunk:
|
||||||
|
- COMPLETES the walk → clean rc=0 (gallery-dl with skip:True exits 0
|
||||||
|
only after exhausting the newest→oldest walk; a chunk cut short by
|
||||||
|
its time-box returns success=False / rc<0 via TimeoutExpired). On
|
||||||
|
completion: state="complete", clear the cursor, return to tick mode.
|
||||||
|
- made PROGRESS (cursor advanced and/or files written) → stay
|
||||||
|
"running", checkpoint the new cursor, bump the chunk counter, and
|
||||||
|
spend one of the safety-cap chunks (backfill_runs_remaining). If the
|
||||||
|
cap is exhausted without finishing → state="stalled".
|
||||||
|
- made NO progress → increment the stall counter; two strikes →
|
||||||
|
state="stalled", clear the cursor (a wedged walk can't loop).
|
||||||
|
|
||||||
|
State is the single source of truth; the cursor lives only inside a
|
||||||
|
running backfill. config_overrides is reassigned (not mutated in place)
|
||||||
|
for SQLAlchemy JSON change detection.
|
||||||
|
"""
|
||||||
|
overrides = ctx["config_overrides"] or {}
|
||||||
|
if overrides.get("_backfill_state") != "running":
|
||||||
|
return # not backfilling — tick mode, nothing to do
|
||||||
|
|
||||||
|
old_cursor = overrides.get("_backfill_cursor")
|
||||||
|
cap_remaining = ctx.get("backfill_runs_remaining", 0) or 0
|
||||||
|
|
||||||
|
src = (await self.async_session.execute(
|
||||||
|
select(Source).where(Source.id == ctx["source_id"])
|
||||||
|
)).scalar_one()
|
||||||
|
new_overrides = dict(src.config_overrides or {})
|
||||||
|
chunks = int(new_overrides.get("_backfill_chunks", 0)) + 1
|
||||||
|
new_overrides["_backfill_chunks"] = chunks
|
||||||
|
# plan #704 (#5): _backfill_posts (the live progress badge) is OWNED by the
|
||||||
|
# ingester now — it writes a monotonic absolute mid-walk at each page
|
||||||
|
# boundary (ingest_core._checkpoint_posts), so the badge climbs DURING a
|
||||||
|
# chunk instead of jumping once per chunk here, and the re-walked resume
|
||||||
|
# page is no longer double-counted. new_overrides (read fresh above)
|
||||||
|
# carries the ingester's committed value forward untouched.
|
||||||
|
|
||||||
|
completed = (
|
||||||
|
dl_result.success
|
||||||
|
and dl_result.error_type is None
|
||||||
|
and dl_result.return_code == 0
|
||||||
|
)
|
||||||
|
if completed:
|
||||||
|
new_overrides["_backfill_state"] = "complete"
|
||||||
|
new_overrides.pop("_backfill_cursor", None)
|
||||||
|
new_overrides.pop("_backfill_cursor_stalls", None)
|
||||||
|
# plan #697: a recovery walk shares this lifecycle; clear its bypass
|
||||||
|
# flag on completion so the next routine tick honors the seen-ledger.
|
||||||
|
new_overrides.pop("_backfill_bypass_seen", None)
|
||||||
|
src.config_overrides = new_overrides
|
||||||
|
src.backfill_runs_remaining = 0
|
||||||
|
return
|
||||||
|
|
||||||
|
# Did not finish. The native ingester checkpoints + resumes via cursor
|
||||||
|
# (carried structurally on the result, plan #704); gallery-dl platforms
|
||||||
|
# have no resumable cursor (every chunk re-walks from the top), so they
|
||||||
|
# advance only by the download archive growing.
|
||||||
|
new_cursor = (
|
||||||
|
dl_result.cursor if uses_native_ingester(ctx["platform"]) else None
|
||||||
|
)
|
||||||
|
advanced = bool(
|
||||||
|
(new_cursor and new_cursor != old_cursor)
|
||||||
|
or dl_result.files_downloaded > 0
|
||||||
|
)
|
||||||
|
if advanced:
|
||||||
|
if new_cursor:
|
||||||
|
new_overrides["_backfill_cursor"] = new_cursor
|
||||||
|
new_overrides.pop("_backfill_cursor_stalls", None)
|
||||||
|
cap_remaining = max(0, cap_remaining - 1)
|
||||||
|
src.backfill_runs_remaining = cap_remaining
|
||||||
|
if cap_remaining == 0:
|
||||||
|
# Safety cap hit before reaching the bottom — pause, don't loop.
|
||||||
|
new_overrides["_backfill_state"] = "stalled"
|
||||||
|
else:
|
||||||
|
stalls = int(new_overrides.get("_backfill_cursor_stalls", 0)) + 1
|
||||||
|
if stalls >= 2:
|
||||||
|
new_overrides["_backfill_state"] = "stalled"
|
||||||
|
new_overrides.pop("_backfill_cursor", None)
|
||||||
|
new_overrides.pop("_backfill_cursor_stalls", None)
|
||||||
|
else:
|
||||||
|
new_overrides["_backfill_cursor_stalls"] = stalls
|
||||||
|
|
||||||
|
src.config_overrides = new_overrides
|
||||||
|
|
||||||
async def _update_source_health(
|
async def _update_source_health(
|
||||||
self, *, source_id: int, status: str, error_message: str | None,
|
self, *, source_id: int, status: str, error_message: str | None,
|
||||||
|
error_type: str | None = None, retry_after_seconds: float | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""FC-3d: update Source.{consecutive_failures, last_error, last_checked_at}.
|
"""FC-3d: update Source.{consecutive_failures, last_error, last_checked_at}.
|
||||||
|
|
||||||
ok -> failures = 0, error = None, checked_at = now
|
ok -> failures = 0, error = None, checked_at = now
|
||||||
error -> failures += 1, error = error_message, checked_at = now
|
error -> failures += 1, error = error_message, checked_at = now
|
||||||
skipped -> failures unchanged, error = None, checked_at = now
|
skipped -> failures unchanged, error = None, checked_at = now
|
||||||
|
|
||||||
|
When error_type == 'rate_limited', also stamps a platform-wide
|
||||||
|
cooldown via scheduler_service.set_platform_cooldown so the next
|
||||||
|
scan tick skips every source on this platform until the cooldown
|
||||||
|
expires. Preventive half of the burst-prevention pair —
|
||||||
|
consecutive_failures still backs the offending source off across
|
||||||
|
ticks.
|
||||||
"""
|
"""
|
||||||
source = (await self.async_session.execute(
|
source = (await self.async_session.execute(
|
||||||
select(Source).where(Source.id == source_id)
|
select(Source).where(Source.id == source_id)
|
||||||
@@ -296,9 +567,27 @@ class DownloadService:
|
|||||||
if status == "ok":
|
if status == "ok":
|
||||||
source.consecutive_failures = 0
|
source.consecutive_failures = 0
|
||||||
source.last_error = None
|
source.last_error = None
|
||||||
|
# alembic 0032 — clear the failure-class chip on success.
|
||||||
|
source.error_type = None
|
||||||
elif status == "error":
|
elif status == "error":
|
||||||
source.consecutive_failures = (source.consecutive_failures or 0) + 1
|
source.consecutive_failures = (source.consecutive_failures or 0) + 1
|
||||||
source.last_error = error_message
|
source.last_error = error_message
|
||||||
|
# alembic 0032 — stamp the failure-class so FailingSourcesCard
|
||||||
|
# can render a colored chip and operators can bulk-triage
|
||||||
|
# by error class without opening Logs per row.
|
||||||
|
source.error_type = error_type
|
||||||
|
if error_type == "rate_limited":
|
||||||
|
# plan #708 B1: honor the server's Retry-After when the native
|
||||||
|
# client surfaced one (clamped to a sane [60, 3600] window so a
|
||||||
|
# tiny hint can't leave the platform effectively un-cooled and a
|
||||||
|
# huge one can't strand it for hours); else the flat default.
|
||||||
|
if retry_after_seconds is not None:
|
||||||
|
seconds = int(min(max(retry_after_seconds, 60), 3600))
|
||||||
|
await set_platform_cooldown(
|
||||||
|
self.async_session, source.platform, seconds=seconds,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
await set_platform_cooldown(self.async_session, source.platform)
|
||||||
elif status == "skipped":
|
elif status == "skipped":
|
||||||
source.last_error = None
|
source.last_error = None
|
||||||
source.last_checked_at = now
|
source.last_checked_at = now
|
||||||
|
|||||||
@@ -11,11 +11,12 @@ from __future__ import annotations
|
|||||||
import re
|
import re
|
||||||
|
|
||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
from sqlalchemy.exc import IntegrityError
|
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from ..models import Artist, Source
|
from ..models import Artist, Source
|
||||||
from ..utils.slug import slugify
|
from ..utils.slug import slugify
|
||||||
|
from .db_helpers import get_or_create
|
||||||
|
from .source_service import BACKFILL_MAX_CHUNKS
|
||||||
|
|
||||||
|
|
||||||
class UnknownPlatformError(Exception):
|
class UnknownPlatformError(Exception):
|
||||||
@@ -86,6 +87,67 @@ class ExtensionService:
|
|||||||
"created_artist": created_artist,
|
"created_artist": created_artist,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async def probe(self, url: str) -> dict:
|
||||||
|
"""Read-only resolution of a creator-page URL against the FC DB.
|
||||||
|
Returns one of:
|
||||||
|
- {state: 'unknown_platform'} — URL didn't match any
|
||||||
|
platform's strict artist-page pattern
|
||||||
|
- {state: 'new', platform, slug} — would create both
|
||||||
|
artist and source on quick-add
|
||||||
|
- {state: 'artist_match', platform, slug, artist}
|
||||||
|
— artist exists, this
|
||||||
|
exact URL isn't a Source yet (collapses the sidecar-synthetic
|
||||||
|
case too — the synthetic anchor counts as an existing artist
|
||||||
|
row but not as a pollable Source for this URL)
|
||||||
|
- {state: 'source_match', platform, slug, artist, source}
|
||||||
|
— exact (artist, platform,
|
||||||
|
url) Source already exists
|
||||||
|
|
||||||
|
Side-effect-free: two SELECTs at most.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
platform, raw_slug = self._derive(url)
|
||||||
|
except (UnknownPlatformError, InvalidUrlError):
|
||||||
|
return {"state": "unknown_platform"}
|
||||||
|
|
||||||
|
slug = slugify(raw_slug)
|
||||||
|
artist = (await self.session.execute(
|
||||||
|
select(Artist).where(Artist.slug == slug)
|
||||||
|
)).scalar_one_or_none()
|
||||||
|
if artist is None:
|
||||||
|
return {"state": "new", "platform": platform, "slug": slug}
|
||||||
|
|
||||||
|
artist_payload = {"id": artist.id, "name": artist.name, "slug": artist.slug}
|
||||||
|
|
||||||
|
source = (await self.session.execute(
|
||||||
|
select(Source).where(
|
||||||
|
Source.artist_id == artist.id,
|
||||||
|
Source.platform == platform,
|
||||||
|
Source.url == url,
|
||||||
|
)
|
||||||
|
)).scalar_one_or_none()
|
||||||
|
if source is None:
|
||||||
|
return {
|
||||||
|
"state": "artist_match",
|
||||||
|
"platform": platform,
|
||||||
|
"slug": slug,
|
||||||
|
"artist": artist_payload,
|
||||||
|
}
|
||||||
|
|
||||||
|
return {
|
||||||
|
"state": "source_match",
|
||||||
|
"platform": platform,
|
||||||
|
"slug": slug,
|
||||||
|
"artist": artist_payload,
|
||||||
|
"source": {
|
||||||
|
"id": source.id,
|
||||||
|
"artist_id": source.artist_id,
|
||||||
|
"platform": source.platform,
|
||||||
|
"url": source.url,
|
||||||
|
"enabled": source.enabled,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
def _derive(self, url: str) -> tuple[str, str]:
|
def _derive(self, url: str) -> tuple[str, str]:
|
||||||
if not isinstance(url, str) or not url.strip():
|
if not isinstance(url, str) or not url.strip():
|
||||||
raise InvalidUrlError("url is empty")
|
raise InvalidUrlError("url is empty")
|
||||||
@@ -107,24 +169,16 @@ class ExtensionService:
|
|||||||
500 against uq_artist_slug.
|
500 against uq_artist_slug.
|
||||||
"""
|
"""
|
||||||
slug = slugify(raw_name)
|
slug = slugify(raw_name)
|
||||||
existing = (await self.session.execute(
|
|
||||||
select(Artist).where(Artist.slug == slug)
|
async def _create() -> Artist:
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
return existing, False
|
|
||||||
sp = await self.session.begin_nested()
|
|
||||||
try:
|
|
||||||
artist = Artist(name=raw_name, slug=slug, is_subscription=True)
|
artist = Artist(name=raw_name, slug=slug, is_subscription=True)
|
||||||
self.session.add(artist)
|
self.session.add(artist)
|
||||||
await self.session.flush()
|
await self.session.flush()
|
||||||
await sp.commit()
|
return artist
|
||||||
return artist, True
|
|
||||||
except IntegrityError:
|
return await get_or_create(
|
||||||
await sp.rollback()
|
self.session, select(Artist).where(Artist.slug == slug), _create
|
||||||
recovered = (await self.session.execute(
|
)
|
||||||
select(Artist).where(Artist.slug == slug)
|
|
||||||
)).scalar_one()
|
|
||||||
return recovered, False
|
|
||||||
|
|
||||||
async def _find_or_create_source(
|
async def _find_or_create_source(
|
||||||
self, *, artist_id: int, platform: str, url: str,
|
self, *, artist_id: int, platform: str, url: str,
|
||||||
@@ -132,33 +186,32 @@ class ExtensionService:
|
|||||||
"""Race-safe — same pattern as _find_or_create_artist above. The
|
"""Race-safe — same pattern as _find_or_create_artist above. The
|
||||||
uq_source_artist_platform_url constraint catches the duplicate
|
uq_source_artist_platform_url constraint catches the duplicate
|
||||||
insert; we roll the savepoint back and re-select."""
|
insert; we roll the savepoint back and re-select."""
|
||||||
existing = (await self.session.execute(
|
select_existing = select(Source).where(
|
||||||
select(Source).where(
|
Source.artist_id == artist_id,
|
||||||
Source.artist_id == artist_id,
|
Source.platform == platform,
|
||||||
Source.platform == platform,
|
Source.url == url,
|
||||||
Source.url == url,
|
)
|
||||||
)
|
|
||||||
)).scalar_one_or_none()
|
async def _create() -> Source:
|
||||||
if existing is not None:
|
# New subscription sources arm run-until-done backfill (plan #693)
|
||||||
return existing, False
|
# so the first ticks walk the full history (otherwise gallery-dl's
|
||||||
sp = await self.session.begin_nested()
|
# exit:20 short-circuits before the archive is built). Mirrors
|
||||||
try:
|
# SourceService.create — without it, Firefox quick-add on a creator
|
||||||
|
# with >20 unsynced posts would surface as "check failed" with no
|
||||||
|
# diagnosis. Audit 2026-06-02.
|
||||||
src = Source(
|
src = Source(
|
||||||
artist_id=artist_id, platform=platform,
|
artist_id=artist_id, platform=platform,
|
||||||
url=url, enabled=True,
|
url=url, enabled=True,
|
||||||
|
config_overrides={"_backfill_state": "running"},
|
||||||
|
backfill_runs_remaining=BACKFILL_MAX_CHUNKS,
|
||||||
)
|
)
|
||||||
self.session.add(src)
|
self.session.add(src)
|
||||||
await self.session.flush()
|
await self.session.flush()
|
||||||
await sp.commit()
|
return src
|
||||||
except IntegrityError:
|
|
||||||
await sp.rollback()
|
src, created = await get_or_create(
|
||||||
recovered = (await self.session.execute(
|
self.session, select_existing, _create
|
||||||
select(Source).where(
|
)
|
||||||
Source.artist_id == artist_id,
|
if created:
|
||||||
Source.platform == platform,
|
await self.session.commit()
|
||||||
Source.url == url,
|
return src, created
|
||||||
)
|
|
||||||
)).scalar_one()
|
|
||||||
return recovered, False
|
|
||||||
await self.session.commit()
|
|
||||||
return src, True
|
|
||||||
|
|||||||
@@ -0,0 +1,248 @@
|
|||||||
|
"""Fetchers for off-platform file hosts (mega / gdrive / mediafire / dropbox /
|
||||||
|
pixeldrain).
|
||||||
|
|
||||||
|
A shared, reusable subsystem: given an external_link URL, fetch the file(s) into
|
||||||
|
a destination directory and report the outcome. The download worker (separate
|
||||||
|
slice) drives these off the external_link ledger; any in-house downloader can
|
||||||
|
call `fetch_external()` directly.
|
||||||
|
|
||||||
|
No single tool covers all five hosts, so a small registry maps host → fetch
|
||||||
|
function behind one signature:
|
||||||
|
|
||||||
|
fetch_external(host, url, dest_dir, *, timeout, should_stop) -> FetchResult
|
||||||
|
|
||||||
|
Backends:
|
||||||
|
- dropbox : force the direct-download variant (dl=1) + stream GET.
|
||||||
|
- pixeldrain : GET the /api/file/{id} endpoint.
|
||||||
|
- mediafire : scrape the download page for the direct link + stream GET.
|
||||||
|
- gdrive : gdown (handles the confirm-token + virus-scan interstitial).
|
||||||
|
- mega : `megatools dl` subprocess (public link incl. #key); needs the
|
||||||
|
`megatools` binary in the runtime image (Debian apt package).
|
||||||
|
|
||||||
|
Public links work credential-free (rule 26); per-host creds are a later Settings
|
||||||
|
concern. Plain-HTTP homelab — no secure-context API. The HTTP / gdown /
|
||||||
|
subprocess calls go through module-level seams so unit tests run without
|
||||||
|
network, gdown, or MEGAcmd.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
from collections.abc import Callable
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from pathlib import Path
|
||||||
|
from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
|
||||||
|
|
||||||
|
import requests
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
_CHUNK = 1 << 16
|
||||||
|
_DEFAULT_TIMEOUT = 600.0
|
||||||
|
_USER_AGENT = (
|
||||||
|
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
|
||||||
|
"(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
|
||||||
|
)
|
||||||
|
# MediaFire's download page embeds the real file URL in a download-button href.
|
||||||
|
_MEDIAFIRE_RE = re.compile(
|
||||||
|
r'href="(https://download[^"]+?\.mediafire\.com/[^"]+)"', re.IGNORECASE
|
||||||
|
)
|
||||||
|
_CD_FILENAME_RE = re.compile(r'filename\*?=(?:UTF-8\'\')?"?([^";]+)"?', re.IGNORECASE)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class FetchResult:
|
||||||
|
files: list[Path] = field(default_factory=list)
|
||||||
|
bytes: int = 0
|
||||||
|
error: str | None = None
|
||||||
|
|
||||||
|
@property
|
||||||
|
def ok(self) -> bool:
|
||||||
|
return self.error is None and bool(self.files)
|
||||||
|
|
||||||
|
|
||||||
|
class ExternalFetchError(Exception):
|
||||||
|
"""A fetch failed in a way worth recording on the link's last_error."""
|
||||||
|
|
||||||
|
|
||||||
|
# -- seams (monkeypatched in tests) ----------------------------------------
|
||||||
|
|
||||||
|
def _http_get(url: str, *, timeout: float, headers: dict | None = None,
|
||||||
|
stream: bool = True) -> requests.Response:
|
||||||
|
hdrs = {"User-Agent": _USER_AGENT}
|
||||||
|
if headers:
|
||||||
|
hdrs.update(headers)
|
||||||
|
return requests.get(url, timeout=timeout, headers=hdrs, stream=stream)
|
||||||
|
|
||||||
|
|
||||||
|
def _gdown_download(url: str, out_dir: str) -> str | None:
|
||||||
|
"""Download a Google Drive url into out_dir via gdown; return the written
|
||||||
|
path. Imported lazily so the dep is optional at module-import time."""
|
||||||
|
# Lazy import: keeps gdown an optional dep that isn't needed to import this
|
||||||
|
# module (e.g. the no-DB unit lane that never exercises a real fetch).
|
||||||
|
import gdown
|
||||||
|
# A trailing sep tells gdown to keep the server-side filename inside out_dir.
|
||||||
|
return gdown.download(url, output=out_dir + os.sep, quiet=True, fuzzy=True)
|
||||||
|
|
||||||
|
|
||||||
|
def _run_mega_get(url: str, out_dir: str, *, timeout: float) -> None:
|
||||||
|
"""Download a mega.nz public link (key in the #fragment) into out_dir via
|
||||||
|
`megatools dl` (the Debian `megatools` package). Raises ExternalFetchError
|
||||||
|
on non-zero exit."""
|
||||||
|
# Fixed argv (not shell): only `url` is external input, passed positionally,
|
||||||
|
# so there's no shell-injection surface.
|
||||||
|
proc = subprocess.run(
|
||||||
|
["megatools", "dl", "--path", out_dir, url],
|
||||||
|
capture_output=True, text=True, timeout=timeout, check=False,
|
||||||
|
)
|
||||||
|
if proc.returncode != 0:
|
||||||
|
raise ExternalFetchError(
|
||||||
|
f"megatools dl exit {proc.returncode}: {(proc.stderr or '').strip()[:300]}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# -- helpers ---------------------------------------------------------------
|
||||||
|
|
||||||
|
def _safe_name(name: str, fallback: str) -> str:
|
||||||
|
name = os.path.basename((name or "").strip().strip('"'))
|
||||||
|
# Strip path separators / control chars; never empty.
|
||||||
|
name = re.sub(r'[<>:"/\\|?*\x00-\x1f]', "_", name).strip(". ")
|
||||||
|
return name or fallback
|
||||||
|
|
||||||
|
|
||||||
|
def _filename_from(resp: requests.Response, url: str, fallback: str) -> str:
|
||||||
|
cd = resp.headers.get("Content-Disposition", "")
|
||||||
|
m = _CD_FILENAME_RE.search(cd)
|
||||||
|
if m:
|
||||||
|
return _safe_name(m.group(1), fallback)
|
||||||
|
path_name = os.path.basename(urlsplit(url).path)
|
||||||
|
return _safe_name(path_name, fallback)
|
||||||
|
|
||||||
|
|
||||||
|
def _stream_to_file(resp: requests.Response, dest: Path,
|
||||||
|
should_stop: Callable[[], bool]) -> int:
|
||||||
|
"""Stream a response body to `dest` (atomic via .part). Returns byte count.
|
||||||
|
Honors should_stop between chunks (partial file removed)."""
|
||||||
|
part = dest.with_name(dest.name + ".part")
|
||||||
|
total = 0
|
||||||
|
try:
|
||||||
|
with part.open("wb") as fh:
|
||||||
|
for chunk in resp.iter_content(chunk_size=_CHUNK):
|
||||||
|
if should_stop():
|
||||||
|
raise ExternalFetchError("stopped")
|
||||||
|
if chunk:
|
||||||
|
fh.write(chunk)
|
||||||
|
total += len(chunk)
|
||||||
|
except BaseException:
|
||||||
|
part.unlink(missing_ok=True)
|
||||||
|
raise
|
||||||
|
os.replace(part, dest)
|
||||||
|
return total
|
||||||
|
|
||||||
|
|
||||||
|
def _get_to_dir(url: str, dest_dir: Path, *, timeout: float,
|
||||||
|
should_stop: Callable[[], bool], fallback: str,
|
||||||
|
headers: dict | None = None) -> FetchResult:
|
||||||
|
resp = _http_get(url, timeout=timeout, headers=headers, stream=True)
|
||||||
|
if resp.status_code != 200:
|
||||||
|
return FetchResult(error=f"HTTP {resp.status_code} for {url}")
|
||||||
|
name = _filename_from(resp, url, fallback)
|
||||||
|
dest = dest_dir / name
|
||||||
|
written = _stream_to_file(resp, dest, should_stop)
|
||||||
|
return FetchResult(files=[dest], bytes=written)
|
||||||
|
|
||||||
|
|
||||||
|
# -- per-host fetchers -----------------------------------------------------
|
||||||
|
|
||||||
|
def _fetch_dropbox(url: str, dest_dir: Path, *, timeout: float,
|
||||||
|
should_stop: Callable[[], bool]) -> FetchResult:
|
||||||
|
# Force the direct-download variant: dl=1 (Dropbox serves an HTML preview
|
||||||
|
# for dl=0). Rewrite/insert the param rather than string-replace so ?dl=0,
|
||||||
|
# &dl=0, and a missing param all resolve.
|
||||||
|
parts = urlsplit(url)
|
||||||
|
q = dict(parse_qsl(parts.query))
|
||||||
|
q["dl"] = "1"
|
||||||
|
direct = urlunsplit(parts._replace(query=urlencode(q)))
|
||||||
|
return _get_to_dir(direct, dest_dir, timeout=timeout,
|
||||||
|
should_stop=should_stop, fallback="dropbox-file")
|
||||||
|
|
||||||
|
|
||||||
|
def _fetch_pixeldrain(url: str, dest_dir: Path, *, timeout: float,
|
||||||
|
should_stop: Callable[[], bool]) -> FetchResult:
|
||||||
|
# /u/{id} (and /l/{id}) → the API file endpoint.
|
||||||
|
file_id = urlsplit(url).path.rstrip("/").split("/")[-1]
|
||||||
|
if not file_id:
|
||||||
|
return FetchResult(error=f"no pixeldrain id in {url}")
|
||||||
|
api = f"https://pixeldrain.com/api/file/{file_id}"
|
||||||
|
return _get_to_dir(api, dest_dir, timeout=timeout,
|
||||||
|
should_stop=should_stop, fallback=f"{file_id}.bin")
|
||||||
|
|
||||||
|
|
||||||
|
def _fetch_mediafire(url: str, dest_dir: Path, *, timeout: float,
|
||||||
|
should_stop: Callable[[], bool]) -> FetchResult:
|
||||||
|
page = _http_get(url, timeout=timeout, stream=False)
|
||||||
|
if page.status_code != 200:
|
||||||
|
return FetchResult(error=f"HTTP {page.status_code} for mediafire page")
|
||||||
|
m = _MEDIAFIRE_RE.search(page.text or "")
|
||||||
|
if not m:
|
||||||
|
return FetchResult(error="mediafire direct link not found on page")
|
||||||
|
return _get_to_dir(m.group(1), dest_dir, timeout=timeout,
|
||||||
|
should_stop=should_stop, fallback="mediafire-file")
|
||||||
|
|
||||||
|
|
||||||
|
def _fetch_gdrive(url: str, dest_dir: Path, *, timeout: float,
|
||||||
|
should_stop: Callable[[], bool]) -> FetchResult:
|
||||||
|
out = _gdown_download(url, str(dest_dir))
|
||||||
|
if not out:
|
||||||
|
return FetchResult(error="gdown returned no file (quota / private?)")
|
||||||
|
p = Path(out)
|
||||||
|
if not p.exists():
|
||||||
|
return FetchResult(error=f"gdown reported {out} but it is missing")
|
||||||
|
return FetchResult(files=[p], bytes=p.stat().st_size)
|
||||||
|
|
||||||
|
|
||||||
|
def _fetch_mega(url: str, dest_dir: Path, *, timeout: float,
|
||||||
|
should_stop: Callable[[], bool]) -> FetchResult:
|
||||||
|
before = set(dest_dir.iterdir()) if dest_dir.exists() else set()
|
||||||
|
_run_mega_get(url, str(dest_dir), timeout=timeout)
|
||||||
|
new = [p for p in dest_dir.iterdir() if p not in before and p.is_file()]
|
||||||
|
if not new:
|
||||||
|
return FetchResult(error="mega-get wrote no new file")
|
||||||
|
return FetchResult(files=new, bytes=sum(p.stat().st_size for p in new))
|
||||||
|
|
||||||
|
|
||||||
|
_REGISTRY: dict[str, Callable[..., FetchResult]] = {
|
||||||
|
"dropbox": _fetch_dropbox,
|
||||||
|
"pixeldrain": _fetch_pixeldrain,
|
||||||
|
"mediafire": _fetch_mediafire,
|
||||||
|
"gdrive": _fetch_gdrive,
|
||||||
|
"mega": _fetch_mega,
|
||||||
|
}
|
||||||
|
|
||||||
|
SUPPORTED_HOSTS = tuple(_REGISTRY)
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_external(host: str, url: str, dest_dir: Path, *,
|
||||||
|
timeout: float = _DEFAULT_TIMEOUT,
|
||||||
|
should_stop: Callable[[], bool] = lambda: False) -> FetchResult:
|
||||||
|
"""Fetch `url` (a `host` link) into `dest_dir`. Returns a FetchResult; never
|
||||||
|
raises — any backend error (transport, non-200, scrape miss, subprocess
|
||||||
|
failure, stop) is captured on `.error` so the worker can record it and move
|
||||||
|
on."""
|
||||||
|
fetcher = _REGISTRY.get(host)
|
||||||
|
if fetcher is None:
|
||||||
|
return FetchResult(error=f"unsupported host {host!r}")
|
||||||
|
dest_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
try:
|
||||||
|
return fetcher(url, dest_dir, timeout=timeout, should_stop=should_stop)
|
||||||
|
except requests.RequestException as exc:
|
||||||
|
return FetchResult(error=f"transport error: {exc}")
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
return FetchResult(error="timed out")
|
||||||
|
except ExternalFetchError as exc:
|
||||||
|
return FetchResult(error=str(exc))
|
||||||
|
except Exception as exc: # never let a backend quirk kill the worker
|
||||||
|
log.warning("external fetch (%s) failed for %s: %s", host, url, exc)
|
||||||
|
return FetchResult(error=f"{type(exc).__name__}: {exc}")
|
||||||
@@ -11,9 +11,15 @@ expected to write.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import shutil
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
|
from datetime import UTC, datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
JPEG_HEAD = b"\xff\xd8\xff"
|
JPEG_HEAD = b"\xff\xd8\xff"
|
||||||
JPEG_TAIL = b"\xff\xd9"
|
JPEG_TAIL = b"\xff\xd9"
|
||||||
|
|
||||||
@@ -108,3 +114,52 @@ def validate_file(path: Path) -> ValidationResult:
|
|||||||
return ValidationResult(ok=True, format="webp", size=size)
|
return ValidationResult(ok=True, format="webp", size=size)
|
||||||
|
|
||||||
return ValidationResult(ok=True, format=None, size=size)
|
return ValidationResult(ok=True, format=None, size=size)
|
||||||
|
|
||||||
|
|
||||||
|
def quarantine_file(
|
||||||
|
images_root: Path,
|
||||||
|
path: Path,
|
||||||
|
artist_slug: str,
|
||||||
|
platform: str,
|
||||||
|
*,
|
||||||
|
url: str | None,
|
||||||
|
result: ValidationResult,
|
||||||
|
) -> Path | None:
|
||||||
|
"""Move a validation-failed file to `_quarantine/<slug>/<platform>` and write
|
||||||
|
a `.quarantine.json` provenance sidecar next to it.
|
||||||
|
|
||||||
|
Returns the destination path, or None if the move itself failed (file left
|
||||||
|
in place — the caller decides what to report). The caller has already run
|
||||||
|
`validate_file` and seen `result.ok is False`. ONE implementation for both
|
||||||
|
download backends — gallery-dl's batch post-process and the native ingester's
|
||||||
|
per-media path — so the quarantine layout + provenance sidecar can't drift.
|
||||||
|
"""
|
||||||
|
quarantine_root = images_root / "_quarantine" / artist_slug / platform
|
||||||
|
try:
|
||||||
|
quarantine_root.mkdir(parents=True, exist_ok=True)
|
||||||
|
dest = quarantine_root / path.name
|
||||||
|
counter = 1
|
||||||
|
while dest.exists():
|
||||||
|
dest = quarantine_root / f"{path.stem}.{counter}{path.suffix}"
|
||||||
|
counter += 1
|
||||||
|
shutil.move(str(path), str(dest))
|
||||||
|
sidecar = dest.with_suffix(dest.suffix + ".quarantine.json")
|
||||||
|
sidecar.write_text(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"original_path": str(path),
|
||||||
|
"source_url": url,
|
||||||
|
"artist_slug": artist_slug,
|
||||||
|
"platform": platform,
|
||||||
|
"format": result.format,
|
||||||
|
"reason": result.reason,
|
||||||
|
"size": result.size,
|
||||||
|
"quarantined_at": datetime.now(UTC).isoformat(),
|
||||||
|
},
|
||||||
|
indent=2,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
except OSError as exc:
|
||||||
|
log.error("Failed to quarantine %s: %s. File left in place.", path, exc)
|
||||||
|
return None
|
||||||
|
return dest
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ from datetime import UTC, datetime
|
|||||||
from enum import StrEnum
|
from enum import StrEnum
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from .file_validator import is_validatable, validate_file
|
from .file_validator import is_validatable, quarantine_file, validate_file
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -39,19 +39,90 @@ class ErrorType(StrEnum):
|
|||||||
HTTP_ERROR = "http_error"
|
HTTP_ERROR = "http_error"
|
||||||
UNSUPPORTED_URL = "unsupported_url"
|
UNSUPPORTED_URL = "unsupported_url"
|
||||||
VALIDATION_FAILED = "validation_failed"
|
VALIDATION_FAILED = "validation_failed"
|
||||||
|
# Native Patreon ingester only (plan #697): a response parsed as JSON but
|
||||||
|
# didn't match the JSON:API shape the ingester depends on (missing `data`,
|
||||||
|
# media lacking `file_name`/`url`, etc.). Distinct from AUTH_ERROR — the fix
|
||||||
|
# is updating the ingester's field-set/parser, not rotating credentials. The
|
||||||
|
# contract test guards against silently shipping this.
|
||||||
|
API_DRIFT = "api_drift"
|
||||||
|
# Run made real progress (downloaded ≥1 file) but did not finish in the
|
||||||
|
# subprocess budget. Distinct from UNKNOWN_ERROR — the downstream status
|
||||||
|
# mapping classifies this as "ok" because the next tick continues.
|
||||||
|
PARTIAL = "partial"
|
||||||
UNKNOWN_ERROR = "unknown_error"
|
UNKNOWN_ERROR = "unknown_error"
|
||||||
|
|
||||||
|
|
||||||
|
# Tick mode (routine cron polls): skip ≤20 contiguous already-archived
|
||||||
|
# items, then exit gallery-dl. Established subscription with zero new
|
||||||
|
# content exits in ~30s of HEAD requests instead of walking to the bottom
|
||||||
|
# of the post history (which can be hours for prolific creators). 20 (not
|
||||||
|
# 5) is operator-set headroom against any edge case where paywalled or
|
||||||
|
# otherwise-non-downloadable items might interleave with archived ones —
|
||||||
|
# 20 contiguous HEADs is still negligible.
|
||||||
|
TICK_SKIP_VALUE = "exit:20"
|
||||||
|
|
||||||
|
# Backfill mode (operator-triggered deep scan): walk the full history,
|
||||||
|
# one TIME-BOXED CHUNK per run (plan #693). config_overrides["_backfill_state"]
|
||||||
|
# == "running" selects this mode; the cursor checkpoint (plan #689) lets each
|
||||||
|
# chunk resume where the last stopped, so the walk advances across chunks
|
||||||
|
# until gallery-dl exits cleanly (= reached the bottom).
|
||||||
|
#
|
||||||
|
# The chunk budget is deliberately FAR below download_source's Celery
|
||||||
|
# soft_time_limit (DOWNLOAD_SOFT_TIME_LIMIT=1350, tasks/download.py), not
|
||||||
|
# "just under" it. Hitting this budget is the NORMAL chunk boundary, not a
|
||||||
|
# failure: subprocess.run raises TimeoutExpired (which captures partial
|
||||||
|
# stdout/stderr + the last emitted cursor), and download_service reclassifies
|
||||||
|
# a chunk that made progress as PARTIAL (status "ok"), not an error. The huge
|
||||||
|
# headroom means a stuck file or a slow chunk can never let Celery's
|
||||||
|
# SoftTimeLimitExceeded preempt TimeoutExpired (the failure mode behind Knuxy
|
||||||
|
# #38275 / Anduo #39912/#40411). Earlier we ran one ~1170s run-to-the-wall
|
||||||
|
# per arming; that died as a timeout error every time on large catalogs.
|
||||||
|
BACKFILL_SKIP_VALUE = True
|
||||||
|
BACKFILL_CHUNK_SECONDS = 600
|
||||||
|
|
||||||
|
|
||||||
|
# Sits well below download_source's Celery soft_time_limit
|
||||||
|
# (DOWNLOAD_SOFT_TIME_LIMIT=1350, tasks/download.py). subprocess.run MUST
|
||||||
|
# raise TimeoutExpired before Celery raises SoftTimeLimitExceeded —
|
||||||
|
# otherwise Celery wins the race, SIGKILLs the worker, in-memory
|
||||||
|
# stdout/stderr is lost, and the DownloadEvent ends up empty-logged with
|
||||||
|
# "stranded by recovery sweep" (operator-flagged 2026-05-31, Knuxy event
|
||||||
|
# #38275; recurred in backfill mode as Anduo #39912). Per-source bumps
|
||||||
|
# still live in source.config_overrides for legitimately long syncs —
|
||||||
|
# keep any override below the soft limit, or the soft-limit salvage path
|
||||||
|
# in tasks/download.py (_finalize_soft_limited) is the only safety net.
|
||||||
|
_DEFAULT_GDL_TIMEOUT_SECONDS = 870
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class SourceConfig:
|
class SourceConfig:
|
||||||
|
"""Per-source overrides loaded from Source.config_overrides JSON.
|
||||||
|
|
||||||
|
Note: the gallery-dl `skip` value (tick vs backfill, see TICK_SKIP_VALUE /
|
||||||
|
BACKFILL_SKIP_VALUE) is NOT carried here — it derives from the
|
||||||
|
Source.backfill_runs_remaining column at the download_service layer
|
||||||
|
and is passed to _build_config_for_source as `skip_value`. Same for
|
||||||
|
the per-run subprocess timeout.
|
||||||
|
|
||||||
|
`resume_cursor` is RUNTIME state, not persisted operator config: the
|
||||||
|
cursor-paged backfill checkpoint (see parse_last_cursor + the
|
||||||
|
download_service backfill lifecycle). It is consumed only by the native
|
||||||
|
Patreon ingester (the sole cursor-paged platform; gallery-dl's Patreon path
|
||||||
|
was removed at the #697 cutover): on a backfill/recovery run the ingester
|
||||||
|
resumes its newest→oldest walk from that pagination cursor instead of
|
||||||
|
restarting from the top. download_service threads it from
|
||||||
|
config_overrides["_backfill_cursor"]; SourceConfig deliberately does NOT read
|
||||||
|
it in from_dict (config_overrides also holds operator config, and
|
||||||
|
resume_cursor must only apply in backfill/recovery mode).
|
||||||
|
"""
|
||||||
content_types: list[str] = field(default_factory=lambda: ["all"])
|
content_types: list[str] = field(default_factory=lambda: ["all"])
|
||||||
sleep: float | None = None
|
sleep: float | None = None
|
||||||
sleep_request: float | None = None
|
sleep_request: float | None = None
|
||||||
directory_pattern: str | None = None
|
directory_pattern: str | None = None
|
||||||
filename_pattern: str | None = None
|
filename_pattern: str | None = None
|
||||||
skip_existing: bool = True
|
|
||||||
save_metadata: bool = True
|
save_metadata: bool = True
|
||||||
timeout: int = 3600
|
timeout: int = _DEFAULT_GDL_TIMEOUT_SECONDS
|
||||||
|
resume_cursor: str | None = None
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_dict(cls, data: dict) -> SourceConfig:
|
def from_dict(cls, data: dict) -> SourceConfig:
|
||||||
@@ -61,9 +132,8 @@ class SourceConfig:
|
|||||||
sleep_request=data.get("sleep_request"),
|
sleep_request=data.get("sleep_request"),
|
||||||
directory_pattern=data.get("directory_pattern"),
|
directory_pattern=data.get("directory_pattern"),
|
||||||
filename_pattern=data.get("filename_pattern"),
|
filename_pattern=data.get("filename_pattern"),
|
||||||
skip_existing=data.get("skip_existing", True),
|
|
||||||
save_metadata=data.get("save_metadata", True),
|
save_metadata=data.get("save_metadata", True),
|
||||||
timeout=data.get("timeout", 3600),
|
timeout=data.get("timeout", _DEFAULT_GDL_TIMEOUT_SECONDS),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -77,6 +147,12 @@ class DownloadResult:
|
|||||||
files_quarantined: int = 0
|
files_quarantined: int = 0
|
||||||
quarantined_paths: list[str] = field(default_factory=list)
|
quarantined_paths: list[str] = field(default_factory=list)
|
||||||
written_paths: list[str] = field(default_factory=list)
|
written_paths: list[str] = field(default_factory=list)
|
||||||
|
# Native ingester only: post-only sidecar paths for posts that have NO
|
||||||
|
# downloadable media (pure-text posts), so the importer can still upsert the
|
||||||
|
# Post + its body. Empty on the gallery-dl path. Phase 3 imports these via
|
||||||
|
# Importer.upsert_post_record (keyed on external_post_id → updates, never
|
||||||
|
# doubles, the same Post a media import would create).
|
||||||
|
post_record_paths: list[str] = field(default_factory=list)
|
||||||
stdout: str = ""
|
stdout: str = ""
|
||||||
stderr: str = ""
|
stderr: str = ""
|
||||||
return_code: int = 0
|
return_code: int = 0
|
||||||
@@ -85,6 +161,51 @@ class DownloadResult:
|
|||||||
duration_seconds: float = 0.0
|
duration_seconds: float = 0.0
|
||||||
started_at: str | None = None
|
started_at: str | None = None
|
||||||
completed_at: str | None = None
|
completed_at: str | None = None
|
||||||
|
# Plan #704 — structured fields the NATIVE ingester populates directly (None
|
||||||
|
# on the gallery-dl path, which keeps the regex-over-stdout route). When set,
|
||||||
|
# phase 3 reads these instead of scraping the text it would otherwise have to
|
||||||
|
# reconstruct: `run_stats` mirrors _compute_run_stats' shape; `cursor` is the
|
||||||
|
# backfill checkpoint the ingester knows exactly (no parse_last_cursor).
|
||||||
|
run_stats: dict | None = None
|
||||||
|
cursor: str | None = None
|
||||||
|
posts_processed: int = 0
|
||||||
|
# Plan #708 B1 — the server's Retry-After seconds on a RATE_LIMITED result, so
|
||||||
|
# the platform cooldown matches the hint instead of a flat default. None when
|
||||||
|
# unknown (no header, or not a rate-limit failure).
|
||||||
|
retry_after_seconds: float | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def extract_errors_warnings(stderr: str) -> str:
|
||||||
|
"""Keep only the `[error]`/`[warning]` lines from a captured stderr blob.
|
||||||
|
|
||||||
|
A generic download-log helper (module-level so the native-ingester result
|
||||||
|
path can shape its logs without reaching through a GalleryDLService instance).
|
||||||
|
"""
|
||||||
|
if not stderr:
|
||||||
|
return ""
|
||||||
|
kept = [
|
||||||
|
line for line in stderr.splitlines()
|
||||||
|
if "][error]" in line.lower() or "][warning]" in line.lower()
|
||||||
|
]
|
||||||
|
return "\n".join(kept)
|
||||||
|
|
||||||
|
|
||||||
|
def truncate_log(text: str, max_bytes: int = 500_000) -> str:
|
||||||
|
"""Cap a captured log to `max_bytes`, eliding the middle (head + tail kept)."""
|
||||||
|
if not text:
|
||||||
|
return text
|
||||||
|
encoded = text.encode("utf-8")
|
||||||
|
if len(encoded) <= max_bytes:
|
||||||
|
return text
|
||||||
|
half = max_bytes // 2
|
||||||
|
head = encoded[:half].decode("utf-8", errors="ignore")
|
||||||
|
tail = encoded[-half:].decode("utf-8", errors="ignore")
|
||||||
|
head_lines = head.count("\n")
|
||||||
|
tail_lines = tail.count("\n")
|
||||||
|
total_lines = text.count("\n")
|
||||||
|
elided = max(0, total_lines - head_lines - tail_lines)
|
||||||
|
marker = f"\n\n... [{elided} lines elided, {len(encoded) - max_bytes} bytes] ...\n\n"
|
||||||
|
return head + marker + tail
|
||||||
|
|
||||||
|
|
||||||
def _summarize_validation_failures(failures: list[dict]) -> str:
|
def _summarize_validation_failures(failures: list[dict]) -> str:
|
||||||
@@ -101,6 +222,40 @@ def _summarize_validation_failures(failures: list[dict]) -> str:
|
|||||||
return f"{n} files quarantined ({top_count}× {top_reason}, mixed)"
|
return f"{n} files quarantined ({top_count}× {top_reason}, mixed)"
|
||||||
|
|
||||||
|
|
||||||
|
# (parse_last_cursor was removed in plan #704: the native ingester now carries
|
||||||
|
# its checkpoint cursor as a structured DownloadResult.cursor field, so there is
|
||||||
|
# no log text to scrape — and gallery-dl platforms never had a cursor.)
|
||||||
|
|
||||||
|
|
||||||
|
def make_run_stats(
|
||||||
|
*,
|
||||||
|
exit_code: int = 0,
|
||||||
|
downloaded_count: int = 0,
|
||||||
|
skipped_count: int = 0,
|
||||||
|
per_item_failures: int = 0,
|
||||||
|
warning_count: int = 0,
|
||||||
|
tier_gated_count: int = 0,
|
||||||
|
quarantined_count: int = 0,
|
||||||
|
dead_lettered_count: int = 0,
|
||||||
|
) -> dict:
|
||||||
|
"""The canonical `run_stats` dict shape, in ONE place.
|
||||||
|
|
||||||
|
Both result producers — gallery-dl's `_compute_run_stats` (log scrape) and the
|
||||||
|
native ingester's per-outcome tally (ingest_core) — build through this so the
|
||||||
|
key set can't drift between backends. Phase 3 + the Logs UI read these keys.
|
||||||
|
"""
|
||||||
|
return {
|
||||||
|
"exit_code": exit_code,
|
||||||
|
"downloaded_count": downloaded_count,
|
||||||
|
"skipped_count": skipped_count,
|
||||||
|
"per_item_failures": per_item_failures,
|
||||||
|
"warning_count": warning_count,
|
||||||
|
"tier_gated_count": tier_gated_count,
|
||||||
|
"quarantined_count": quarantined_count,
|
||||||
|
"dead_lettered_count": dead_lettered_count,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
class GalleryDLService:
|
class GalleryDLService:
|
||||||
"""Service for executing gallery-dl downloads."""
|
"""Service for executing gallery-dl downloads."""
|
||||||
|
|
||||||
@@ -134,16 +289,10 @@ class GalleryDLService:
|
|||||||
"permission denied", "tier required", "pledge required",
|
"permission denied", "tier required", "pledge required",
|
||||||
]
|
]
|
||||||
|
|
||||||
# Per-platform defaults. Lifted from GS — same six platforms FC supports.
|
# Per-platform defaults for the gallery-dl-backed platforms. Patreon was
|
||||||
|
# removed at the plan-#697 cutover — it now uses the native ingester
|
||||||
|
# (services/patreon_ingester.py), not gallery-dl.
|
||||||
PLATFORM_DEFAULTS = {
|
PLATFORM_DEFAULTS = {
|
||||||
"patreon": {
|
|
||||||
"content_types": ["images", "image_large", "attachments", "postfile", "content"],
|
|
||||||
"directory": ["{date:%Y-%m-%d}_{id}_{title[:40]}"],
|
|
||||||
"filename": "{num:>02}_{filename}.{extension}",
|
|
||||||
"videos": True,
|
|
||||||
"embeds": True,
|
|
||||||
"cursor": True,
|
|
||||||
},
|
|
||||||
"subscribestar": {
|
"subscribestar": {
|
||||||
"content_types": ["all"],
|
"content_types": ["all"],
|
||||||
"directory": ["{date:%Y-%m-%d}_{id}_{title[:40]}"],
|
"directory": ["{date:%Y-%m-%d}_{id}_{title[:40]}"],
|
||||||
@@ -202,7 +351,12 @@ class GalleryDLService:
|
|||||||
"skip": True,
|
"skip": True,
|
||||||
"sleep": self._rate_limit,
|
"sleep": self._rate_limit,
|
||||||
"sleep-request": max(0.5, self._rate_limit / 4),
|
"sleep-request": max(0.5, self._rate_limit / 4),
|
||||||
"retries": 3,
|
# 2 (not 3) retries — a stuck CDN host shouldn't burn a whole
|
||||||
|
# backfill chunk on one file. Anduo #40838 spent ~600s on a
|
||||||
|
# single image (4 extractor HEAD retries × 30s + 4 downloader
|
||||||
|
# GET retries × 120s); halving retries + the downloader
|
||||||
|
# timeout below caps a wedged file at ~1-2 min instead.
|
||||||
|
"retries": 2,
|
||||||
"timeout": 30.0,
|
"timeout": 30.0,
|
||||||
"verify": True,
|
"verify": True,
|
||||||
"postprocessors": [
|
"postprocessors": [
|
||||||
@@ -217,8 +371,17 @@ class GalleryDLService:
|
|||||||
"downloader": {
|
"downloader": {
|
||||||
"part": True,
|
"part": True,
|
||||||
"part-directory": str(self._config_dir / "temp"),
|
"part-directory": str(self._config_dir / "temp"),
|
||||||
"retries": 3,
|
# See the extractor retries note above (Anduo #40838): 2
|
||||||
"timeout": 120.0,
|
# retries + a 60s read-timeout (was 120) so a stalled
|
||||||
|
# connection fails fast. 60s is a per-read timeout, not a
|
||||||
|
# transfer cap — large GIFs that keep streaming are unaffected.
|
||||||
|
"retries": 2,
|
||||||
|
"timeout": 60.0,
|
||||||
|
# NOTE: the Patreon/Mux yt-dlp Referer/Origin forwarding lived
|
||||||
|
# here until the plan-#697 cutover. It was Patreon-specific (and
|
||||||
|
# would have wrongly tagged the other platforms' yt-dlp fetches);
|
||||||
|
# Patreon video is now handled by the native ingester's
|
||||||
|
# downloader (patreon_downloader._VIDEO_HEADERS), so it's gone.
|
||||||
},
|
},
|
||||||
"output": {"progress": True},
|
"output": {"progress": True},
|
||||||
}
|
}
|
||||||
@@ -233,7 +396,18 @@ class GalleryDLService:
|
|||||||
platform: str,
|
platform: str,
|
||||||
source_config: SourceConfig,
|
source_config: SourceConfig,
|
||||||
artist_slug: str,
|
artist_slug: str,
|
||||||
|
skip_value: bool | str = BACKFILL_SKIP_VALUE,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
|
"""`skip_value` controls gallery-dl's archive-walk behavior:
|
||||||
|
- True (BACKFILL_SKIP_VALUE): walk full post history, skipping
|
||||||
|
archived items but continuing past them. Used in backfill mode.
|
||||||
|
- "exit:20" (TICK_SKIP_VALUE): exit gallery-dl after 20
|
||||||
|
contiguous archived items. Used in tick (routine catch-up)
|
||||||
|
mode for fast no-op syncs on creators with deep history.
|
||||||
|
- False: don't skip — redownload everything (not used in FC).
|
||||||
|
The caller (download_service) chooses based on
|
||||||
|
Source.backfill_runs_remaining.
|
||||||
|
"""
|
||||||
config = json.loads(json.dumps(self._get_default_config())) # deep copy
|
config = json.loads(json.dumps(self._get_default_config())) # deep copy
|
||||||
|
|
||||||
destination = str(self.images_root / artist_slug / platform)
|
destination = str(self.images_root / artist_slug / platform)
|
||||||
@@ -243,7 +417,7 @@ class GalleryDLService:
|
|||||||
config["extractor"]["sleep"] = source_config.sleep
|
config["extractor"]["sleep"] = source_config.sleep
|
||||||
if source_config.sleep_request is not None:
|
if source_config.sleep_request is not None:
|
||||||
config["extractor"]["sleep-request"] = source_config.sleep_request
|
config["extractor"]["sleep-request"] = source_config.sleep_request
|
||||||
config["extractor"]["skip"] = source_config.skip_existing
|
config["extractor"]["skip"] = skip_value
|
||||||
|
|
||||||
if source_config.save_metadata:
|
if source_config.save_metadata:
|
||||||
config["extractor"]["postprocessors"] = [
|
config["extractor"]["postprocessors"] = [
|
||||||
@@ -259,14 +433,7 @@ class GalleryDLService:
|
|||||||
|
|
||||||
platform_section = config["extractor"].setdefault(platform, {})
|
platform_section = config["extractor"].setdefault(platform, {})
|
||||||
|
|
||||||
if platform == "patreon":
|
if platform == "hentaifoundry":
|
||||||
if "all" in source_config.content_types:
|
|
||||||
platform_section["files"] = [
|
|
||||||
"images", "image_large", "attachments", "postfile", "content",
|
|
||||||
]
|
|
||||||
else:
|
|
||||||
platform_section["files"] = source_config.content_types
|
|
||||||
elif platform == "hentaifoundry":
|
|
||||||
if "pictures" in source_config.content_types or "all" in source_config.content_types:
|
if "pictures" in source_config.content_types or "all" in source_config.content_types:
|
||||||
platform_section["include"] = "all"
|
platform_section["include"] = "all"
|
||||||
|
|
||||||
@@ -360,7 +527,17 @@ class GalleryDLService:
|
|||||||
if return_code in (1, 4) and (skip_line_count > 0 or has_skip_text) and not has_actual_error:
|
if return_code in (1, 4) and (skip_line_count > 0 or has_skip_text) and not has_actual_error:
|
||||||
return ErrorType.NO_NEW_CONTENT, "No new content to download"
|
return ErrorType.NO_NEW_CONTENT, "No new content to download"
|
||||||
|
|
||||||
if return_code in (1, 4) and not has_actual_error:
|
# Tier-gated classification used to require `return_code in (1, 4)`,
|
||||||
|
# which silently fell through to UNKNOWN_ERROR when gallery-dl
|
||||||
|
# returned a different exit code for mixed-failure runs (e.g.
|
||||||
|
# paywall warnings + a missing yt-dlp dep flipping the exit bits).
|
||||||
|
# The artist then surfaced as "needs attention" purely because a
|
||||||
|
# paywall blocked posts the operator wasn't paying to see —
|
||||||
|
# operator-flagged 2026-05-31. Now: if no source-level error
|
||||||
|
# category fired AND tier-gated warnings are present, classify
|
||||||
|
# as TIER_LIMITED regardless of return code. Same priority order
|
||||||
|
# as before (auth/rate/access/not_found/network/http still win).
|
||||||
|
if not has_actual_error:
|
||||||
tier_gated_lines = [
|
tier_gated_lines = [
|
||||||
line for line in combined.split("\n")
|
line for line in combined.split("\n")
|
||||||
if "][warning]" in line and "not allowed to view post" in line
|
if "][warning]" in line and "not allowed to view post" in line
|
||||||
@@ -372,6 +549,22 @@ class GalleryDLService:
|
|||||||
f"Subscription tier does not grant access to {count} post{'s' if count != 1 else ''}",
|
f"Subscription tier does not grant access to {count} post{'s' if count != 1 else ''}",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Partial-success: the subprocess exited non-zero (typically because
|
||||||
|
# the wall-clock timeout fired mid-walk), but it had downloaded ≥1
|
||||||
|
# file by then and no source-level error category fired. The work
|
||||||
|
# the run DID do is real; gallery-dl's archive will pick up where
|
||||||
|
# it left off on the next tick. Mapped to status="ok" downstream
|
||||||
|
# (download_service.py) so this doesn't flag the source as
|
||||||
|
# "needs attention." Operator-flagged 2026-06-01 after a Knuxy
|
||||||
|
# patreon run downloaded hundreds of files then ran red on timeout.
|
||||||
|
files_downloaded = self._count_downloaded_files(stdout)
|
||||||
|
if not has_actual_error and files_downloaded > 0:
|
||||||
|
return (
|
||||||
|
ErrorType.PARTIAL,
|
||||||
|
f"Downloaded {files_downloaded} file{'s' if files_downloaded != 1 else ''}; "
|
||||||
|
"run did not complete in budget — next tick will continue",
|
||||||
|
)
|
||||||
|
|
||||||
return ErrorType.UNKNOWN_ERROR, f"Unknown error (return code: {return_code})"
|
return ErrorType.UNKNOWN_ERROR, f"Unknown error (return code: {return_code})"
|
||||||
|
|
||||||
def _count_downloaded_files(self, stdout: str) -> int:
|
def _count_downloaded_files(self, stdout: str) -> int:
|
||||||
@@ -400,10 +593,6 @@ class GalleryDLService:
|
|||||||
if not written_paths:
|
if not written_paths:
|
||||||
return quarantined_relpaths, failures
|
return quarantined_relpaths, failures
|
||||||
|
|
||||||
quarantine_root = (
|
|
||||||
self.images_root / "_quarantine" / artist_slug / platform
|
|
||||||
)
|
|
||||||
|
|
||||||
for path in written_paths:
|
for path in written_paths:
|
||||||
if not is_validatable(path):
|
if not is_validatable(path):
|
||||||
continue
|
continue
|
||||||
@@ -414,34 +603,14 @@ class GalleryDLService:
|
|||||||
continue
|
continue
|
||||||
if result.ok:
|
if result.ok:
|
||||||
continue
|
continue
|
||||||
try:
|
# Shared move+sidecar (file_validator.quarantine_file) — same impl the
|
||||||
quarantine_root.mkdir(parents=True, exist_ok=True)
|
# native ingester uses, so the layout + provenance sidecar can't drift.
|
||||||
dest = quarantine_root / path.name
|
dest = quarantine_file(
|
||||||
counter = 1
|
self.images_root, path, artist_slug, platform,
|
||||||
while dest.exists():
|
url=url, result=result,
|
||||||
dest = quarantine_root / f"{path.stem}.{counter}{path.suffix}"
|
)
|
||||||
counter += 1
|
if dest is None:
|
||||||
path.rename(dest)
|
continue # move failed → left in place, not counted
|
||||||
sidecar = dest.with_suffix(dest.suffix + ".quarantine.json")
|
|
||||||
sidecar.write_text(
|
|
||||||
json.dumps(
|
|
||||||
{
|
|
||||||
"original_path": str(path),
|
|
||||||
"source_url": url,
|
|
||||||
"artist_slug": artist_slug,
|
|
||||||
"platform": platform,
|
|
||||||
"format": result.format,
|
|
||||||
"reason": result.reason,
|
|
||||||
"size": result.size,
|
|
||||||
"quarantined_at": datetime.now(UTC).isoformat(),
|
|
||||||
},
|
|
||||||
indent=2,
|
|
||||||
)
|
|
||||||
)
|
|
||||||
except OSError as exc:
|
|
||||||
log.error("Failed to quarantine %s: %s. File left in place.", path, exc)
|
|
||||||
continue
|
|
||||||
|
|
||||||
log.warning(
|
log.warning(
|
||||||
"Quarantined corrupt file: %s → %s (%s: %s)",
|
"Quarantined corrupt file: %s → %s (%s: %s)",
|
||||||
path, dest, result.format, result.reason,
|
path, dest, result.format, result.reason,
|
||||||
@@ -479,41 +648,24 @@ class GalleryDLService:
|
|||||||
if "][warning]" in line.lower() and "not allowed to view post" in line.lower()
|
if "][warning]" in line.lower() and "not allowed to view post" in line.lower()
|
||||||
)
|
)
|
||||||
|
|
||||||
return {
|
return make_run_stats(
|
||||||
"exit_code": return_code,
|
exit_code=return_code,
|
||||||
"downloaded_count": self._count_downloaded_files(stdout),
|
downloaded_count=self._count_downloaded_files(stdout),
|
||||||
"skipped_count": skipped_stdout + skipped_stderr,
|
skipped_count=skipped_stdout + skipped_stderr,
|
||||||
"per_item_failures": per_item_failures,
|
per_item_failures=per_item_failures,
|
||||||
"warning_count": warning_count,
|
warning_count=warning_count,
|
||||||
"tier_gated_count": tier_gated_count,
|
tier_gated_count=tier_gated_count,
|
||||||
}
|
)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _extract_errors_warnings(stderr: str) -> str:
|
def _extract_errors_warnings(stderr: str) -> str:
|
||||||
if not stderr:
|
# Thin delegator to the module-level helper (kept for existing callers).
|
||||||
return ""
|
return extract_errors_warnings(stderr)
|
||||||
kept = [
|
|
||||||
line for line in stderr.splitlines()
|
|
||||||
if "][error]" in line.lower() or "][warning]" in line.lower()
|
|
||||||
]
|
|
||||||
return "\n".join(kept)
|
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _truncate_log(text: str, max_bytes: int = 500_000) -> str:
|
def _truncate_log(text: str, max_bytes: int = 500_000) -> str:
|
||||||
if not text:
|
# Thin delegator to the module-level helper (kept for existing callers).
|
||||||
return text
|
return truncate_log(text, max_bytes)
|
||||||
encoded = text.encode("utf-8")
|
|
||||||
if len(encoded) <= max_bytes:
|
|
||||||
return text
|
|
||||||
half = max_bytes // 2
|
|
||||||
head = encoded[:half].decode("utf-8", errors="ignore")
|
|
||||||
tail = encoded[-half:].decode("utf-8", errors="ignore")
|
|
||||||
head_lines = head.count("\n")
|
|
||||||
tail_lines = tail.count("\n")
|
|
||||||
total_lines = text.count("\n")
|
|
||||||
elided = max(0, total_lines - head_lines - tail_lines)
|
|
||||||
marker = f"\n\n... [{elided} lines elided, {len(encoded) - max_bytes} bytes] ...\n\n"
|
|
||||||
return head + marker + tail
|
|
||||||
|
|
||||||
async def download(
|
async def download(
|
||||||
self,
|
self,
|
||||||
@@ -523,6 +675,7 @@ class GalleryDLService:
|
|||||||
source_config: SourceConfig | None = None,
|
source_config: SourceConfig | None = None,
|
||||||
cookies_path: str | None = None,
|
cookies_path: str | None = None,
|
||||||
auth_token: str | None = None,
|
auth_token: str | None = None,
|
||||||
|
skip_value: bool | str = BACKFILL_SKIP_VALUE,
|
||||||
) -> DownloadResult:
|
) -> DownloadResult:
|
||||||
start_time = time.time()
|
start_time = time.time()
|
||||||
started_at = datetime.now(UTC).isoformat()
|
started_at = datetime.now(UTC).isoformat()
|
||||||
@@ -530,7 +683,9 @@ class GalleryDLService:
|
|||||||
if source_config is None:
|
if source_config is None:
|
||||||
source_config = SourceConfig()
|
source_config = SourceConfig()
|
||||||
|
|
||||||
config = self._build_config_for_source(platform, source_config, artist_slug)
|
config = self._build_config_for_source(
|
||||||
|
platform, source_config, artist_slug, skip_value=skip_value,
|
||||||
|
)
|
||||||
|
|
||||||
if cookies_path:
|
if cookies_path:
|
||||||
config["extractor"]["cookies"] = cookies_path
|
config["extractor"]["cookies"] = cookies_path
|
||||||
@@ -632,13 +787,57 @@ class GalleryDLService:
|
|||||||
started_at=started_at, completed_at=completed_at,
|
started_at=started_at, completed_at=completed_at,
|
||||||
)
|
)
|
||||||
|
|
||||||
except subprocess.TimeoutExpired:
|
except subprocess.TimeoutExpired as e:
|
||||||
duration = time.time() - start_time
|
duration = time.time() - start_time
|
||||||
log.error("Download timeout for %s/%s after %.1fs", artist_slug, platform, duration)
|
# subprocess.run(text=True) makes these str if non-None, but the
|
||||||
|
# caller may have raised TimeoutExpired manually with None or
|
||||||
|
# bytes (tests do); coerce both cases to str.
|
||||||
|
partial_stdout = e.stdout or ""
|
||||||
|
partial_stderr = e.stderr or ""
|
||||||
|
if isinstance(partial_stdout, bytes):
|
||||||
|
partial_stdout = partial_stdout.decode("utf-8", "replace")
|
||||||
|
if isinstance(partial_stderr, bytes):
|
||||||
|
partial_stderr = partial_stderr.decode("utf-8", "replace")
|
||||||
|
|
||||||
|
files_so_far = self._count_downloaded_files(partial_stdout)
|
||||||
|
written_so_far = [str(p) for p in self._written_paths(partial_stdout)]
|
||||||
|
stderr_lines = partial_stderr.strip().splitlines()
|
||||||
|
tail_hint = stderr_lines[-1] if stderr_lines else "no stderr output"
|
||||||
|
|
||||||
|
# If the partial output already shows a rate-limit pattern, the
|
||||||
|
# timeout was almost certainly gallery-dl spinning on retries —
|
||||||
|
# promote to RATE_LIMITED so _update_source_health stamps the
|
||||||
|
# platform cooldown (same code path as a clean-exit rate limit).
|
||||||
|
# Otherwise stay TIMEOUT and let the captured stdout/stderr +
|
||||||
|
# files_so_far tell the operator whether it was "lots of
|
||||||
|
# content" vs "stuck retrying" vs "hung silent".
|
||||||
|
combined = (partial_stdout + "\n" + partial_stderr).lower()
|
||||||
|
if any(p in combined for p in self.RATE_LIMIT_PATTERNS):
|
||||||
|
error_type = ErrorType.RATE_LIMITED
|
||||||
|
error_message = (
|
||||||
|
f"Rate-limited and never completed within "
|
||||||
|
f"{source_config.timeout}s ({files_so_far} files written)"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
error_type = ErrorType.TIMEOUT
|
||||||
|
error_message = (
|
||||||
|
f"Download timed out after {source_config.timeout}s — "
|
||||||
|
f"{files_so_far} file(s) written; last stderr: {tail_hint}"
|
||||||
|
)
|
||||||
|
|
||||||
|
log.error(
|
||||||
|
"Download timeout for %s/%s after %.1fs (%d files written, "
|
||||||
|
"last stderr: %s)",
|
||||||
|
artist_slug, platform, duration, files_so_far, tail_hint,
|
||||||
|
)
|
||||||
|
|
||||||
return DownloadResult(
|
return DownloadResult(
|
||||||
success=False, url=url, artist_slug=artist_slug, platform=platform,
|
success=False, url=url, artist_slug=artist_slug, platform=platform,
|
||||||
error_type=ErrorType.TIMEOUT,
|
files_downloaded=files_so_far,
|
||||||
error_message=f"Download timed out after {source_config.timeout} seconds",
|
written_paths=written_so_far,
|
||||||
|
stdout=partial_stdout, stderr=partial_stderr,
|
||||||
|
return_code=-1, # killed by timeout, no real exit code
|
||||||
|
error_type=error_type, error_message=error_message,
|
||||||
duration_seconds=duration,
|
duration_seconds=duration,
|
||||||
started_at=started_at,
|
started_at=started_at,
|
||||||
completed_at=datetime.now(UTC).isoformat(),
|
completed_at=datetime.now(UTC).isoformat(),
|
||||||
@@ -658,3 +857,70 @@ class GalleryDLService:
|
|||||||
Path(temp_config_path).unlink() # noqa: ASYNC240
|
Path(temp_config_path).unlink() # noqa: ASYNC240
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
async def verify(
|
||||||
|
self,
|
||||||
|
url: str,
|
||||||
|
artist_slug: str,
|
||||||
|
platform: str,
|
||||||
|
source_config: SourceConfig | None = None,
|
||||||
|
cookies_path: str | None = None,
|
||||||
|
auth_token: str | None = None,
|
||||||
|
timeout: float = 45.0, # noqa: ASYNC109 — subprocess.run timeout, not a coroutine deadline
|
||||||
|
) -> tuple[bool, str]:
|
||||||
|
"""Test that credentials authenticate against `url` WITHOUT
|
||||||
|
downloading anything. Runs gallery-dl in --simulate mode limited
|
||||||
|
to the first item; if auth is bad the extractor errors before it
|
||||||
|
can list, which _categorize_error flags as AUTH_ERROR. Returns
|
||||||
|
(ok, message). Used by the credential Verify button."""
|
||||||
|
if source_config is None:
|
||||||
|
source_config = SourceConfig()
|
||||||
|
config = self._build_config_for_source(platform, source_config, artist_slug)
|
||||||
|
if cookies_path:
|
||||||
|
config["extractor"]["cookies"] = cookies_path
|
||||||
|
if auth_token and platform == "discord":
|
||||||
|
config["extractor"].setdefault("discord", {})["token"] = auth_token
|
||||||
|
if auth_token and platform == "pixiv":
|
||||||
|
config["extractor"].setdefault("pixiv", {})["refresh-token"] = auth_token
|
||||||
|
|
||||||
|
with tempfile.NamedTemporaryFile(
|
||||||
|
mode="w", suffix=".json", delete=False, dir=str(self._config_dir),
|
||||||
|
) as fh:
|
||||||
|
json.dump(config, fh, indent=2)
|
||||||
|
temp_config_path = fh.name
|
||||||
|
try:
|
||||||
|
cmd = [
|
||||||
|
sys.executable, "-m", "gallery_dl",
|
||||||
|
"--config", temp_config_path,
|
||||||
|
"--simulate", "--range", "1-1", "--verbose", url,
|
||||||
|
]
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
proc = await loop.run_in_executor(
|
||||||
|
None,
|
||||||
|
lambda: subprocess.run(
|
||||||
|
cmd, capture_output=True, text=True, timeout=timeout,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
etype, msg = self._categorize_error(proc.returncode, proc.stdout, proc.stderr)
|
||||||
|
# TIER_LIMITED proves auth worked — gallery-dl reached the
|
||||||
|
# post, was told it's tier-gated. The download path treats
|
||||||
|
# this as success (line 712); verify must too, or operators
|
||||||
|
# rotate working cookies for no reason. Audit 2026-06-02.
|
||||||
|
if proc.returncode == 0 or etype in (
|
||||||
|
ErrorType.NO_NEW_CONTENT, ErrorType.TIER_LIMITED,
|
||||||
|
):
|
||||||
|
return True, "Credentials valid — the feed authenticated."
|
||||||
|
if etype == ErrorType.AUTH_ERROR:
|
||||||
|
return False, msg
|
||||||
|
# Network / not-found / rate-limit / unknown: inconclusive,
|
||||||
|
# not a definitive credential failure. Surface the reason.
|
||||||
|
return False, f"Could not confirm ({etype.value}): {msg}"
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
return False, f"Verification timed out after {timeout:.0f}s"
|
||||||
|
except Exception as exc: # noqa: BLE001
|
||||||
|
return False, f"Verification error: {exc}"
|
||||||
|
finally:
|
||||||
|
try:
|
||||||
|
Path(temp_config_path).unlink() # noqa: ASYNC240
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|||||||
@@ -14,44 +14,39 @@ Decoding rejects malformed cursors with a ValueError; the API layer
|
|||||||
translates that to HTTP 400.
|
translates that to HTTP 400.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import base64
|
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
from urllib.parse import quote
|
||||||
|
|
||||||
from sqlalchemy import Select, and_, exists, func, or_, select
|
from sqlalchemy import Select, and_, distinct, exists, func, or_, select
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
from sqlalchemy.orm import aliased
|
||||||
|
|
||||||
from ..models import Artist, ImageProvenance, ImageRecord, Post, Source, Tag
|
from ..models import Artist, ImageProvenance, ImageRecord, Post, Source, Tag
|
||||||
from ..models.tag import image_tag
|
from ..models.tag import image_tag
|
||||||
|
from .pagination import decode_cursor, encode_cursor
|
||||||
|
from .tag_query import fandom_join_alias, serialize_tag, tag_columns
|
||||||
|
|
||||||
CURSOR_SEPARATOR = "|"
|
# Reserved `platform` filter value selecting images with NO platformed
|
||||||
|
# provenance (filesystem imports). Returned by facets() as a null-valued
|
||||||
|
# bucket; the frontend maps that null back to this sentinel in the URL so the
|
||||||
def encode_cursor(effective_date: datetime, image_id: int) -> str:
|
# bucket is selectable. Underscore-wrapped so it can't collide with a real
|
||||||
raw = f"{effective_date.isoformat()}{CURSOR_SEPARATOR}{image_id}"
|
# gallery-dl platform name (patreon/pixiv/...).
|
||||||
return base64.urlsafe_b64encode(raw.encode()).decode()
|
UNSOURCED_PLATFORM = "__unsourced__"
|
||||||
|
|
||||||
|
|
||||||
def decode_cursor(cursor: str) -> tuple[datetime, int]:
|
|
||||||
try:
|
|
||||||
raw = base64.urlsafe_b64decode(cursor.encode()).decode()
|
|
||||||
ts_part, id_part = raw.split(CURSOR_SEPARATOR, 1)
|
|
||||||
return datetime.fromisoformat(ts_part), int(id_part)
|
|
||||||
except Exception as exc:
|
|
||||||
raise ValueError(f"invalid cursor: {cursor!r}") from exc
|
|
||||||
|
|
||||||
|
|
||||||
def _effective_date_col():
|
def _effective_date_col():
|
||||||
"""SQL expression: COALESCE(post.post_date, image_record.created_at).
|
"""The materialized gallery sort key: image_record.effective_date
|
||||||
|
(alembic 0035) = COALESCE(primary post's post_date, created_at),
|
||||||
|
maintained at write time by the importer.
|
||||||
|
|
||||||
Used as the canonical sort/group/filter key across the gallery so
|
Canonical sort/group/filter key across the gallery so images attached
|
||||||
images backfilled with primary_post_id (e.g. via tag_apply phase 4)
|
to a post surface at their original publish date, not their FC import
|
||||||
surface at their original publish date, not their FC import date.
|
date — and, now that it's a single indexed column rather than a
|
||||||
Images without a Post (or with Post.post_date NULL) fall back to
|
COALESCE across the Post outer join, the cursor scroll is an index
|
||||||
image_record.created_at and still order coherently against
|
range scan instead of a full re-sort per page.
|
||||||
post-attached ones.
|
|
||||||
"""
|
"""
|
||||||
return func.coalesce(Post.post_date, ImageRecord.created_at)
|
return ImageRecord.effective_date
|
||||||
|
|
||||||
|
|
||||||
def _outer_join_primary_post(stmt: Select) -> Select:
|
def _outer_join_primary_post(stmt: Select) -> Select:
|
||||||
@@ -90,21 +85,136 @@ class TimelineBucket:
|
|||||||
count: int
|
count: int
|
||||||
|
|
||||||
|
|
||||||
def thumbnail_url(sha256_hex: str, mime: str) -> str:
|
@dataclass(frozen=True)
|
||||||
# Quart serves /images/* via the frontend blueprint (FC-1); thumbnails go
|
class GalleryFacets:
|
||||||
# under /images/thumbs/. The MIME determines the extension.
|
total: int # images matching the FULL active filter
|
||||||
|
platforms: list[dict] # [{"value": str|None, "count": int}], null = unsourced
|
||||||
|
untagged: int # how many the Untagged flag would isolate
|
||||||
|
no_artist: int # how many the No-artist flag would isolate
|
||||||
|
date_min: datetime | None
|
||||||
|
date_max: datetime | None
|
||||||
|
|
||||||
|
|
||||||
|
def thumbnail_url(thumbnail_path: str | None, sha256_hex: str, mime: str) -> str:
|
||||||
|
"""Return the URL to fetch a thumbnail.
|
||||||
|
|
||||||
|
Prefers the stored thumbnail_path verbatim — Quart serves /images/*
|
||||||
|
1:1 from the volume (frontend.py:20-36), so the URL IS the disk
|
||||||
|
path. Falls back to deriving from (sha256, mime) only when the
|
||||||
|
record's thumbnail_path is NULL (thumbnailer hasn't run yet); that
|
||||||
|
URL will 404 until backfill catches it, same as before the path
|
||||||
|
was tracked.
|
||||||
|
|
||||||
|
Pre-2026-05-30 this was derived only from (sha256, mime), which
|
||||||
|
disagreed with the actual on-disk extension when the thumbnailer
|
||||||
|
chose its format from transparency rather than MIME — every PNG
|
||||||
|
source without alpha (extension was .jpg on disk) and every WebP
|
||||||
|
source with alpha (extension was .png on disk) silently 404'd
|
||||||
|
despite the thumbnail file existing.
|
||||||
|
"""
|
||||||
|
if thumbnail_path:
|
||||||
|
return thumbnail_path
|
||||||
|
# Fallback for records with no thumbnail recorded yet — preserves
|
||||||
|
# prior behavior (URL exists but 404s until backfill regenerates).
|
||||||
ext = ".png" if mime in ("image/png", "image/gif") else ".jpg"
|
ext = ".png" if mime in ("image/png", "image/gif") else ".jpg"
|
||||||
bucket = sha256_hex[:3]
|
bucket = sha256_hex[:3]
|
||||||
return f"/images/thumbs/{bucket}/{sha256_hex}{ext}"
|
return f"/images/thumbs/{bucket}/{sha256_hex}{ext}"
|
||||||
|
|
||||||
|
|
||||||
def _require_single_filter(tag_id, post_id, artist_id) -> None:
|
def image_url(path: str) -> str:
|
||||||
if sum(x is not None for x in (tag_id, post_id, artist_id)) > 1:
|
"""Return the URL to fetch the full-size original from /images.
|
||||||
|
|
||||||
|
The on-disk `path` mirrors the source tree (artist/post folders), so it can
|
||||||
|
contain characters that are special in a URL — most importantly '#' (post
|
||||||
|
titles like 'BLUE#59'), but also spaces, '?', '%'. The serve_image route
|
||||||
|
URL-decodes its <path:subpath> fine, but only if the browser sends the whole
|
||||||
|
path; an unencoded '#' is parsed as a fragment, so '#59/01_timelapse.jpg'
|
||||||
|
never reaches the server and the original 404s while the (hash-named)
|
||||||
|
thumbnail still loads. Percent-encode the path, keeping '/' as the segment
|
||||||
|
separator. Operator-flagged 2026-06-12."""
|
||||||
|
rel = path.split("/images/", 1)[-1]
|
||||||
|
return f"/images/{quote(rel, safe='/')}"
|
||||||
|
|
||||||
|
|
||||||
|
def _require_single_filter(tag_ids, post_id, artist_id) -> None:
|
||||||
|
"""post_id is the post-detail view — it can't combine with the
|
||||||
|
composable filters. tag_ids + artist_id (+ media_type) compose freely
|
||||||
|
(AND)."""
|
||||||
|
if post_id is not None and (tag_ids or artist_id is not None):
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
"tag_id, post_id, artist_id are mutually exclusive"
|
"post_id cannot be combined with tag or artist filters"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _apply_scope(
|
||||||
|
stmt, *, tag_ids, post_id, artist_id, media_type,
|
||||||
|
platform=None, untagged=False, no_artist=False,
|
||||||
|
date_from=None, date_to=None,
|
||||||
|
):
|
||||||
|
"""Apply the composable gallery filters to a statement.
|
||||||
|
|
||||||
|
All clauses are correlated EXISTS / scalar predicates on ImageRecord, so
|
||||||
|
they AND together without row-multiplication and don't require any join to
|
||||||
|
be present on `stmt` (the artist/platform paths alias Post/Source inside
|
||||||
|
their own EXISTS).
|
||||||
|
|
||||||
|
- tag_ids: image must carry ALL of them — one correlated EXISTS per tag.
|
||||||
|
- post_id / artist_id: provenance EXISTS (post_id is exclusive, guarded
|
||||||
|
by _require_single_filter).
|
||||||
|
- media_type: 'image' | 'video' narrows by mime prefix.
|
||||||
|
- platform: EXISTS a provenance→source with that platform; the
|
||||||
|
UNSOURCED_PLATFORM sentinel inverts it (NO platformed provenance).
|
||||||
|
- untagged: NOT EXISTS any image_tag row.
|
||||||
|
- no_artist: ImageRecord.artist_id IS NULL.
|
||||||
|
- date_from / date_to: half-open [from, to) bounds on effective_date.
|
||||||
|
"""
|
||||||
|
for tid in tag_ids or []:
|
||||||
|
stmt = stmt.where(
|
||||||
|
exists().where(
|
||||||
|
image_tag.c.image_record_id == ImageRecord.id,
|
||||||
|
image_tag.c.tag_id == tid,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
prov = _provenance_clause(post_id, artist_id)
|
||||||
|
if prov is not None:
|
||||||
|
stmt = stmt.where(prov)
|
||||||
|
if media_type == "image":
|
||||||
|
stmt = stmt.where(ImageRecord.mime.like("image/%"))
|
||||||
|
elif media_type == "video":
|
||||||
|
stmt = stmt.where(ImageRecord.mime.like("video/%"))
|
||||||
|
if platform is not None:
|
||||||
|
stmt = stmt.where(_platform_clause(platform))
|
||||||
|
if untagged:
|
||||||
|
stmt = stmt.where(
|
||||||
|
~exists().where(image_tag.c.image_record_id == ImageRecord.id)
|
||||||
|
)
|
||||||
|
if no_artist:
|
||||||
|
stmt = stmt.where(ImageRecord.artist_id.is_(None))
|
||||||
|
eff = _effective_date_col()
|
||||||
|
if date_from is not None:
|
||||||
|
stmt = stmt.where(eff >= date_from)
|
||||||
|
if date_to is not None:
|
||||||
|
stmt = stmt.where(eff < date_to)
|
||||||
|
return stmt
|
||||||
|
|
||||||
|
|
||||||
|
def _platform_clause(platform):
|
||||||
|
"""Correlated EXISTS on a provenance row whose Source carries `platform`.
|
||||||
|
The UNSOURCED_PLATFORM sentinel inverts to NOT EXISTS(any sourced
|
||||||
|
provenance) — i.e. filesystem-imported content with no platform."""
|
||||||
|
src = aliased(Source)
|
||||||
|
if platform == UNSOURCED_PLATFORM:
|
||||||
|
return ~exists().where(
|
||||||
|
ImageProvenance.image_record_id == ImageRecord.id,
|
||||||
|
ImageProvenance.source_id == src.id,
|
||||||
|
)
|
||||||
|
return exists().where(
|
||||||
|
ImageProvenance.image_record_id == ImageRecord.id,
|
||||||
|
ImageProvenance.source_id == src.id,
|
||||||
|
src.platform == platform,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _provenance_clause(post_id, artist_id):
|
def _provenance_clause(post_id, artist_id):
|
||||||
"""Correlated EXISTS clause (NOT a join) so an image with multiple
|
"""Correlated EXISTS clause (NOT a join) so an image with multiple
|
||||||
matching provenance rows is returned exactly once and the
|
matching provenance rows is returned exactly once and the
|
||||||
@@ -115,14 +225,48 @@ def _provenance_clause(post_id, artist_id):
|
|||||||
ImageProvenance.post_id == post_id,
|
ImageProvenance.post_id == post_id,
|
||||||
)
|
)
|
||||||
if artist_id is not None:
|
if artist_id is not None:
|
||||||
|
# Use Post.artist_id (alembic 0030 denormalized column) instead
|
||||||
|
# of joining through ImageProvenance.source_id → Source.artist_id.
|
||||||
|
# The denormalization is the always-present linkage; the source
|
||||||
|
# path now drops NULL-source provenance rows (filesystem-imported
|
||||||
|
# content) which would otherwise vanish from artist-filtered
|
||||||
|
# gallery views.
|
||||||
|
# ALIAS Post: the gallery query outer-joins Post on
|
||||||
|
# ImageRecord.primary_post_id (`_outer_join_primary_post`).
|
||||||
|
# SQLAlchemy would otherwise correlate a bare `Post` reference
|
||||||
|
# in this EXISTS subquery to that outer Post (which is NULL for
|
||||||
|
# images with no primary post), and the filter would silently
|
||||||
|
# match nothing.
|
||||||
|
post_inner = aliased(Post)
|
||||||
return exists().where(
|
return exists().where(
|
||||||
ImageProvenance.image_record_id == ImageRecord.id,
|
ImageProvenance.image_record_id == ImageRecord.id,
|
||||||
ImageProvenance.source_id == Source.id,
|
ImageProvenance.post_id == post_inner.id,
|
||||||
Source.artist_id == artist_id,
|
post_inner.artist_id == artist_id,
|
||||||
)
|
)
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _gallery_images(rows, artists: dict[int, dict]) -> list[GalleryImage]:
|
||||||
|
"""Build GalleryImage list from (record, posted_at, eff_date) rows + the
|
||||||
|
artist hydration map. Shared by scroll() and similar()."""
|
||||||
|
return [
|
||||||
|
GalleryImage(
|
||||||
|
id=record.id,
|
||||||
|
path=record.path,
|
||||||
|
sha256=record.sha256,
|
||||||
|
mime=record.mime,
|
||||||
|
width=record.width,
|
||||||
|
height=record.height,
|
||||||
|
created_at=record.created_at,
|
||||||
|
effective_date=eff_date,
|
||||||
|
posted_at=posted_at,
|
||||||
|
thumbnail_url=thumbnail_url(record.thumbnail_path, record.sha256, record.mime),
|
||||||
|
artist=artists.get(record.id),
|
||||||
|
)
|
||||||
|
for record, posted_at, eff_date in rows
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
async def _artists_for(session, image_ids: list[int]) -> dict[int, dict]:
|
async def _artists_for(session, image_ids: list[int]) -> dict[int, dict]:
|
||||||
"""Map image_id -> {"name","slug"} via the canonical
|
"""Map image_id -> {"name","slug"} via the canonical
|
||||||
image_record.artist_id (FC-2d-vii-c). Bounded by page size."""
|
image_record.artist_id (FC-2d-vii-c). Bounded by page size."""
|
||||||
@@ -147,35 +291,50 @@ class GalleryService:
|
|||||||
self,
|
self,
|
||||||
cursor: str | None,
|
cursor: str | None,
|
||||||
limit: int = 50,
|
limit: int = 50,
|
||||||
tag_id: int | None = None,
|
tag_ids: list[int] | None = None,
|
||||||
post_id: int | None = None,
|
post_id: int | None = None,
|
||||||
artist_id: int | None = None,
|
artist_id: int | None = None,
|
||||||
|
media_type: str | None = None,
|
||||||
|
sort: str = "newest",
|
||||||
|
platform: str | None = None,
|
||||||
|
untagged: bool = False,
|
||||||
|
no_artist: bool = False,
|
||||||
|
date_from: datetime | None = None,
|
||||||
|
date_to: datetime | None = None,
|
||||||
) -> GalleryPage:
|
) -> GalleryPage:
|
||||||
if limit < 1 or limit > 200:
|
if limit < 1 or limit > 200:
|
||||||
raise ValueError("limit must be between 1 and 200")
|
raise ValueError("limit must be between 1 and 200")
|
||||||
_require_single_filter(tag_id, post_id, artist_id)
|
_require_single_filter(tag_ids, post_id, artist_id)
|
||||||
|
|
||||||
eff = _effective_date_col()
|
eff = _effective_date_col()
|
||||||
stmt = select(ImageRecord, Post.post_date, eff.label("eff"))
|
stmt = select(ImageRecord, Post.post_date, eff.label("eff"))
|
||||||
stmt = _outer_join_primary_post(stmt)
|
stmt = _outer_join_primary_post(stmt)
|
||||||
if tag_id is not None:
|
stmt = _apply_scope(
|
||||||
stmt = stmt.join(image_tag, image_tag.c.image_record_id == ImageRecord.id).where(
|
stmt, tag_ids=tag_ids, post_id=post_id,
|
||||||
image_tag.c.tag_id == tag_id
|
artist_id=artist_id, media_type=media_type,
|
||||||
)
|
platform=platform, untagged=untagged, no_artist=no_artist,
|
||||||
prov = _provenance_clause(post_id, artist_id)
|
date_from=date_from, date_to=date_to,
|
||||||
if prov is not None:
|
)
|
||||||
stmt = stmt.where(prov)
|
|
||||||
|
|
||||||
|
descending = sort != "oldest"
|
||||||
if cursor:
|
if cursor:
|
||||||
cur_ts, cur_id = decode_cursor(cursor)
|
cur_ts, cur_id = decode_cursor(cursor)
|
||||||
stmt = stmt.where(
|
# The cursor is just (last eff, last id); the request's sort
|
||||||
or_(
|
# decides which side of it the next page lies on.
|
||||||
eff < cur_ts,
|
if descending:
|
||||||
and_(eff == cur_ts, ImageRecord.id < cur_id),
|
stmt = stmt.where(
|
||||||
|
or_(eff < cur_ts, and_(eff == cur_ts, ImageRecord.id < cur_id))
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
stmt = stmt.where(
|
||||||
|
or_(eff > cur_ts, and_(eff == cur_ts, ImageRecord.id > cur_id))
|
||||||
)
|
)
|
||||||
)
|
|
||||||
|
|
||||||
stmt = stmt.order_by(eff.desc(), ImageRecord.id.desc()).limit(limit + 1)
|
if descending:
|
||||||
|
stmt = stmt.order_by(eff.desc(), ImageRecord.id.desc())
|
||||||
|
else:
|
||||||
|
stmt = stmt.order_by(eff.asc(), ImageRecord.id.asc())
|
||||||
|
stmt = stmt.limit(limit + 1)
|
||||||
rows = (await self.session.execute(stmt)).all()
|
rows = (await self.session.execute(stmt)).all()
|
||||||
|
|
||||||
next_cursor = None
|
next_cursor = None
|
||||||
@@ -187,22 +346,7 @@ class GalleryService:
|
|||||||
artists = await _artists_for(
|
artists = await _artists_for(
|
||||||
self.session, [r[0].id for r in rows]
|
self.session, [r[0].id for r in rows]
|
||||||
)
|
)
|
||||||
images = [
|
images = _gallery_images(rows, artists)
|
||||||
GalleryImage(
|
|
||||||
id=record.id,
|
|
||||||
path=record.path,
|
|
||||||
sha256=record.sha256,
|
|
||||||
mime=record.mime,
|
|
||||||
width=record.width,
|
|
||||||
height=record.height,
|
|
||||||
created_at=record.created_at,
|
|
||||||
effective_date=eff_date,
|
|
||||||
posted_at=posted_at,
|
|
||||||
thumbnail_url=thumbnail_url(record.sha256, record.mime),
|
|
||||||
artist=artists.get(record.id),
|
|
||||||
)
|
|
||||||
for record, posted_at, eff_date in rows
|
|
||||||
]
|
|
||||||
return GalleryPage(
|
return GalleryPage(
|
||||||
images=images,
|
images=images,
|
||||||
next_cursor=next_cursor,
|
next_cursor=next_cursor,
|
||||||
@@ -211,9 +355,15 @@ class GalleryService:
|
|||||||
|
|
||||||
async def timeline(
|
async def timeline(
|
||||||
self,
|
self,
|
||||||
tag_id: int | None = None,
|
tag_ids: list[int] | None = None,
|
||||||
post_id: int | None = None,
|
post_id: int | None = None,
|
||||||
artist_id: int | None = None,
|
artist_id: int | None = None,
|
||||||
|
media_type: str | None = None,
|
||||||
|
platform: str | None = None,
|
||||||
|
untagged: bool = False,
|
||||||
|
no_artist: bool = False,
|
||||||
|
date_from: datetime | None = None,
|
||||||
|
date_to: datetime | None = None,
|
||||||
) -> list[TimelineBucket]:
|
) -> list[TimelineBucket]:
|
||||||
eff = _effective_date_col()
|
eff = _effective_date_col()
|
||||||
year_col = func.date_part("year", eff).label("yr")
|
year_col = func.date_part("year", eff).label("yr")
|
||||||
@@ -222,25 +372,28 @@ class GalleryService:
|
|||||||
year_col, month_col, func.count(ImageRecord.id).label("cnt")
|
year_col, month_col, func.count(ImageRecord.id).label("cnt")
|
||||||
)
|
)
|
||||||
stmt = _outer_join_primary_post(stmt)
|
stmt = _outer_join_primary_post(stmt)
|
||||||
_require_single_filter(tag_id, post_id, artist_id)
|
_require_single_filter(tag_ids, post_id, artist_id)
|
||||||
if tag_id is not None:
|
stmt = _apply_scope(
|
||||||
stmt = stmt.join(image_tag, image_tag.c.image_record_id == ImageRecord.id).where(
|
stmt, tag_ids=tag_ids, post_id=post_id,
|
||||||
image_tag.c.tag_id == tag_id
|
artist_id=artist_id, media_type=media_type,
|
||||||
)
|
platform=platform, untagged=untagged, no_artist=no_artist,
|
||||||
prov = _provenance_clause(post_id, artist_id)
|
date_from=date_from, date_to=date_to,
|
||||||
if prov is not None:
|
)
|
||||||
stmt = stmt.where(prov)
|
|
||||||
stmt = stmt.group_by(year_col, month_col).order_by(year_col.desc(), month_col.desc())
|
stmt = stmt.group_by(year_col, month_col).order_by(year_col.desc(), month_col.desc())
|
||||||
rows = (await self.session.execute(stmt)).all()
|
rows = (await self.session.execute(stmt)).all()
|
||||||
return [TimelineBucket(year=int(r.yr), month=int(r.mo), count=int(r.cnt)) for r in rows]
|
return [TimelineBucket(year=int(r.yr), month=int(r.mo), count=int(r.cnt)) for r in rows]
|
||||||
|
|
||||||
async def jump_cursor(
|
async def jump_cursor(
|
||||||
self, year: int, month: int, tag_id: int | None = None,
|
self, year: int, month: int, tag_ids: list[int] | None = None,
|
||||||
post_id: int | None = None, artist_id: int | None = None,
|
post_id: int | None = None, artist_id: int | None = None,
|
||||||
|
media_type: str | None = None, sort: str = "newest",
|
||||||
|
platform: str | None = None, untagged: bool = False,
|
||||||
|
no_artist: bool = False, date_from: datetime | None = None,
|
||||||
|
date_to: datetime | None = None,
|
||||||
) -> str | None:
|
) -> str | None:
|
||||||
"""Returns a cursor that, when passed to scroll(), positions at the
|
"""Returns a cursor that, when passed to scroll() with the same sort,
|
||||||
first image of the given year-month (by effective_date, not
|
positions at the first image of the given year-month. None if the
|
||||||
created_at). None if the bucket is empty.
|
bucket is empty.
|
||||||
"""
|
"""
|
||||||
from sqlalchemy import extract
|
from sqlalchemy import extract
|
||||||
|
|
||||||
@@ -250,34 +403,176 @@ class GalleryService:
|
|||||||
extract("month", eff) == month,
|
extract("month", eff) == month,
|
||||||
)
|
)
|
||||||
stmt = _outer_join_primary_post(stmt)
|
stmt = _outer_join_primary_post(stmt)
|
||||||
_require_single_filter(tag_id, post_id, artist_id)
|
_require_single_filter(tag_ids, post_id, artist_id)
|
||||||
if tag_id is not None:
|
stmt = _apply_scope(
|
||||||
stmt = stmt.join(image_tag, image_tag.c.image_record_id == ImageRecord.id).where(
|
stmt, tag_ids=tag_ids, post_id=post_id,
|
||||||
image_tag.c.tag_id == tag_id
|
artist_id=artist_id, media_type=media_type,
|
||||||
)
|
platform=platform, untagged=untagged, no_artist=no_artist,
|
||||||
prov = _provenance_clause(post_id, artist_id)
|
date_from=date_from, date_to=date_to,
|
||||||
if prov is not None:
|
)
|
||||||
stmt = stmt.where(prov)
|
descending = sort != "oldest"
|
||||||
stmt = stmt.order_by(eff.desc(), ImageRecord.id.desc()).limit(1)
|
if descending:
|
||||||
first = (await self.session.execute(stmt)).first()
|
stmt = stmt.order_by(eff.desc(), ImageRecord.id.desc())
|
||||||
|
else:
|
||||||
|
stmt = stmt.order_by(eff.asc(), ImageRecord.id.asc())
|
||||||
|
first = (await self.session.execute(stmt.limit(1))).first()
|
||||||
if first is None:
|
if first is None:
|
||||||
return None
|
return None
|
||||||
record, eff_date = first
|
record, eff_date = first
|
||||||
# Cursor is exclusive; we encode a cursor with id+1 so the row itself
|
# Cursor is exclusive; nudge the id one past the boundary row (in the
|
||||||
# is the first result in the next scroll().
|
# scan direction) so the row itself is the first result of scroll().
|
||||||
return encode_cursor(eff_date, record.id + 1)
|
boundary = record.id + 1 if descending else record.id - 1
|
||||||
|
return encode_cursor(eff_date, boundary)
|
||||||
|
|
||||||
|
async def facets(
|
||||||
|
self, *, tag_ids: list[int] | None = None,
|
||||||
|
post_id: int | None = None, artist_id: int | None = None,
|
||||||
|
media_type: str | None = None, platform: str | None = None,
|
||||||
|
untagged: bool = False, no_artist: bool = False,
|
||||||
|
date_from: datetime | None = None, date_to: datetime | None = None,
|
||||||
|
) -> GalleryFacets:
|
||||||
|
"""Live facet counts scoped to the current filter. Each facet GROUP is
|
||||||
|
computed with all OTHER active filters applied but its OWN selection
|
||||||
|
ignored ("minus-self"), so sibling options stay visible/switchable.
|
||||||
|
No outer join is needed — every clause is a correlated EXISTS or a
|
||||||
|
column predicate on ImageRecord.
|
||||||
|
"""
|
||||||
|
_require_single_filter(tag_ids, post_id, artist_id)
|
||||||
|
common = {
|
||||||
|
"tag_ids": tag_ids, "post_id": post_id,
|
||||||
|
"artist_id": artist_id, "media_type": media_type,
|
||||||
|
}
|
||||||
|
|
||||||
|
# total — the full active filter (the headline result count).
|
||||||
|
total = (await self.session.execute(
|
||||||
|
_apply_scope(
|
||||||
|
select(func.count(ImageRecord.id)), **common,
|
||||||
|
platform=platform, untagged=untagged, no_artist=no_artist,
|
||||||
|
date_from=date_from, date_to=date_to,
|
||||||
|
)
|
||||||
|
)).scalar_one()
|
||||||
|
|
||||||
|
# platforms — scope minus the platform selection. Inner-join
|
||||||
|
# provenance→source and COUNT(DISTINCT image) per platform (a
|
||||||
|
# cross-posted image counts under each of its platforms).
|
||||||
|
plat_scope = {
|
||||||
|
**common, "untagged": untagged, "no_artist": no_artist,
|
||||||
|
"date_from": date_from, "date_to": date_to,
|
||||||
|
}
|
||||||
|
src = aliased(Source)
|
||||||
|
plat_stmt = (
|
||||||
|
select(src.platform, func.count(distinct(ImageRecord.id)))
|
||||||
|
.select_from(ImageRecord)
|
||||||
|
.join(ImageProvenance, ImageProvenance.image_record_id == ImageRecord.id)
|
||||||
|
.join(src, src.id == ImageProvenance.source_id)
|
||||||
|
)
|
||||||
|
plat_stmt = _apply_scope(plat_stmt, **plat_scope).group_by(src.platform)
|
||||||
|
platforms = [
|
||||||
|
{"value": p, "count": c}
|
||||||
|
for p, c in (await self.session.execute(plat_stmt)).all()
|
||||||
|
]
|
||||||
|
# Unsourced (filesystem) bucket — same minus-platform scope.
|
||||||
|
unsourced = (await self.session.execute(
|
||||||
|
_apply_scope(
|
||||||
|
select(func.count(ImageRecord.id)), **plat_scope,
|
||||||
|
platform=UNSOURCED_PLATFORM,
|
||||||
|
)
|
||||||
|
)).scalar_one()
|
||||||
|
if unsourced:
|
||||||
|
platforms.append({"value": None, "count": unsourced})
|
||||||
|
|
||||||
|
# curation flags — each minus its OWN flag.
|
||||||
|
untagged_count = (await self.session.execute(
|
||||||
|
_apply_scope(
|
||||||
|
select(func.count(ImageRecord.id)), **common,
|
||||||
|
platform=platform, no_artist=no_artist,
|
||||||
|
date_from=date_from, date_to=date_to, untagged=True,
|
||||||
|
)
|
||||||
|
)).scalar_one()
|
||||||
|
no_artist_count = (await self.session.execute(
|
||||||
|
_apply_scope(
|
||||||
|
select(func.count(ImageRecord.id)), **common,
|
||||||
|
platform=platform, untagged=untagged,
|
||||||
|
date_from=date_from, date_to=date_to, no_artist=True,
|
||||||
|
)
|
||||||
|
)).scalar_one()
|
||||||
|
|
||||||
|
# date bounds — scope minus the date params (those drive the picker).
|
||||||
|
eff = _effective_date_col()
|
||||||
|
dmin, dmax = (await self.session.execute(
|
||||||
|
_apply_scope(
|
||||||
|
select(func.min(eff), func.max(eff)), **common,
|
||||||
|
platform=platform, untagged=untagged, no_artist=no_artist,
|
||||||
|
)
|
||||||
|
)).one()
|
||||||
|
|
||||||
|
return GalleryFacets(
|
||||||
|
total=total, platforms=platforms,
|
||||||
|
untagged=untagged_count, no_artist=no_artist_count,
|
||||||
|
date_min=dmin, date_max=dmax,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def similar(
|
||||||
|
self, image_id: int, limit: int = 100, *,
|
||||||
|
tag_ids: list[int] | None = None, artist_id: int | None = None,
|
||||||
|
media_type: str | None = None, platform: str | None = None,
|
||||||
|
untagged: bool = False, no_artist: bool = False,
|
||||||
|
date_from: datetime | None = None, date_to: datetime | None = None,
|
||||||
|
) -> list[GalleryImage] | None:
|
||||||
|
"""Visual "more like this": images ranked by cosine distance to
|
||||||
|
`image_id`'s SigLIP embedding (pgvector, HNSW-indexed — alembic 0036).
|
||||||
|
No ML inference here; the embedding was computed at import.
|
||||||
|
|
||||||
|
Returns None if the source image doesn't exist (→ 404), [] if it has
|
||||||
|
no embedding (a video / not-yet-embedded). Composes with the Phase-1/2
|
||||||
|
scope filters (AND) but REPLACES the date sort — always nearest-first,
|
||||||
|
bounded to `limit` (no cursor; distance-ranking has no date cursor).
|
||||||
|
"""
|
||||||
|
if limit < 1 or limit > 200:
|
||||||
|
raise ValueError("limit must be between 1 and 200")
|
||||||
|
src = await self.session.get(ImageRecord, image_id)
|
||||||
|
if src is None:
|
||||||
|
return None
|
||||||
|
if src.siglip_embedding is None:
|
||||||
|
return []
|
||||||
|
|
||||||
|
distance = ImageRecord.siglip_embedding.cosine_distance(src.siglip_embedding)
|
||||||
|
eff = _effective_date_col()
|
||||||
|
stmt = select(ImageRecord, Post.post_date, eff.label("eff"))
|
||||||
|
stmt = _outer_join_primary_post(stmt)
|
||||||
|
stmt = stmt.where(
|
||||||
|
ImageRecord.siglip_embedding.is_not(None),
|
||||||
|
ImageRecord.id != image_id,
|
||||||
|
)
|
||||||
|
stmt = _apply_scope(
|
||||||
|
stmt, tag_ids=tag_ids, post_id=None,
|
||||||
|
artist_id=artist_id, media_type=media_type,
|
||||||
|
platform=platform, untagged=untagged, no_artist=no_artist,
|
||||||
|
date_from=date_from, date_to=date_to,
|
||||||
|
)
|
||||||
|
stmt = stmt.order_by(distance.asc()).limit(limit)
|
||||||
|
rows = (await self.session.execute(stmt)).all()
|
||||||
|
artists = await _artists_for(self.session, [r[0].id for r in rows])
|
||||||
|
return _gallery_images(rows, artists)
|
||||||
|
|
||||||
async def get_image_with_tags(self, image_id: int) -> dict | None:
|
async def get_image_with_tags(self, image_id: int) -> dict | None:
|
||||||
record = await self.session.get(ImageRecord, image_id)
|
record = await self.session.get(ImageRecord, image_id)
|
||||||
if record is None:
|
if record is None:
|
||||||
return None
|
return None
|
||||||
|
# Self-join Tag to resolve a character's fandom NAME (not just id) so the
|
||||||
|
# modal chip can label it without an N+1 (shared tag_query helpers).
|
||||||
|
fandom_alias = fandom_join_alias()
|
||||||
tag_stmt = (
|
tag_stmt = (
|
||||||
select(Tag)
|
select(*tag_columns(fandom_alias))
|
||||||
.join(image_tag, image_tag.c.tag_id == Tag.id)
|
.select_from(
|
||||||
|
Tag.__table__
|
||||||
|
.join(image_tag, image_tag.c.tag_id == Tag.id)
|
||||||
|
.outerjoin(fandom_alias, Tag.fandom_id == fandom_alias.c.id)
|
||||||
|
)
|
||||||
.where(image_tag.c.image_record_id == image_id)
|
.where(image_tag.c.image_record_id == image_id)
|
||||||
.order_by(Tag.kind.asc(), Tag.name.asc())
|
.order_by(Tag.kind.asc(), Tag.name.asc())
|
||||||
)
|
)
|
||||||
tags = (await self.session.execute(tag_stmt)).scalars().all()
|
tags = (await self.session.execute(tag_stmt)).all()
|
||||||
# Fetch the canonical post.post_date for this image (if any) so
|
# Fetch the canonical post.post_date for this image (if any) so
|
||||||
# the modal can show "Posted on <date>" alongside import date.
|
# the modal can show "Posted on <date>" alongside import date.
|
||||||
posted_at = None
|
posted_at = None
|
||||||
@@ -304,38 +599,26 @@ class GalleryService:
|
|||||||
"height": record.height,
|
"height": record.height,
|
||||||
"size_bytes": record.size_bytes,
|
"size_bytes": record.size_bytes,
|
||||||
"integrity_status": record.integrity_status,
|
"integrity_status": record.integrity_status,
|
||||||
|
# Phase 3: lets the modal hide the "Related"/find-similar surface
|
||||||
|
# for images that have no embedding yet (videos / pending ML).
|
||||||
|
"has_embedding": record.siglip_embedding is not None,
|
||||||
"created_at": record.created_at.isoformat(),
|
"created_at": record.created_at.isoformat(),
|
||||||
"posted_at": posted_at.isoformat() if posted_at else None,
|
"posted_at": posted_at.isoformat() if posted_at else None,
|
||||||
"thumbnail_url": thumbnail_url(record.sha256, record.mime),
|
"thumbnail_url": thumbnail_url(record.thumbnail_path, record.sha256, record.mime),
|
||||||
"image_url": f"/images/{record.path.split('/images/', 1)[-1]}",
|
"image_url": image_url(record.path),
|
||||||
"artist": (
|
"artist": (
|
||||||
{"id": artist.id, "name": artist.name, "slug": artist.slug}
|
{"id": artist.id, "name": artist.name, "slug": artist.slug}
|
||||||
if artist is not None else None
|
if artist is not None else None
|
||||||
),
|
),
|
||||||
"tags": [
|
"tags": [serialize_tag(t) for t in tags],
|
||||||
{
|
|
||||||
"id": t.id,
|
|
||||||
"name": t.name,
|
|
||||||
"kind": t.kind.value if hasattr(t.kind, "value") else t.kind,
|
|
||||||
"fandom_id": t.fandom_id,
|
|
||||||
}
|
|
||||||
for t in tags
|
|
||||||
],
|
|
||||||
"neighbors": neighbors,
|
"neighbors": neighbors,
|
||||||
}
|
}
|
||||||
|
|
||||||
async def _neighbors(self, record: ImageRecord) -> dict:
|
async def _neighbors(self, record: ImageRecord) -> dict:
|
||||||
# Compute the boundary image's effective_date in Python (one query
|
# The boundary image's sort key is materialized on the row now
|
||||||
# below + the SELECT we already have on `record`) and use it for
|
# (alembic 0035) — read it directly instead of re-deriving COALESCE
|
||||||
# the neighbor comparison. Cheaper than re-deriving in SQL via
|
# via an extra Post lookup.
|
||||||
# correlated subquery.
|
boundary_eff = record.effective_date
|
||||||
boundary_eff = record.created_at
|
|
||||||
if record.primary_post_id is not None:
|
|
||||||
post_date = (await self.session.execute(
|
|
||||||
select(Post.post_date).where(Post.id == record.primary_post_id)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if post_date is not None:
|
|
||||||
boundary_eff = post_date
|
|
||||||
|
|
||||||
eff = _effective_date_col()
|
eff = _effective_date_col()
|
||||||
prev_stmt = _outer_join_primary_post(
|
prev_stmt = _outer_join_primary_post(
|
||||||
|
|||||||
+372
-146
@@ -23,6 +23,7 @@ from sqlalchemy.orm import Session
|
|||||||
|
|
||||||
from ..models import (
|
from ..models import (
|
||||||
Artist,
|
Artist,
|
||||||
|
ExternalLink,
|
||||||
ImageProvenance,
|
ImageProvenance,
|
||||||
ImageRecord,
|
ImageRecord,
|
||||||
ImportSettings,
|
ImportSettings,
|
||||||
@@ -31,12 +32,19 @@ from ..models import (
|
|||||||
Source,
|
Source,
|
||||||
)
|
)
|
||||||
from ..utils import safe_probe
|
from ..utils import safe_probe
|
||||||
from ..utils.paths import derive_subdir, derive_top_level_artist, hash_suffixed_name
|
from ..utils.paths import (
|
||||||
|
derive_subdir,
|
||||||
|
derive_top_level_artist,
|
||||||
|
filehash_from_url,
|
||||||
|
hash_suffixed_name,
|
||||||
|
safe_ext,
|
||||||
|
)
|
||||||
from ..utils.phash import compute_phash, find_similar
|
from ..utils.phash import compute_phash, find_similar
|
||||||
from ..utils.sidecar import find_sidecar, parse_sidecar
|
from ..utils.sidecar import find_sidecar, parse_sidecar
|
||||||
from ..utils.slug import slugify
|
from ..utils.slug import slugify
|
||||||
from .archive_extractor import extract_archive, is_archive
|
from .archive_extractor import extract_archive, is_archive
|
||||||
from .attachment_store import AttachmentStore
|
from .attachment_store import AttachmentStore
|
||||||
|
from .link_extract import extract_external_links
|
||||||
from .thumbnailer import Thumbnailer
|
from .thumbnailer import Thumbnailer
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
@@ -82,27 +90,9 @@ def is_video(path: Path) -> bool:
|
|||||||
|
|
||||||
def _safe_ext(path: Path) -> str:
|
def _safe_ext(path: Path) -> str:
|
||||||
"""Conservatively extract a file extension for PostAttachment.ext
|
"""Conservatively extract a file extension for PostAttachment.ext
|
||||||
(varchar(32)).
|
(varchar(32)). Thin wrapper over the shared `utils.paths.safe_ext` (kept for
|
||||||
|
the Path-typed call sites + the [[path_suffix_sanitize]] memory pointer)."""
|
||||||
gallery-dl produces some filenames with URL-encoded query-string
|
return safe_ext(path)
|
||||||
artifacts embedded into the basename (e.g.
|
|
||||||
`79507046_media_..._https___www.patreon.com_media-u_Z0FBQUFBQm5q...`).
|
|
||||||
`Path.suffix` finds the LAST dot and returns everything after, which
|
|
||||||
in those cases yields a 50+ char "extension" of mostly base64-ish
|
|
||||||
junk. That blows the column. Operator-flagged 2026-05-25.
|
|
||||||
|
|
||||||
Real extensions are short and alphanumeric. We accept anything ≤ 16
|
|
||||||
chars where every post-dot character is alphanumeric; anything else
|
|
||||||
means the input wasn't a real extension and we return the empty
|
|
||||||
string. ext is nullable-ish (empty string still satisfies NOT NULL)
|
|
||||||
and consumers should treat "" as "no known extension".
|
|
||||||
"""
|
|
||||||
suffix = path.suffix.lower()
|
|
||||||
if not suffix or len(suffix) > 16:
|
|
||||||
return ""
|
|
||||||
if not all(c.isalnum() for c in suffix[1:]):
|
|
||||||
return ""
|
|
||||||
return suffix
|
|
||||||
|
|
||||||
|
|
||||||
def _mime_for(path: Path) -> str:
|
def _mime_for(path: Path) -> str:
|
||||||
@@ -204,6 +194,32 @@ class Importer:
|
|||||||
(phash, width or 0, height or 0, image_id)
|
(phash, width or 0, height or 0, image_id)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def _get_or_create(self, stmt, factory):
|
||||||
|
"""Race-safe find-or-create. Run `stmt` (scalar_one_or_none); if a
|
||||||
|
row exists, return it. Otherwise open a savepoint and INSERT
|
||||||
|
``factory()``; on IntegrityError (a concurrent worker inserted the
|
||||||
|
same row first) roll the savepoint back — NOT the outer transaction,
|
||||||
|
which would lose the surrounding scan's progress — and re-run `stmt`
|
||||||
|
(scalar_one) to return the row the other worker created.
|
||||||
|
|
||||||
|
Centralizes the pattern shared by _find_or_create_source and
|
||||||
|
_find_or_create_post. The plain SELECT-then-INSERT version lost
|
||||||
|
races under the 5-min recovery sweep (operator-flagged
|
||||||
|
2026-05-26)."""
|
||||||
|
existing = self.session.execute(stmt).scalar_one_or_none()
|
||||||
|
if existing is not None:
|
||||||
|
return existing
|
||||||
|
sp = self.session.begin_nested()
|
||||||
|
try:
|
||||||
|
row = factory()
|
||||||
|
self.session.add(row)
|
||||||
|
self.session.flush()
|
||||||
|
sp.commit()
|
||||||
|
return row
|
||||||
|
except IntegrityError:
|
||||||
|
sp.rollback()
|
||||||
|
return self.session.execute(stmt).scalar_one()
|
||||||
|
|
||||||
def _find_or_create_source(
|
def _find_or_create_source(
|
||||||
self, *, artist_id: int, platform: str, url: str,
|
self, *, artist_id: int, platform: str, url: str,
|
||||||
) -> Source:
|
) -> Source:
|
||||||
@@ -222,53 +238,35 @@ class Importer:
|
|||||||
and re-select — the concurrent op just created the row we
|
and re-select — the concurrent op just created the row we
|
||||||
wanted, so the second select will find it.
|
wanted, so the second select will find it.
|
||||||
"""
|
"""
|
||||||
existing = self.session.execute(
|
stmt = select(Source).where(
|
||||||
select(Source).where(
|
Source.artist_id == artist_id,
|
||||||
Source.artist_id == artist_id,
|
Source.platform == platform,
|
||||||
Source.platform == platform,
|
Source.url == url,
|
||||||
Source.url == url,
|
)
|
||||||
)
|
return self._get_or_create(
|
||||||
).scalar_one_or_none()
|
stmt,
|
||||||
if existing is not None:
|
lambda: Source(artist_id=artist_id, platform=platform, url=url),
|
||||||
return existing
|
)
|
||||||
sp = self.session.begin_nested()
|
|
||||||
try:
|
|
||||||
row = Source(artist_id=artist_id, platform=platform, url=url)
|
|
||||||
self.session.add(row)
|
|
||||||
self.session.flush()
|
|
||||||
sp.commit()
|
|
||||||
return row
|
|
||||||
except IntegrityError:
|
|
||||||
sp.rollback()
|
|
||||||
return self.session.execute(
|
|
||||||
select(Source).where(
|
|
||||||
Source.artist_id == artist_id,
|
|
||||||
Source.platform == platform,
|
|
||||||
Source.url == url,
|
|
||||||
)
|
|
||||||
).scalar_one()
|
|
||||||
|
|
||||||
def _source_for_sidecar(
|
def _lookup_source_for_sidecar(
|
||||||
self, *, artist_id: int, platform: str, artist_slug: str,
|
self, *, artist_id: int, platform: str,
|
||||||
) -> Source:
|
) -> Source | None:
|
||||||
"""Filesystem-import sidecar Source resolver.
|
"""Find the real subscription Source for (artist, platform), or
|
||||||
|
None if no subscription exists.
|
||||||
|
|
||||||
Source represents a subscription feed (one per artist+platform — the
|
Pre-alembic-0030 this method would CREATE a synthetic
|
||||||
gallery-dl URL polled by the FC-3 downloader). The filesystem importer
|
`sidecar:<platform>:<slug>` Source when no real one existed —
|
||||||
used to call _find_or_create_source(url=sd.post_url), which created
|
because `Post.source_id` was NOT NULL and the importer needed
|
||||||
one Source row per post URL — 100s of junk Sources per artist, all
|
something to attach Posts to. Alembic 0030 relaxed both
|
||||||
with enabled=True, polluting the artist detail page and tricking the
|
`Post.source_id` and `ImageProvenance.source_id` to nullable, so
|
||||||
subscription checker into trying to poll patreon post URLs as feeds.
|
synthetic anchors are obsolete; the importer now leaves
|
||||||
Operator-flagged 2026-05-26.
|
source_id as None when no subscription exists for the (artist,
|
||||||
|
platform). Operator-asked 2026-06-01: synthetic Sources had
|
||||||
New behaviour: if any Source row exists for (artist_id, platform),
|
leaked into the Subscriptions UI as phantom subscriptions and
|
||||||
reuse it regardless of its URL — the artist's real subscription Source
|
the operator wanted the data model to truthfully say "this
|
||||||
(created by the downloader / extension / UI) is the canonical
|
content has no live subscription."
|
||||||
attachment point for filesystem-imported posts. If none exists, create
|
|
||||||
ONE synthetic anchor with url='sidecar:<platform>:<artist_slug>' and
|
|
||||||
enabled=False (so the subscription checker doesn't poll it).
|
|
||||||
"""
|
"""
|
||||||
existing = self.session.execute(
|
stmt = (
|
||||||
select(Source)
|
select(Source)
|
||||||
.where(
|
.where(
|
||||||
Source.artist_id == artist_id,
|
Source.artist_id == artist_id,
|
||||||
@@ -276,63 +274,65 @@ class Importer:
|
|||||||
)
|
)
|
||||||
.order_by(Source.id.asc())
|
.order_by(Source.id.asc())
|
||||||
.limit(1)
|
.limit(1)
|
||||||
).scalar_one_or_none()
|
)
|
||||||
if existing is not None:
|
return self.session.execute(stmt).scalar_one_or_none()
|
||||||
return existing
|
|
||||||
synthetic_url = f"sidecar:{platform}:{artist_slug}"
|
|
||||||
sp = self.session.begin_nested()
|
|
||||||
try:
|
|
||||||
row = Source(
|
|
||||||
artist_id=artist_id,
|
|
||||||
platform=platform,
|
|
||||||
url=synthetic_url,
|
|
||||||
enabled=False,
|
|
||||||
)
|
|
||||||
self.session.add(row)
|
|
||||||
self.session.flush()
|
|
||||||
sp.commit()
|
|
||||||
return row
|
|
||||||
except IntegrityError:
|
|
||||||
sp.rollback()
|
|
||||||
return self.session.execute(
|
|
||||||
select(Source)
|
|
||||||
.where(
|
|
||||||
Source.artist_id == artist_id,
|
|
||||||
Source.platform == platform,
|
|
||||||
)
|
|
||||||
.order_by(Source.id.asc())
|
|
||||||
.limit(1)
|
|
||||||
).scalar_one()
|
|
||||||
|
|
||||||
def _find_or_create_post(
|
def _find_or_create_post(
|
||||||
self, *, source_id: int, external_post_id: str,
|
self, *, source_id: int | None, external_post_id: str,
|
||||||
|
artist_id: int,
|
||||||
) -> Post:
|
) -> Post:
|
||||||
"""Race-safe find-or-create on `post` keyed by
|
"""Race-safe find-or-create on `post`. Keyed by
|
||||||
(source_id, external_post_id). Mirrors `_find_or_create_source`
|
(source_id, external_post_id) when source_id is set — the
|
||||||
— same savepoint + IntegrityError-recovery pattern."""
|
`uq_post_source_external_id` constraint guards. For NULL-source
|
||||||
existing = self.session.execute(
|
posts the existence check matches on (artist_id, external_post_id),
|
||||||
select(Post).where(
|
which the partial unique index `uq_post_artist_external_id_null_source`
|
||||||
|
(alembic 0030) guards. Same savepoint + IntegrityError-recovery
|
||||||
|
pattern as the rest of the helpers."""
|
||||||
|
if source_id is not None:
|
||||||
|
stmt = select(Post).where(
|
||||||
Post.source_id == source_id,
|
Post.source_id == source_id,
|
||||||
Post.external_post_id == external_post_id,
|
Post.external_post_id == external_post_id,
|
||||||
)
|
)
|
||||||
).scalar_one_or_none()
|
else:
|
||||||
if existing is not None:
|
stmt = select(Post).where(
|
||||||
return existing
|
Post.source_id.is_(None),
|
||||||
sp = self.session.begin_nested()
|
Post.artist_id == artist_id,
|
||||||
try:
|
Post.external_post_id == external_post_id,
|
||||||
row = Post(source_id=source_id, external_post_id=external_post_id)
|
)
|
||||||
self.session.add(row)
|
return self._get_or_create(
|
||||||
self.session.flush()
|
stmt,
|
||||||
sp.commit()
|
lambda: Post(
|
||||||
return row
|
source_id=source_id,
|
||||||
except IntegrityError:
|
artist_id=artist_id,
|
||||||
sp.rollback()
|
external_post_id=external_post_id,
|
||||||
return self.session.execute(
|
),
|
||||||
select(Post).where(
|
)
|
||||||
Post.source_id == source_id,
|
|
||||||
Post.external_post_id == external_post_id,
|
def _sync_external_links(self, post: Post) -> None:
|
||||||
)
|
"""Record off-platform file-host links (mega/gdrive/mediafire/dropbox/
|
||||||
).scalar_one()
|
pixeldrain) found in the post body, so they're never silently dropped
|
||||||
|
and the download worker can fetch them later. Shared by every platform's
|
||||||
|
import (runs off Post.description). INSERT-MISSING only — an existing row
|
||||||
|
keeps its status/attempts (a re-import must not reset a link already
|
||||||
|
downloaded or dead-lettered). Identity is (post_id, url); the full url
|
||||||
|
incl. #fragment is preserved by the extractor."""
|
||||||
|
links = extract_external_links(post.description)
|
||||||
|
if not links:
|
||||||
|
return
|
||||||
|
self.session.flush() # ensure post.id is assigned before we reference it
|
||||||
|
existing = set(self.session.execute(
|
||||||
|
select(ExternalLink.url).where(ExternalLink.post_id == post.id)
|
||||||
|
).scalars().all())
|
||||||
|
for link in links:
|
||||||
|
if link.url in existing:
|
||||||
|
continue
|
||||||
|
self.session.add(ExternalLink(
|
||||||
|
post_id=post.id,
|
||||||
|
artist_id=post.artist_id,
|
||||||
|
host=link.host,
|
||||||
|
url=link.url,
|
||||||
|
label=link.label,
|
||||||
|
))
|
||||||
|
|
||||||
def import_one(self, source: Path) -> ImportResult:
|
def import_one(self, source: Path) -> ImportResult:
|
||||||
"""Dispatch by kind. Media → normal pipeline. Archive → extract
|
"""Dispatch by kind. Media → normal pipeline. Archive → extract
|
||||||
@@ -372,12 +372,14 @@ class Importer:
|
|||||||
return None
|
return None
|
||||||
sd = parse_sidecar(data)
|
sd = parse_sidecar(data)
|
||||||
platform = sd.platform or "unknown"
|
platform = sd.platform or "unknown"
|
||||||
src = self._source_for_sidecar(
|
src = self._lookup_source_for_sidecar(
|
||||||
artist_id=artist.id, platform=platform, artist_slug=artist.slug,
|
artist_id=artist.id, platform=platform,
|
||||||
)
|
)
|
||||||
epid = sd.external_post_id or sc.stem
|
epid = sd.external_post_id or sc.stem
|
||||||
return self._find_or_create_post(
|
return self._find_or_create_post(
|
||||||
source_id=src.id, external_post_id=epid,
|
source_id=src.id if src else None,
|
||||||
|
external_post_id=epid,
|
||||||
|
artist_id=artist.id,
|
||||||
)
|
)
|
||||||
|
|
||||||
def _capture_attachment(
|
def _capture_attachment(
|
||||||
@@ -388,13 +390,39 @@ class Importer:
|
|||||||
artist = self._resolve_artist(source)
|
artist = self._resolve_artist(source)
|
||||||
post = self._post_for_sidecar(source, artist)
|
post = self._post_for_sidecar(source, artist)
|
||||||
sha = _sha256_of(source)
|
sha = _sha256_of(source)
|
||||||
existing = self.session.execute(
|
post_id = post.id if post else None
|
||||||
select(PostAttachment).where(PostAttachment.sha256 == sha)
|
# Dedup is PER-POST, not global: a non-art file the creator attaches to
|
||||||
).scalar_one_or_none()
|
# many posts (a standard pdf/zip/link-card) must get a row on EVERY post
|
||||||
if existing is None:
|
# so none is left a bare shell. The on-disk blob stays sha-deduped
|
||||||
|
# (attachments.store is sha-addressed + idempotent); only the rows are
|
||||||
|
# per-post. PostAttachment's unique is partial (post_id, sha256) for
|
||||||
|
# real posts and (sha256) for the NULL-post filesystem case. Before
|
||||||
|
# 2026-06-08 the global UNIQUE(sha256) left every post after the first
|
||||||
|
# with no attachment row → 1589 empty Anduo shells.
|
||||||
|
if post_id is not None:
|
||||||
|
select_existing = select(PostAttachment).where(
|
||||||
|
PostAttachment.post_id == post_id,
|
||||||
|
PostAttachment.sha256 == sha,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
select_existing = select(PostAttachment).where(
|
||||||
|
PostAttachment.post_id.is_(None),
|
||||||
|
PostAttachment.sha256 == sha,
|
||||||
|
)
|
||||||
|
existing = self.session.execute(select_existing).scalar_one_or_none()
|
||||||
|
if existing is not None:
|
||||||
|
self.session.commit()
|
||||||
|
return ImportResult(status="attached")
|
||||||
|
# Savepoint + IntegrityError recovery — the partial UNIQUE means two
|
||||||
|
# workers can both pass the SELECT and only the second INSERT fails.
|
||||||
|
# Without savepoint, the outer transaction poisons and the calling task
|
||||||
|
# crashes. attachments.store is sha-addressed so both workers race to
|
||||||
|
# write the same target path; shutil.copy2 + rename is idempotent.
|
||||||
|
sp = self.session.begin_nested()
|
||||||
|
try:
|
||||||
stored = self.attachments.store(source, sha)
|
stored = self.attachments.store(source, sha)
|
||||||
self.session.add(PostAttachment(
|
self.session.add(PostAttachment(
|
||||||
post_id=post.id if post else None,
|
post_id=post_id,
|
||||||
artist_id=artist.id if artist else None,
|
artist_id=artist.id if artist else None,
|
||||||
sha256=sha,
|
sha256=sha,
|
||||||
path=stored,
|
path=stored,
|
||||||
@@ -404,10 +432,19 @@ class Importer:
|
|||||||
size_bytes=source.stat().st_size,
|
size_bytes=source.stat().st_size,
|
||||||
))
|
))
|
||||||
self.session.flush()
|
self.session.flush()
|
||||||
|
sp.commit()
|
||||||
|
except IntegrityError:
|
||||||
|
sp.rollback()
|
||||||
|
# Lost the race — the other worker's row for this (post, sha) wins.
|
||||||
|
self.session.execute(select_existing).scalar_one()
|
||||||
self.session.commit()
|
self.session.commit()
|
||||||
return ImportResult(status="attached")
|
return ImportResult(status="attached")
|
||||||
|
|
||||||
def _import_archive(self, source: Path) -> ImportResult:
|
def _import_archive(
|
||||||
|
self, source: Path, *,
|
||||||
|
artist: Artist | None = None,
|
||||||
|
source_row: Source | None = None,
|
||||||
|
) -> ImportResult:
|
||||||
# Layer-3 isolation: bomb-size guard + integrity test in a
|
# Layer-3 isolation: bomb-size guard + integrity test in a
|
||||||
# spawned child BEFORE extracting in this process. A
|
# spawned child BEFORE extracting in this process. A
|
||||||
# decompression bomb or a native-lib crash on a malformed
|
# decompression bomb or a native-lib crash on a malformed
|
||||||
@@ -415,6 +452,14 @@ class Importer:
|
|||||||
# instead of OOMing/segfaulting the import worker. extract_archive
|
# instead of OOMing/segfaulting the import worker. extract_archive
|
||||||
# is already fail-soft for plain exceptions, so this only adds
|
# is already fail-soft for plain exceptions, so this only adds
|
||||||
# the hard-crash protection.
|
# the hard-crash protection.
|
||||||
|
#
|
||||||
|
# Audit 2026-06-02: optional artist/source_row kwargs let the
|
||||||
|
# download path thread its explicit subscription context
|
||||||
|
# through instead of having _resolve_artist re-derive from
|
||||||
|
# path-walk (which works by coincidence today because gallery-dl
|
||||||
|
# lays files out under /images/<artist_slug>/...). Filesystem
|
||||||
|
# import still calls bare _import_archive(source) and falls
|
||||||
|
# back to the path-walk derivation as before.
|
||||||
probe = safe_probe.probe_archive(source)
|
probe = safe_probe.probe_archive(source)
|
||||||
if not probe.ok:
|
if not probe.ok:
|
||||||
if probe.crashed:
|
if probe.crashed:
|
||||||
@@ -426,34 +471,55 @@ class Importer:
|
|||||||
# still preserve the archive file itself as an attachment so
|
# still preserve the archive file itself as an attachment so
|
||||||
# nothing silently vanishes, matching extract_archive's
|
# nothing silently vanishes, matching extract_archive's
|
||||||
# fail-soft contract.
|
# fail-soft contract.
|
||||||
artist = self._resolve_artist(source)
|
artist_use = artist if artist is not None else self._resolve_artist(source)
|
||||||
post = self._post_for_sidecar(source, artist)
|
post = self._post_for_sidecar(source, artist_use)
|
||||||
self._capture_attachment(source, post=post, artist=artist, resolved=True)
|
self._capture_attachment(
|
||||||
return ImportResult(status="attached")
|
source, post=post, artist=artist_use, resolved=True,
|
||||||
|
)
|
||||||
|
reason = f"archive probe rejected, captured unextracted: {probe.reason}"
|
||||||
|
log.warning("%s: %s", source.name, reason)
|
||||||
|
return ImportResult(status="attached", error=reason)
|
||||||
|
|
||||||
artist = self._resolve_artist(source)
|
artist_use = artist if artist is not None else self._resolve_artist(source)
|
||||||
post = self._post_for_sidecar(source, artist)
|
post = self._post_for_sidecar(source, artist_use)
|
||||||
member_ids: list[int] = []
|
member_ids: list[int] = []
|
||||||
|
member_total = 0
|
||||||
with extract_archive(source) as members:
|
with extract_archive(source) as members:
|
||||||
for _name, member_path in members:
|
for _name, member_path in members:
|
||||||
|
member_total += 1
|
||||||
if not is_supported(member_path):
|
if not is_supported(member_path):
|
||||||
continue # non-media preserved via the stored archive
|
continue # non-media preserved via the stored archive
|
||||||
res = self._import_media(member_path, source)
|
res = self._import_media(
|
||||||
|
member_path, source, explicit_source=source_row,
|
||||||
|
)
|
||||||
if res.status in ("imported", "superseded") and res.image_id:
|
if res.status in ("imported", "superseded") and res.image_id:
|
||||||
member_ids.append(res.image_id)
|
member_ids.append(res.image_id)
|
||||||
# Preserve the archive itself (links to the same Post/Artist).
|
# Preserve the archive itself (links to the same Post/Artist).
|
||||||
self._capture_attachment(
|
self._capture_attachment(
|
||||||
source, post=post, artist=artist, resolved=True
|
source, post=post, artist=artist_use, resolved=True
|
||||||
)
|
)
|
||||||
if member_ids:
|
if member_ids:
|
||||||
return ImportResult(
|
return ImportResult(
|
||||||
status="imported", image_id=member_ids[0],
|
status="imported", image_id=member_ids[0],
|
||||||
member_image_ids=member_ids,
|
member_image_ids=member_ids,
|
||||||
)
|
)
|
||||||
return ImportResult(status="attached")
|
# No images landed — surface WHY so a post showing "no images" beside an
|
||||||
|
# archive is diagnosable instead of silent. Zero members usually means
|
||||||
|
# the extractor backend is missing/failed (unar for rar, py7zr for 7z)
|
||||||
|
# or the file is corrupt; non-zero-but-no-images means it held only
|
||||||
|
# non-media files.
|
||||||
|
reason = (
|
||||||
|
"archive extracted but held no supported image/video members"
|
||||||
|
if member_total
|
||||||
|
else "archive yielded no members (unsupported/corrupt, or the "
|
||||||
|
"extractor backend failed)"
|
||||||
|
)
|
||||||
|
log.warning("%s: %s", source.name, reason)
|
||||||
|
return ImportResult(status="attached", error=reason)
|
||||||
|
|
||||||
def _import_media(
|
def _import_media(
|
||||||
self, source: Path, attribution_path: Path
|
self, source: Path, attribution_path: Path,
|
||||||
|
*, explicit_source: Source | None = None,
|
||||||
) -> ImportResult:
|
) -> ImportResult:
|
||||||
"""The media import pipeline (filters, dedup, copy, provenance).
|
"""The media import pipeline (filters, dedup, copy, provenance).
|
||||||
|
|
||||||
@@ -535,6 +601,15 @@ class Importer:
|
|||||||
existing = self.session.execute(existing_stmt).scalar_one_or_none()
|
existing = self.session.execute(existing_stmt).scalar_one_or_none()
|
||||||
if existing:
|
if existing:
|
||||||
if not self.deep:
|
if not self.deep:
|
||||||
|
# Enrich-on-duplicate (parity with attach_in_place): a re-scanned
|
||||||
|
# file that is a byte-dup of an existing image still links its
|
||||||
|
# post via _apply_sidecar, so a cross-posted image shows on every
|
||||||
|
# post. Artist is derived from the sidecar here (not yet resolved).
|
||||||
|
self._apply_sidecar(
|
||||||
|
existing, attribution_path, None,
|
||||||
|
explicit_source=explicit_source,
|
||||||
|
)
|
||||||
|
self.session.commit()
|
||||||
return ImportResult(
|
return ImportResult(
|
||||||
status="skipped", skip_reason=SkipReason.duplicate_hash,
|
status="skipped", skip_reason=SkipReason.duplicate_hash,
|
||||||
image_id=existing.id, error="sha256 already present",
|
image_id=existing.id, error="sha256 already present",
|
||||||
@@ -561,6 +636,14 @@ class Importer:
|
|||||||
candidates, self.settings.phash_threshold,
|
candidates, self.settings.phash_threshold,
|
||||||
)
|
)
|
||||||
if rel == "larger_exists":
|
if rel == "larger_exists":
|
||||||
|
# Enrich-on-duplicate (parity with attach_in_place).
|
||||||
|
larger = self.session.get(ImageRecord, match_id)
|
||||||
|
if larger is not None:
|
||||||
|
self._apply_sidecar(
|
||||||
|
larger, attribution_path, None,
|
||||||
|
explicit_source=explicit_source,
|
||||||
|
)
|
||||||
|
self.session.commit()
|
||||||
return ImportResult(
|
return ImportResult(
|
||||||
status="skipped",
|
status="skipped",
|
||||||
skip_reason=SkipReason.duplicate_phash,
|
skip_reason=SkipReason.duplicate_phash,
|
||||||
@@ -600,7 +683,15 @@ class Importer:
|
|||||||
artist = self._attach_artist(record, artist_name)
|
artist = self._attach_artist(record, artist_name)
|
||||||
|
|
||||||
# Sidecar provenance (best-effort; never fails the import).
|
# Sidecar provenance (best-effort; never fails the import).
|
||||||
self._apply_sidecar(record, attribution_path, artist)
|
# explicit_source lets the FC-3c download path bind the new
|
||||||
|
# ImageProvenance row to its subscription Source instead of
|
||||||
|
# having _apply_sidecar re-derive via _lookup_source_for_sidecar.
|
||||||
|
# Audit 2026-06-02 — archive members extracted from a
|
||||||
|
# subscription-downloaded zip previously lost subscription
|
||||||
|
# linkage if the on-disk layout didn't match assumptions.
|
||||||
|
self._apply_sidecar(
|
||||||
|
record, attribution_path, artist, explicit_source=explicit_source,
|
||||||
|
)
|
||||||
|
|
||||||
# Thumbnail is queued separately by the calling task; the importer
|
# Thumbnail is queued separately by the calling task; the importer
|
||||||
# does not generate thumbnails inline so the import queue stays moving.
|
# does not generate thumbnails inline so the import queue stays moving.
|
||||||
@@ -648,6 +739,67 @@ class Importer:
|
|||||||
self.session.commit()
|
self.session.commit()
|
||||||
return ImportResult(status="refreshed", image_id=existing.id)
|
return ImportResult(status="refreshed", image_id=existing.id)
|
||||||
|
|
||||||
|
def upsert_post_record(
|
||||||
|
self, sidecar: Path, *, artist: Artist | None = None,
|
||||||
|
source: Source | None = None,
|
||||||
|
) -> bool:
|
||||||
|
"""Upsert the Post for a post-ONLY sidecar (a media-less post), so the
|
||||||
|
artist archive includes text posts (their body + external links).
|
||||||
|
|
||||||
|
Reuses `_find_or_create_post` (keyed on external_post_id) so it UPDATES
|
||||||
|
the SAME Post a media import would create — never doubles. Fields are
|
||||||
|
FILLED, never clobbered with empty: parse_sidecar yields None for empty
|
||||||
|
values, and a None field is left untouched (an empty feed body never
|
||||||
|
wipes a populated one). Returns True if a Post was upserted, False if the
|
||||||
|
sidecar was unusable (parse failure / no artist)."""
|
||||||
|
try:
|
||||||
|
data = json.loads(sidecar.read_text("utf-8"))
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
raise ValueError("sidecar JSON is not an object")
|
||||||
|
except Exception as exc:
|
||||||
|
log.warning("post-record sidecar parse failed for %s: %s", sidecar, exc)
|
||||||
|
return False
|
||||||
|
|
||||||
|
sd = parse_sidecar(data)
|
||||||
|
|
||||||
|
if artist is None:
|
||||||
|
name = self._sidecar_artist_name(data)
|
||||||
|
artist = self._upsert_artist(name) if name else None
|
||||||
|
if artist is None:
|
||||||
|
log.warning("post-record sidecar %s has no artist; skipping", sidecar)
|
||||||
|
return False
|
||||||
|
|
||||||
|
if source is not None:
|
||||||
|
src = source
|
||||||
|
else:
|
||||||
|
platform = sd.platform or "unknown"
|
||||||
|
src = self._lookup_source_for_sidecar(
|
||||||
|
artist_id=artist.id, platform=platform,
|
||||||
|
)
|
||||||
|
|
||||||
|
epid = sd.external_post_id or sidecar.stem
|
||||||
|
post = self._find_or_create_post(
|
||||||
|
source_id=src.id if src else None,
|
||||||
|
external_post_id=epid,
|
||||||
|
artist_id=artist.id,
|
||||||
|
)
|
||||||
|
if post.artist_id is None:
|
||||||
|
post.artist_id = artist.id
|
||||||
|
if sd.post_url is not None:
|
||||||
|
post.post_url = sd.post_url
|
||||||
|
if sd.post_title is not None:
|
||||||
|
post.post_title = sd.post_title
|
||||||
|
if sd.post_date is not None:
|
||||||
|
post.post_date = sd.post_date
|
||||||
|
if sd.description is not None:
|
||||||
|
post.description = sd.description
|
||||||
|
if sd.attachment_count is not None:
|
||||||
|
post.attachment_count = sd.attachment_count
|
||||||
|
post.raw_metadata = sd.raw
|
||||||
|
self._sync_external_links(post)
|
||||||
|
self.session.commit()
|
||||||
|
return True
|
||||||
|
|
||||||
def attach_in_place(
|
def attach_in_place(
|
||||||
self,
|
self,
|
||||||
path: Path,
|
path: Path,
|
||||||
@@ -664,16 +816,36 @@ class Importer:
|
|||||||
them through. The sidecar JSON gallery-dl emits next to each
|
them through. The sidecar JSON gallery-dl emits next to each
|
||||||
downloaded file is read by `_apply_sidecar` via `find_sidecar`.
|
downloaded file is read by `_apply_sidecar` via `find_sidecar`.
|
||||||
|
|
||||||
|
File-type dispatch parity with `import_one` (FC-2d-iii): zips,
|
||||||
|
PDFs, audio etc. become PostAttachments; archives are extracted.
|
||||||
|
Without this dispatch, gallery-dl-downloaded non-media bounced
|
||||||
|
back as `skipped+invalid_image`, which DownloadService counted
|
||||||
|
as an ingest error and flipped otherwise-successful runs to
|
||||||
|
status="error". Operator-flagged 2026-06-02 after a Lustria
|
||||||
|
patreon run with a 94MB OST zip went red despite 21 successful
|
||||||
|
image attaches.
|
||||||
|
|
||||||
Caller's responsibilities after this returns:
|
Caller's responsibilities after this returns:
|
||||||
- duplicate_hash / duplicate_phash skip → delete the on-disk file
|
- duplicate_hash / duplicate_phash skip → delete the on-disk file
|
||||||
- superseded → file stays where it is (now canonical)
|
- superseded → file stays where it is (now canonical)
|
||||||
- imported → file stays where it is
|
- imported → file stays where it is
|
||||||
|
- attached → the file's been copied into the attachments store;
|
||||||
|
caller may delete the on-disk original (mirrors duplicate_hash)
|
||||||
- failed → file untouched; caller decides
|
- failed → file untouched; caller decides
|
||||||
"""
|
"""
|
||||||
if not is_supported(path):
|
if path.suffix.lower() == ".json":
|
||||||
return ImportResult(
|
return ImportResult(
|
||||||
status="skipped", skip_reason=SkipReason.invalid_image,
|
status="skipped", skip_reason=SkipReason.invalid_image,
|
||||||
error=f"unsupported extension {path.suffix}",
|
error="sidecar json is metadata, not content",
|
||||||
|
)
|
||||||
|
if is_archive(path):
|
||||||
|
return self._import_archive(
|
||||||
|
path, artist=artist, source_row=source,
|
||||||
|
)
|
||||||
|
if not is_supported(path):
|
||||||
|
post = self._post_for_sidecar(path, artist) if artist else None
|
||||||
|
return self._capture_attachment(
|
||||||
|
path, post=post, artist=artist, resolved=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Format / dimension / transparency filters (mirror _import_media).
|
# Format / dimension / transparency filters (mirror _import_media).
|
||||||
@@ -716,6 +888,15 @@ class Importer:
|
|||||||
select(ImageRecord).where(ImageRecord.sha256 == sha)
|
select(ImageRecord).where(ImageRecord.sha256 == sha)
|
||||||
).scalar_one_or_none()
|
).scalar_one_or_none()
|
||||||
if existing:
|
if existing:
|
||||||
|
# Enrich-on-duplicate (image_provenance docstring / spec §3): the
|
||||||
|
# same bytes appearing in another post must show on THAT post too,
|
||||||
|
# so append a provenance row for the new post rather than dropping
|
||||||
|
# the linkage. primary_post_id stays on the first post — _apply_sidecar
|
||||||
|
# only sets it when NULL. Without this the new post is left with no
|
||||||
|
# image AND (if its other content also deduped) no attachment, i.e. a
|
||||||
|
# bare shell — operator-flagged 2026-06-08 (1589 empty Anduo posts).
|
||||||
|
self._apply_sidecar(existing, path, artist, explicit_source=source)
|
||||||
|
self.session.commit()
|
||||||
return ImportResult(
|
return ImportResult(
|
||||||
status="skipped", skip_reason=SkipReason.duplicate_hash,
|
status="skipped", skip_reason=SkipReason.duplicate_hash,
|
||||||
image_id=existing.id, error="sha256 already present",
|
image_id=existing.id, error="sha256 already present",
|
||||||
@@ -736,6 +917,15 @@ class Importer:
|
|||||||
candidates, self.settings.phash_threshold,
|
candidates, self.settings.phash_threshold,
|
||||||
)
|
)
|
||||||
if rel == "larger_exists":
|
if rel == "larger_exists":
|
||||||
|
# Enrich-on-duplicate: link the near-dup's post to the
|
||||||
|
# existing larger image so it shows on both (see duplicate_hash
|
||||||
|
# above). smaller_exists already links via _supersede.
|
||||||
|
larger = self.session.get(ImageRecord, match_id)
|
||||||
|
if larger is not None:
|
||||||
|
self._apply_sidecar(
|
||||||
|
larger, path, artist, explicit_source=source,
|
||||||
|
)
|
||||||
|
self.session.commit()
|
||||||
return ImportResult(
|
return ImportResult(
|
||||||
status="skipped",
|
status="skipped",
|
||||||
skip_reason=SkipReason.duplicate_phash,
|
skip_reason=SkipReason.duplicate_phash,
|
||||||
@@ -745,7 +935,8 @@ class Importer:
|
|||||||
if rel == "smaller_exists":
|
if rel == "smaller_exists":
|
||||||
target = self.session.get(ImageRecord, match_id)
|
target = self.session.get(ImageRecord, match_id)
|
||||||
self._supersede(
|
self._supersede(
|
||||||
target, path, sha, phash, width, height, new_path=path
|
target, path, sha, phash, width, height,
|
||||||
|
new_path=path, artist=artist, source_row=source,
|
||||||
)
|
)
|
||||||
return ImportResult(status="superseded", image_id=match_id)
|
return ImportResult(status="superseded", image_id=match_id)
|
||||||
|
|
||||||
@@ -856,18 +1047,27 @@ class Importer:
|
|||||||
if record.artist_id is None:
|
if record.artist_id is None:
|
||||||
record.artist_id = artist.id
|
record.artist_id = artist.id
|
||||||
|
|
||||||
|
# #830 Phase 2: persist the file's CDN source identity (NULL-only, so a
|
||||||
|
# re-import / enrich-on-duplicate never clobbers it) — the filehash is
|
||||||
|
# the join key that lets post_feed_service remap the body's inline
|
||||||
|
# `<img src=CDN>` to this local copy at render time.
|
||||||
|
if sd.source_url and record.source_filehash is None:
|
||||||
|
record.source_url = sd.source_url
|
||||||
|
record.source_filehash = filehash_from_url(sd.source_url)
|
||||||
|
|
||||||
if explicit_source is not None:
|
if explicit_source is not None:
|
||||||
src = explicit_source
|
src = explicit_source
|
||||||
else:
|
else:
|
||||||
platform = sd.platform or "unknown"
|
platform = sd.platform or "unknown"
|
||||||
src = self._source_for_sidecar(
|
src = self._lookup_source_for_sidecar(
|
||||||
artist_id=artist.id, platform=platform,
|
artist_id=artist.id, platform=platform,
|
||||||
artist_slug=artist.slug,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
epid = sd.external_post_id or sc.stem
|
epid = sd.external_post_id or sc.stem
|
||||||
post = self._find_or_create_post(
|
post = self._find_or_create_post(
|
||||||
source_id=src.id, external_post_id=epid,
|
source_id=src.id if src else None,
|
||||||
|
external_post_id=epid,
|
||||||
|
artist_id=artist.id,
|
||||||
)
|
)
|
||||||
if sd.post_url is not None:
|
if sd.post_url is not None:
|
||||||
post.post_url = sd.post_url
|
post.post_url = sd.post_url
|
||||||
@@ -880,6 +1080,7 @@ class Importer:
|
|||||||
if sd.attachment_count is not None:
|
if sd.attachment_count is not None:
|
||||||
post.attachment_count = sd.attachment_count
|
post.attachment_count = sd.attachment_count
|
||||||
post.raw_metadata = sd.raw
|
post.raw_metadata = sd.raw
|
||||||
|
self._sync_external_links(post)
|
||||||
|
|
||||||
# Race-safe (image_record_id, post_id) upsert — mirrors the
|
# Race-safe (image_record_id, post_id) upsert — mirrors the
|
||||||
# _find_or_create_source/post savepoint pattern. The plain
|
# _find_or_create_source/post savepoint pattern. The plain
|
||||||
@@ -903,7 +1104,7 @@ class Importer:
|
|||||||
ImageProvenance(
|
ImageProvenance(
|
||||||
image_record_id=record.id,
|
image_record_id=record.id,
|
||||||
post_id=post.id,
|
post_id=post.id,
|
||||||
source_id=src.id,
|
source_id=src.id if src else None,
|
||||||
captured_metadata=sd.raw,
|
captured_metadata=sd.raw,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
@@ -913,6 +1114,14 @@ class Importer:
|
|||||||
sp.rollback()
|
sp.rollback()
|
||||||
if record.primary_post_id is None:
|
if record.primary_post_id is None:
|
||||||
record.primary_post_id = post.id
|
record.primary_post_id = post.id
|
||||||
|
# Keep the denormalized gallery sort key (alembic 0035) aligned with
|
||||||
|
# the primary post's publish date so /scroll orders off
|
||||||
|
# ix_image_record_effective_date instead of COALESCE-ing across the
|
||||||
|
# post join. Only override when THIS post is the primary AND carries
|
||||||
|
# a date; otherwise the column keeps its created_at-equivalent server
|
||||||
|
# default (matches the old COALESCE(post_date, created_at) fallback).
|
||||||
|
if record.primary_post_id == post.id and post.post_date is not None:
|
||||||
|
record.effective_date = post.post_date
|
||||||
self.session.flush()
|
self.session.flush()
|
||||||
|
|
||||||
def _copy_to_library(
|
def _copy_to_library(
|
||||||
@@ -939,6 +1148,8 @@ class Importer:
|
|||||||
self, existing: ImageRecord, source: Path, sha: str,
|
self, existing: ImageRecord, source: Path, sha: str,
|
||||||
phash: str, width: int | None, height: int | None,
|
phash: str, width: int | None, height: int | None,
|
||||||
*, new_path: Path | None = None,
|
*, new_path: Path | None = None,
|
||||||
|
artist: Artist | None = None,
|
||||||
|
source_row: Source | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Replace `existing`'s file with the larger `source`, keeping the
|
"""Replace `existing`'s file with the larger `source`, keeping the
|
||||||
row id (so tags/series/curation stay attached). ML is cleared so
|
row id (so tags/series/curation stay attached). ML is cleared so
|
||||||
@@ -973,11 +1184,20 @@ class Importer:
|
|||||||
existing.height = height
|
existing.height = height
|
||||||
existing.thumbnail_path = None
|
existing.thumbnail_path = None
|
||||||
existing.integrity_status = "unknown"
|
existing.integrity_status = "unknown"
|
||||||
existing.tagger_predictions = None
|
|
||||||
existing.tagger_model_version = None
|
existing.tagger_model_version = None
|
||||||
existing.siglip_embedding = None
|
existing.siglip_embedding = None
|
||||||
existing.siglip_model_version = None
|
existing.siglip_model_version = None
|
||||||
existing.centroid_scores = None
|
existing.centroid_scores = None
|
||||||
|
# #768: predictions also live in the normalized image_prediction table
|
||||||
|
# now — clear them so a re-imported file re-derives a fresh set.
|
||||||
|
from sqlalchemy import delete as _delete
|
||||||
|
|
||||||
|
from ..models import ImagePrediction as _ImagePrediction
|
||||||
|
self.session.execute(
|
||||||
|
_delete(_ImagePrediction).where(
|
||||||
|
_ImagePrediction.image_record_id == existing.id
|
||||||
|
)
|
||||||
|
)
|
||||||
# created_at intentionally preserved; updated_at auto-bumps.
|
# created_at intentionally preserved; updated_at auto-bumps.
|
||||||
self.session.flush()
|
self.session.flush()
|
||||||
self.session.commit()
|
self.session.commit()
|
||||||
@@ -990,8 +1210,14 @@ class Importer:
|
|||||||
# _apply_sidecar resolves artist from the sidecar itself if the
|
# _apply_sidecar resolves artist from the sidecar itself if the
|
||||||
# existing row has none, and is internally guarded against
|
# existing row has none, and is internally guarded against
|
||||||
# missing-or-malformed sidecars (silent return).
|
# missing-or-malformed sidecars (silent return).
|
||||||
|
# Audit 2026-06-02: thread artist/source_row from the
|
||||||
|
# download-path caller (attach_in_place smaller_exists branch)
|
||||||
|
# so the supersede preserves explicit subscription linkage
|
||||||
|
# instead of re-deriving via path-walk.
|
||||||
try:
|
try:
|
||||||
self._apply_sidecar(existing, source, None)
|
self._apply_sidecar(
|
||||||
|
existing, source, artist, explicit_source=source_row,
|
||||||
|
)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
# Don't unwind the supersede DB swap if sidecar parsing
|
# Don't unwind the supersede DB swap if sidecar parsing
|
||||||
# blows up unexpectedly — the file replacement is the
|
# blows up unexpectedly — the file replacement is the
|
||||||
|
|||||||
@@ -0,0 +1,677 @@
|
|||||||
|
"""Platform-agnostic native-ingest core (plan #706, build on #697/#703/#704/#705).
|
||||||
|
|
||||||
|
The orchestration that drives a native subscription walk — page a feed →
|
||||||
|
extract media → tiered skip (seen-ledger / on-disk / dead-letter) → download →
|
||||||
|
mark-seen / record-failures / checkpoint-cursor → return a gallery-dl-shaped
|
||||||
|
`DownloadResult`, across tick/backfill/recovery modes — is identical for every
|
||||||
|
platform. Only four things are platform-specific, and they're INJECTED at
|
||||||
|
construction by a thin adapter (e.g. `PatreonIngester`):
|
||||||
|
|
||||||
|
- `client` — `.iter_posts(feed_id, cursor)` yielding `(post, included,
|
||||||
|
page_cursor)` + `.extract_media(post, included) -> [media]`.
|
||||||
|
- `downloader`— `.download_post(post, media, artist_slug, is_seen,
|
||||||
|
should_stop) -> [MediaOutcome]` (status in downloaded/
|
||||||
|
skipped_seen/skipped_disk/quarantined/error;
|
||||||
|
`.path`/`.error`/`.post_id`). `should_stop()` is polled
|
||||||
|
between media so the time-box is honoured mid-post.
|
||||||
|
- ledger — `seen_model` + `failed_model` SQLAlchemy models (+ their
|
||||||
|
on-conflict UNIQUE constraint names) and a `ledger_key(media)`.
|
||||||
|
- failure map — the adapter overrides `_failure_result` (platform exception
|
||||||
|
→ DownloadResult.error_type) and supplies `error_base` (the
|
||||||
|
exception type the walk catches) + `platform` (result label).
|
||||||
|
|
||||||
|
Everything DB touches a SHORT-LIVED sync session from the injected sessionmaker —
|
||||||
|
never held across a network fetch ([[db-connection-held-across-subprocess]]).
|
||||||
|
Plain-HTTP homelab: no secure-context Web API.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import time
|
||||||
|
from collections.abc import Callable
|
||||||
|
|
||||||
|
from sqlalchemy import delete, func, select, text
|
||||||
|
from sqlalchemy.dialects.postgresql import insert as pg_insert
|
||||||
|
|
||||||
|
from .gallery_dl import DownloadResult, ErrorType, make_run_stats
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Stop a tick after this many CONTIGUOUS already-have-it media (seen-ledger or
|
||||||
|
# on-disk) — the cheap native equivalent of gallery-dl's `exit:20`, now free of
|
||||||
|
# per-file HEADs. Headroom against paywalled/undownloadable items interleaving.
|
||||||
|
_TICK_SEEN_THRESHOLD = 20
|
||||||
|
|
||||||
|
# plan #705 #7: after this many failed download/validate attempts a media is
|
||||||
|
# "dead-lettered" and skipped on routine tick/backfill walks (recovery still
|
||||||
|
# re-attempts it). Stops a permanently-broken media re-erroring forever.
|
||||||
|
DEAD_LETTER_THRESHOLD = 3
|
||||||
|
# last_error is Text but bound it so a giant traceback doesn't bloat the row.
|
||||||
|
_ERROR_MAX = 1000
|
||||||
|
|
||||||
|
# plan #709: throttle the live-progress write to the running DownloadEvent to one
|
||||||
|
# every ~5s — a steady cadence for the Downloads view regardless of how big/slow a
|
||||||
|
# page is (page boundaries can be minutes apart on image-dense backfills, so a
|
||||||
|
# page-tied update would lurch). Trivial churn (~one single-row UPDATE / 5s).
|
||||||
|
_LIVE_PROGRESS_INTERVAL = 5.0
|
||||||
|
|
||||||
|
|
||||||
|
class Ingester:
|
||||||
|
"""Generic native-ingest orchestration. Subclass with a platform adapter
|
||||||
|
(see the module docstring) — or construct directly with the keyword seams."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
client,
|
||||||
|
downloader,
|
||||||
|
session_factory: Callable[[], object],
|
||||||
|
seen_model,
|
||||||
|
failed_model,
|
||||||
|
seen_constraint: str,
|
||||||
|
failed_constraint: str,
|
||||||
|
ledger_key: Callable[[object], str],
|
||||||
|
platform: str,
|
||||||
|
error_base: type[Exception],
|
||||||
|
):
|
||||||
|
self.client = client
|
||||||
|
self.downloader = downloader
|
||||||
|
self.session_factory = session_factory
|
||||||
|
self._seen_model = seen_model
|
||||||
|
self._failed_model = failed_model
|
||||||
|
self._seen_constraint = seen_constraint
|
||||||
|
self._failed_constraint = failed_constraint
|
||||||
|
self._ledger_key = ledger_key
|
||||||
|
self._platform = platform
|
||||||
|
self._error_base = error_base
|
||||||
|
|
||||||
|
# -- public ------------------------------------------------------------
|
||||||
|
|
||||||
|
def run(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
source_id: int,
|
||||||
|
campaign_id: str,
|
||||||
|
artist_slug: str,
|
||||||
|
url: str,
|
||||||
|
mode: str,
|
||||||
|
resume_cursor: str | None = None,
|
||||||
|
time_budget_seconds: float = 870.0,
|
||||||
|
seen_threshold: int = _TICK_SEEN_THRESHOLD,
|
||||||
|
posts_base: int = 0,
|
||||||
|
event_id: int | None = None,
|
||||||
|
) -> DownloadResult:
|
||||||
|
"""Walk + download for one source, returning a gallery-dl-shaped result.
|
||||||
|
|
||||||
|
`mode` is "tick" | "backfill" | "recovery". Recovery bypasses the tier-1
|
||||||
|
seen-ledger AND the dead-letter ledger (tier-2 disk still skips kept
|
||||||
|
files). The walk stops on:
|
||||||
|
- budget exhaustion (time_budget_seconds) → TIMEOUT / PARTIAL
|
||||||
|
- tick early-out (seen_threshold contiguous seen) → success
|
||||||
|
- reaching the bottom of the feed → success (rc 0)
|
||||||
|
A client-level failure (drift / auth / network) fails the whole run loud.
|
||||||
|
"""
|
||||||
|
bypass_seen = mode == "recovery"
|
||||||
|
# Only deep walks checkpoint their cursor mid-flight (plan #705 #6); a
|
||||||
|
# tick has no resumable backfill state.
|
||||||
|
checkpoint = mode in ("backfill", "recovery")
|
||||||
|
ledger_key = self._ledger_key
|
||||||
|
# Optional seams (Patreon native ingester): capture pure-text posts that
|
||||||
|
# have NO downloadable media so the artist archive is complete. Absent on
|
||||||
|
# stub clients/downloaders (unit tests) → media-less posts skipped as before.
|
||||||
|
post_record_key = getattr(self.client, "post_record_key", None)
|
||||||
|
write_post_record = getattr(self.downloader, "write_post_record", None)
|
||||||
|
start = time.monotonic()
|
||||||
|
last_live = start # plan #709: last live-progress write timestamp
|
||||||
|
log_lines: list[str] = []
|
||||||
|
written: list[str] = []
|
||||||
|
post_records: list[str] = []
|
||||||
|
quarantined_paths: list[str] = []
|
||||||
|
downloaded = 0
|
||||||
|
errors = 0
|
||||||
|
quarantined = 0
|
||||||
|
dead_lettered = 0
|
||||||
|
skipped_count = 0
|
||||||
|
posts_processed = 0
|
||||||
|
# Net-new posts THIS chunk for the live progress badge (plan #704 #5);
|
||||||
|
# excludes the re-walked resume page so _backfill_posts stays a monotonic
|
||||||
|
# absolute across chunks instead of an inflating sum. posts_processed
|
||||||
|
# stays the gross per-chunk count used for the run summary.
|
||||||
|
chunk_new_posts = 0
|
||||||
|
consecutive_seen = 0
|
||||||
|
emitted_cursor: str | None = None
|
||||||
|
reached_bottom = False
|
||||||
|
budget_hit = False
|
||||||
|
early_out = False
|
||||||
|
stopped = False # plan #708 B4: operator hit Stop mid-walk
|
||||||
|
cancel_armed = False # latched once we observe a live "running" state
|
||||||
|
|
||||||
|
def _result(
|
||||||
|
*, success: bool, return_code: int,
|
||||||
|
error_type: ErrorType | None, error_message: str | None,
|
||||||
|
) -> DownloadResult:
|
||||||
|
# plan #704: return STRUCTURED data — phase 3 reads run_stats/cursor
|
||||||
|
# directly instead of regex-scraping a reconstructed stdout. stdout
|
||||||
|
# stays a human-readable summary (no fake `Cursor:` lines).
|
||||||
|
return DownloadResult(
|
||||||
|
success=success,
|
||||||
|
url=url,
|
||||||
|
artist_slug=artist_slug,
|
||||||
|
platform=self._platform,
|
||||||
|
files_downloaded=downloaded,
|
||||||
|
files_quarantined=quarantined,
|
||||||
|
quarantined_paths=list(quarantined_paths),
|
||||||
|
written_paths=written,
|
||||||
|
post_record_paths=list(post_records),
|
||||||
|
stdout="\n".join(log_lines),
|
||||||
|
stderr="",
|
||||||
|
return_code=return_code,
|
||||||
|
error_type=error_type,
|
||||||
|
error_message=error_message,
|
||||||
|
duration_seconds=time.monotonic() - start,
|
||||||
|
cursor=emitted_cursor,
|
||||||
|
posts_processed=posts_processed,
|
||||||
|
run_stats=make_run_stats(
|
||||||
|
exit_code=return_code,
|
||||||
|
downloaded_count=downloaded,
|
||||||
|
skipped_count=skipped_count,
|
||||||
|
per_item_failures=errors,
|
||||||
|
quarantined_count=quarantined,
|
||||||
|
dead_lettered_count=dead_lettered,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
for post, included, page_cursor in self.client.iter_posts(
|
||||||
|
campaign_id, cursor=resume_cursor
|
||||||
|
):
|
||||||
|
# Checkpoint the cursor that FETCHED this page the moment we
|
||||||
|
# START it — so a chunk cut mid-page resumes the page, not the one
|
||||||
|
# after it. Carried as DownloadResult.cursor (plan #704).
|
||||||
|
if page_cursor and page_cursor != emitted_cursor:
|
||||||
|
emitted_cursor = page_cursor
|
||||||
|
# plan #705 #6: persist the cursor at each page boundary so a
|
||||||
|
# worker SIGKILL mid-chunk resumes near the crash, not the
|
||||||
|
# chunk start. (phase 3 still writes the final cursor — same
|
||||||
|
# value; this is the crash-safety net.) plan #704 #5: persist
|
||||||
|
# the live posts count alongside it so the badge climbs DURING
|
||||||
|
# the chunk, not only when it ends.
|
||||||
|
if checkpoint:
|
||||||
|
# plan #708 B4: an operator Stop pops `_backfill_state` —
|
||||||
|
# bail at the page boundary (progress already checkpointed)
|
||||||
|
# before more network work, so the live chunk halts
|
||||||
|
# promptly instead of running to its time-box. LATCH on the
|
||||||
|
# first observed "running" state, so a run invoked WITHOUT a
|
||||||
|
# running state (a unit test, or a stale call) never
|
||||||
|
# spuriously self-cancels. A short SELECT, never held.
|
||||||
|
if self._still_running(source_id):
|
||||||
|
cancel_armed = True
|
||||||
|
elif cancel_armed:
|
||||||
|
stopped = True
|
||||||
|
break
|
||||||
|
self._checkpoint_cursor(source_id, emitted_cursor)
|
||||||
|
self._checkpoint_posts(source_id, posts_base + chunk_new_posts)
|
||||||
|
|
||||||
|
# Time-box check at the post boundary (coarse, like a gallery-dl
|
||||||
|
# chunk). Backfill/recovery resume from emitted_cursor next chunk.
|
||||||
|
if time.monotonic() - start >= time_budget_seconds:
|
||||||
|
budget_hit = True
|
||||||
|
break
|
||||||
|
|
||||||
|
posts_processed += 1
|
||||||
|
# The resume page (its cursor == resume_cursor) was already
|
||||||
|
# counted by the chunk that checkpointed it — don't re-count it
|
||||||
|
# into the persisted badge (plan #704 #5). First chunk has
|
||||||
|
# resume_cursor None, so everything counts.
|
||||||
|
if not (resume_cursor and page_cursor == resume_cursor):
|
||||||
|
chunk_new_posts += 1
|
||||||
|
# Capture the post body + external links ONCE per post (gated by
|
||||||
|
# the synthetic post key in the seen-ledger), for EVERY post —
|
||||||
|
# whether or not it has downloadable media. This is what makes a
|
||||||
|
# backfill/recovery re-walk RECAPTURE bodies + links for posts
|
||||||
|
# whose media is already on disk: re-downloading existing media
|
||||||
|
# never fills links the system never had, so the body recapture
|
||||||
|
# has to ride the walk itself. Detail-fetch (for an empty feed
|
||||||
|
# body) happens at most once per post — the gate then spares it on
|
||||||
|
# later walks. bypass_seen (recovery) re-captures unconditionally.
|
||||||
|
if post_record_key and write_post_record:
|
||||||
|
rk = post_record_key(post)
|
||||||
|
if rk is not None:
|
||||||
|
pkey, ppid = rk
|
||||||
|
already = (
|
||||||
|
set() if bypass_seen
|
||||||
|
else self._seen_keys(source_id, [pkey])
|
||||||
|
)
|
||||||
|
if pkey not in already:
|
||||||
|
rec_path = write_post_record(post, artist_slug)
|
||||||
|
if rec_path is not None:
|
||||||
|
post_records.append(str(rec_path))
|
||||||
|
self._mark_seen(source_id, [(pkey, ppid)])
|
||||||
|
|
||||||
|
media = self.client.extract_media(post, included)
|
||||||
|
if not media:
|
||||||
|
continue
|
||||||
|
|
||||||
|
keys = [ledger_key(m) for m in media]
|
||||||
|
# Recovery bypasses BOTH the seen-ledger AND the dead-letter
|
||||||
|
# ledger (the operator's "try everything again"); routine walks
|
||||||
|
# skip seen + dead media (tier-1 + tier-1.5, plan #705 #7).
|
||||||
|
dead = set() if bypass_seen else self._dead_keys(source_id, keys)
|
||||||
|
seen = (
|
||||||
|
set()
|
||||||
|
if bypass_seen
|
||||||
|
else self._seen_keys(source_id, keys)
|
||||||
|
)
|
||||||
|
skip = seen | dead
|
||||||
|
|
||||||
|
def _is_skip(m, _skip=skip) -> bool:
|
||||||
|
return ledger_key(m) in _skip
|
||||||
|
|
||||||
|
# Honour the time-box DURING a media-dense post too, not only at
|
||||||
|
# the per-post boundary below — else one heavy post can blow the
|
||||||
|
# chunk budget out to the Celery soft limit (Pocketacer, 2026-06-07).
|
||||||
|
outcomes = self.downloader.download_post(
|
||||||
|
post, media, artist_slug, is_seen=_is_skip,
|
||||||
|
should_stop=lambda: time.monotonic() - start >= time_budget_seconds,
|
||||||
|
)
|
||||||
|
|
||||||
|
to_mark: list[tuple[str, str]] = []
|
||||||
|
to_clear: list[str] = [] # recovered → drop any dead-letter row
|
||||||
|
to_fail: list[tuple[str, str, str]] = [] # (key, post_id, error)
|
||||||
|
for media_item, outcome in zip(media, outcomes, strict=False):
|
||||||
|
key = ledger_key(media_item)
|
||||||
|
if key in dead:
|
||||||
|
dead_lettered += 1 # skipped because previously dead
|
||||||
|
if outcome.status == "downloaded":
|
||||||
|
downloaded += 1
|
||||||
|
if outcome.path is not None:
|
||||||
|
written.append(str(outcome.path))
|
||||||
|
to_mark.append((key, media_item.post_id))
|
||||||
|
to_clear.append(key)
|
||||||
|
consecutive_seen = 0
|
||||||
|
elif outcome.status == "skipped_disk":
|
||||||
|
# Already on disk (a prior run). Reconcile the ledger so a
|
||||||
|
# later tick skips it at tier-1 without a disk stat, but
|
||||||
|
# do NOT re-feed it to phase 3 — attach_in_place would see
|
||||||
|
# the duplicate sha256 and unlink the on-disk copy.
|
||||||
|
to_mark.append((key, media_item.post_id))
|
||||||
|
to_clear.append(key)
|
||||||
|
skipped_count += 1
|
||||||
|
consecutive_seen += 1
|
||||||
|
elif outcome.status == "skipped_seen":
|
||||||
|
skipped_count += 1
|
||||||
|
consecutive_seen += 1
|
||||||
|
elif outcome.status == "quarantined":
|
||||||
|
# New content that failed validation (corrupt) — counted
|
||||||
|
# distinctly so the run surfaces a real quarantined total.
|
||||||
|
# Not marked seen (a later walk may re-fetch a fixed file);
|
||||||
|
# it IS new content, so it breaks the run-of-seen. Counts
|
||||||
|
# toward the dead-letter ledger (plan #705 #7).
|
||||||
|
quarantined += 1
|
||||||
|
if outcome.path is not None:
|
||||||
|
quarantined_paths.append(str(outcome.path))
|
||||||
|
to_fail.append((key, media_item.post_id, outcome.error or "quarantined"))
|
||||||
|
consecutive_seen = 0
|
||||||
|
elif outcome.status == "error":
|
||||||
|
errors += 1
|
||||||
|
to_fail.append((key, media_item.post_id, outcome.error or "error"))
|
||||||
|
# An error neither advances nor resets the run-of-seen.
|
||||||
|
|
||||||
|
if mode == "tick" and consecutive_seen >= seen_threshold:
|
||||||
|
early_out = True
|
||||||
|
break
|
||||||
|
|
||||||
|
# Persist ledger changes AFTER the network fetch, on short
|
||||||
|
# sessions: mark downloaded/on-disk seen, clear any dead-letter
|
||||||
|
# for recovered media, and record failures (plan #705 #7).
|
||||||
|
if to_mark:
|
||||||
|
self._mark_seen(source_id, to_mark)
|
||||||
|
if to_clear:
|
||||||
|
self._clear_failures(source_id, to_clear)
|
||||||
|
if to_fail:
|
||||||
|
self._record_failures(source_id, to_fail)
|
||||||
|
|
||||||
|
# plan #709: time-throttled live progress to the running event so
|
||||||
|
# the Downloads view ticks ~every 5s, independent of page size.
|
||||||
|
now = time.monotonic()
|
||||||
|
if event_id is not None and (now - last_live) >= _LIVE_PROGRESS_INTERVAL:
|
||||||
|
last_live = now
|
||||||
|
self._write_live_progress(event_id, {
|
||||||
|
"downloaded": downloaded,
|
||||||
|
"skipped": skipped_count,
|
||||||
|
"errors": errors,
|
||||||
|
"quarantined": quarantined,
|
||||||
|
"posts": posts_processed,
|
||||||
|
})
|
||||||
|
|
||||||
|
if early_out:
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
reached_bottom = True
|
||||||
|
except self._error_base as exc:
|
||||||
|
# The platform's client-error base — _failure_result (adapter)
|
||||||
|
# maps it to a typed error.
|
||||||
|
return self._failure_result(exc, _result)
|
||||||
|
|
||||||
|
# plan #708 B4: a Stop already popped the backfill state (incl. cursor +
|
||||||
|
# posts), so don't re-write them — return PARTIAL (reads as "ok/progress",
|
||||||
|
# the lifecycle no-ops since state is gone) instead of a false "complete".
|
||||||
|
if stopped:
|
||||||
|
return _result(
|
||||||
|
success=False, return_code=-1,
|
||||||
|
error_type=ErrorType.PARTIAL,
|
||||||
|
error_message=f"Stopped by operator: {downloaded} file(s) this chunk",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Final authoritative posts count for the badge — captures the last page
|
||||||
|
# after the last boundary write and the time-box break (plan #704 #5).
|
||||||
|
if checkpoint:
|
||||||
|
self._checkpoint_posts(source_id, posts_base + chunk_new_posts)
|
||||||
|
|
||||||
|
if errors:
|
||||||
|
log_lines.append(f"{errors} media item(s) failed")
|
||||||
|
if quarantined:
|
||||||
|
log_lines.append(f"{quarantined} media item(s) quarantined (invalid)")
|
||||||
|
if dead_lettered:
|
||||||
|
log_lines.append(f"{dead_lettered} media item(s) skipped (dead-lettered)")
|
||||||
|
log_lines.append(
|
||||||
|
f"{self._platform} ingest ({mode}): {downloaded} downloaded, "
|
||||||
|
f"{skipped_count} skipped, {quarantined} quarantined, "
|
||||||
|
f"{dead_lettered} dead-lettered, {errors} error(s), "
|
||||||
|
f"{posts_processed} post(s)"
|
||||||
|
+ (", reached end" if reached_bottom else "")
|
||||||
|
+ (", time-boxed" if budget_hit else "")
|
||||||
|
)
|
||||||
|
|
||||||
|
if budget_hit:
|
||||||
|
# A chunk that hit its time-box but made forward progress is a
|
||||||
|
# NORMAL chunk boundary, not a failure (PARTIAL → status "ok"); the
|
||||||
|
# next chunk resumes from the emitted cursor. No progress → TIMEOUT,
|
||||||
|
# which feeds download_service's backfill stall-guard. rc<0 mirrors
|
||||||
|
# subprocess TimeoutExpired so completion detection stays false.
|
||||||
|
made_progress = downloaded > 0 or emitted_cursor != resume_cursor
|
||||||
|
if made_progress:
|
||||||
|
return _result(
|
||||||
|
success=False, return_code=-1,
|
||||||
|
error_type=ErrorType.PARTIAL,
|
||||||
|
error_message=(
|
||||||
|
f"Backfill chunk: {downloaded} file(s) — continuing"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
return _result(
|
||||||
|
success=False, return_code=-1,
|
||||||
|
error_type=ErrorType.TIMEOUT,
|
||||||
|
error_message="Chunk timed out with no progress",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Normal success: reached the bottom, or a tick that early-outed. rc 0 +
|
||||||
|
# error_type None is REQUIRED for a backfill/recovery walk that reached
|
||||||
|
# the bottom to be marked COMPLETE by
|
||||||
|
# download_service._apply_backfill_lifecycle — so we return None even
|
||||||
|
# when downloaded == 0 (a re-confirming walk that found nothing new still
|
||||||
|
# completed). success=True maps to status "ok" regardless. A tick that
|
||||||
|
# early-outed also returns here; ticks never set backfill state so the
|
||||||
|
# lifecycle is a no-op for them.
|
||||||
|
return _result(
|
||||||
|
success=True, return_code=0,
|
||||||
|
error_type=None, error_message=None,
|
||||||
|
)
|
||||||
|
|
||||||
|
# -- preview (dry-run) -------------------------------------------------
|
||||||
|
|
||||||
|
def preview(
|
||||||
|
self,
|
||||||
|
source_id: int,
|
||||||
|
campaign_id: str,
|
||||||
|
*,
|
||||||
|
page_limit: int = 3,
|
||||||
|
sample_size: int = 10,
|
||||||
|
) -> dict:
|
||||||
|
"""Dry-run (plan #708 B4): walk up to `page_limit` pages and count media
|
||||||
|
NOT already in the seen/dead ledgers, WITHOUT downloading anything.
|
||||||
|
|
||||||
|
Read-only — only the seen/dead SELECTs touch the DB (short sessions). Lets
|
||||||
|
an operator gauge "is this source worth a backfill?" cheaply. Returns:
|
||||||
|
{total_new, posts_scanned, pages_scanned, has_more,
|
||||||
|
sample: [{title, date, new}, ...]} # sample = posts with new media
|
||||||
|
A client-level failure (auth/drift) propagates to the caller.
|
||||||
|
"""
|
||||||
|
total_new = 0
|
||||||
|
posts_scanned = 0
|
||||||
|
pages_scanned = 0
|
||||||
|
has_more = False
|
||||||
|
sample: list[dict] = []
|
||||||
|
unset = object()
|
||||||
|
last_page: object = unset
|
||||||
|
for post, included, page_cursor in self.client.iter_posts(
|
||||||
|
campaign_id, cursor=None
|
||||||
|
):
|
||||||
|
if page_cursor != last_page:
|
||||||
|
last_page = page_cursor
|
||||||
|
pages_scanned += 1
|
||||||
|
if pages_scanned > page_limit:
|
||||||
|
has_more = True
|
||||||
|
pages_scanned = page_limit
|
||||||
|
break
|
||||||
|
posts_scanned += 1
|
||||||
|
media = self.client.extract_media(post, included)
|
||||||
|
if not media:
|
||||||
|
continue
|
||||||
|
keys = [self._ledger_key(m) for m in media]
|
||||||
|
skip = self._seen_keys(source_id, keys) | self._dead_keys(source_id, keys)
|
||||||
|
new_count = sum(1 for m in media if self._ledger_key(m) not in skip)
|
||||||
|
total_new += new_count
|
||||||
|
if new_count > 0 and len(sample) < sample_size:
|
||||||
|
meta = self.client.post_meta(post)
|
||||||
|
sample.append(
|
||||||
|
{
|
||||||
|
"title": meta.get("title") or "(untitled)",
|
||||||
|
"date": meta.get("date"),
|
||||||
|
"new": new_count,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"total_new": total_new,
|
||||||
|
"posts_scanned": posts_scanned,
|
||||||
|
"pages_scanned": pages_scanned,
|
||||||
|
"has_more": has_more,
|
||||||
|
"sample": sample,
|
||||||
|
}
|
||||||
|
|
||||||
|
# -- failure mapping (adapter overrides) -------------------------------
|
||||||
|
|
||||||
|
def _failure_result(self, exc: Exception, _result) -> DownloadResult:
|
||||||
|
"""Map a platform client-error to a typed failed DownloadResult. The base
|
||||||
|
gives a safe default; adapters override with their exception taxonomy."""
|
||||||
|
log.warning("%s ingest failed: %s", self._platform, exc)
|
||||||
|
return _result(
|
||||||
|
success=False, return_code=1,
|
||||||
|
error_type=ErrorType.UNKNOWN_ERROR, error_message=str(exc),
|
||||||
|
)
|
||||||
|
|
||||||
|
# -- seen-ledger (short-lived sessions) --------------------------------
|
||||||
|
|
||||||
|
def _seen_keys(self, source_id: int, keys: list[str]) -> set[str]:
|
||||||
|
"""Which of `keys` are already in the seen-ledger for this source.
|
||||||
|
|
||||||
|
One short SELECT on its own session — opened and closed without any
|
||||||
|
network in between (the GETs happen after, in download_post).
|
||||||
|
"""
|
||||||
|
if not keys:
|
||||||
|
return set()
|
||||||
|
with self.session_factory() as session:
|
||||||
|
rows = session.execute(
|
||||||
|
select(self._seen_model.filehash).where(
|
||||||
|
self._seen_model.source_id == source_id,
|
||||||
|
self._seen_model.filehash.in_(keys),
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
return set(rows)
|
||||||
|
|
||||||
|
def _checkpoint_cursor(self, source_id: int, cursor: str) -> None:
|
||||||
|
"""Persist the in-progress backfill cursor mid-walk (plan #705 #6).
|
||||||
|
|
||||||
|
ATOMIC, single-key UPDATE: cast the JSON column to jsonb, set just
|
||||||
|
`_backfill_cursor`, cast back — so it never clobbers operator config or
|
||||||
|
the other backfill keys (no read-modify-write race). The in-flight guard
|
||||||
|
means only this source's one download runs at a time; a concurrent
|
||||||
|
operator stop is benign (a stray cursor with no `_backfill_state` is
|
||||||
|
ignored by tick mode and cleared on the next start).
|
||||||
|
"""
|
||||||
|
with self.session_factory() as session:
|
||||||
|
session.execute(
|
||||||
|
text(
|
||||||
|
"UPDATE source SET config_overrides = jsonb_set("
|
||||||
|
" coalesce(config_overrides::jsonb, '{}'::jsonb),"
|
||||||
|
" '{_backfill_cursor}', to_jsonb(cast(:cur AS text))"
|
||||||
|
")::json WHERE id = :sid"
|
||||||
|
),
|
||||||
|
{"cur": cursor, "sid": source_id},
|
||||||
|
)
|
||||||
|
session.commit()
|
||||||
|
|
||||||
|
def _write_live_progress(self, event_id: int, counts: dict) -> None:
|
||||||
|
"""Throttled mid-walk write of live counts to the RUNNING download_event
|
||||||
|
(plan #709) so the Downloads view shows progress before the chunk
|
||||||
|
finishes. A short session (never held across the walk); the `status =
|
||||||
|
'running'` guard avoids clobbering an event phase 3 already finalized.
|
||||||
|
`metadata` is JSONB — jsonb_set sets just the `live` key, leaving the rest
|
||||||
|
for phase 3 to overwrite with the final run_stats."""
|
||||||
|
with self.session_factory() as session:
|
||||||
|
session.execute(
|
||||||
|
text(
|
||||||
|
"UPDATE download_event SET metadata = jsonb_set("
|
||||||
|
" coalesce(metadata, '{}'::jsonb), '{live}',"
|
||||||
|
" cast(:live AS jsonb)) "
|
||||||
|
"WHERE id = :eid AND status = 'running'"
|
||||||
|
),
|
||||||
|
{"live": json.dumps(counts), "eid": event_id},
|
||||||
|
)
|
||||||
|
session.commit()
|
||||||
|
|
||||||
|
def _still_running(self, source_id: int) -> bool:
|
||||||
|
"""True while the source is armed for a deep walk (plan #708 B4).
|
||||||
|
|
||||||
|
An operator Stop (`source_service.stop_backfill`) pops `_backfill_state`,
|
||||||
|
so a False here means "cancel this chunk now". One short SELECT on its own
|
||||||
|
session — never held across the walk
|
||||||
|
([[db-connection-held-across-subprocess]])."""
|
||||||
|
with self.session_factory() as session:
|
||||||
|
state = session.execute(
|
||||||
|
text(
|
||||||
|
"SELECT config_overrides::jsonb ->> '_backfill_state' "
|
||||||
|
"FROM source WHERE id = :sid"
|
||||||
|
),
|
||||||
|
{"sid": source_id},
|
||||||
|
).scalar_one_or_none()
|
||||||
|
return state == "running"
|
||||||
|
|
||||||
|
def _checkpoint_posts(self, source_id: int, posts: int) -> None:
|
||||||
|
"""Persist the live backfill posts-processed count mid-walk (plan #704 #5).
|
||||||
|
|
||||||
|
Same atomic single-key jsonb_set dance as _checkpoint_cursor, on the
|
||||||
|
`_backfill_posts` key (cast to a JSON number) — so the progress badge
|
||||||
|
climbs DURING a chunk without clobbering operator config or the cursor.
|
||||||
|
The ingester OWNS this key now; download_service no longer accumulates it
|
||||||
|
post-chunk (which lagged a whole chunk and over-counted the resume page).
|
||||||
|
"""
|
||||||
|
with self.session_factory() as session:
|
||||||
|
session.execute(
|
||||||
|
text(
|
||||||
|
"UPDATE source SET config_overrides = jsonb_set("
|
||||||
|
" coalesce(config_overrides::jsonb, '{}'::jsonb),"
|
||||||
|
" '{_backfill_posts}', to_jsonb(cast(:posts AS int))"
|
||||||
|
")::json WHERE id = :sid"
|
||||||
|
),
|
||||||
|
{"posts": posts, "sid": source_id},
|
||||||
|
)
|
||||||
|
session.commit()
|
||||||
|
|
||||||
|
def _mark_seen(self, source_id: int, items: list[tuple[str, str]]) -> None:
|
||||||
|
"""Idempotent upsert of (filehash, post_id) seen-ledger rows for a page.
|
||||||
|
|
||||||
|
ON CONFLICT DO NOTHING against the (source_id, filehash) UNIQUE so a
|
||||||
|
re-sighting — or a concurrent walk — is a harmless no-op
|
||||||
|
([[scalar_one_or_none-duplicates]]: never check-then-insert without the
|
||||||
|
DB constraint backing it). De-dup the batch locally first so a single
|
||||||
|
page can't present the same key twice to one INSERT.
|
||||||
|
"""
|
||||||
|
seen_local: set[str] = set()
|
||||||
|
values = []
|
||||||
|
for key, post_id in items:
|
||||||
|
if key in seen_local:
|
||||||
|
continue
|
||||||
|
seen_local.add(key)
|
||||||
|
values.append(
|
||||||
|
{"source_id": source_id, "filehash": key, "post_id": post_id}
|
||||||
|
)
|
||||||
|
if not values:
|
||||||
|
return
|
||||||
|
with self.session_factory() as session:
|
||||||
|
stmt = pg_insert(self._seen_model).values(values)
|
||||||
|
stmt = stmt.on_conflict_do_nothing(constraint=self._seen_constraint)
|
||||||
|
session.execute(stmt)
|
||||||
|
session.commit()
|
||||||
|
|
||||||
|
# -- dead-letter ledger (plan #705 #7) ---------------------------------
|
||||||
|
|
||||||
|
def _dead_keys(self, source_id: int, keys: list[str]) -> set[str]:
|
||||||
|
"""Which of `keys` have failed >= DEAD_LETTER_THRESHOLD times (dead).
|
||||||
|
One short SELECT; recovery never calls this (it re-attempts dead media)."""
|
||||||
|
if not keys:
|
||||||
|
return set()
|
||||||
|
with self.session_factory() as session:
|
||||||
|
rows = session.execute(
|
||||||
|
select(self._failed_model.filehash).where(
|
||||||
|
self._failed_model.source_id == source_id,
|
||||||
|
self._failed_model.filehash.in_(keys),
|
||||||
|
self._failed_model.attempts >= DEAD_LETTER_THRESHOLD,
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
return set(rows)
|
||||||
|
|
||||||
|
def _record_failures(
|
||||||
|
self, source_id: int, items: list[tuple[str, str, str]]
|
||||||
|
) -> None:
|
||||||
|
"""Upsert-increment the dead-letter ledger for failed media. On conflict
|
||||||
|
bump `attempts` and refresh last_error/last_failed_at (UNIQUE backs the
|
||||||
|
upsert — no check-then-insert). De-dup the batch (one row/key, last error
|
||||||
|
wins)."""
|
||||||
|
by_key: dict[str, str] = {}
|
||||||
|
for key, _post_id, err in items:
|
||||||
|
by_key[key] = (err or "")[:_ERROR_MAX]
|
||||||
|
if not by_key:
|
||||||
|
return
|
||||||
|
values = [
|
||||||
|
{"source_id": source_id, "filehash": k, "attempts": 1, "last_error": e}
|
||||||
|
for k, e in by_key.items()
|
||||||
|
]
|
||||||
|
with self.session_factory() as session:
|
||||||
|
stmt = pg_insert(self._failed_model).values(values)
|
||||||
|
stmt = stmt.on_conflict_do_update(
|
||||||
|
constraint=self._failed_constraint,
|
||||||
|
set_={
|
||||||
|
"attempts": self._failed_model.attempts + 1,
|
||||||
|
"last_error": stmt.excluded.last_error,
|
||||||
|
"last_failed_at": func.now(),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
session.execute(stmt)
|
||||||
|
session.commit()
|
||||||
|
|
||||||
|
def _clear_failures(self, source_id: int, keys: list[str]) -> None:
|
||||||
|
"""Drop dead-letter rows for media that just downloaded cleanly — they
|
||||||
|
recovered. A no-op DELETE for keys that were never failing."""
|
||||||
|
unique = list(dict.fromkeys(keys))
|
||||||
|
if not unique:
|
||||||
|
return
|
||||||
|
with self.session_factory() as session:
|
||||||
|
session.execute(
|
||||||
|
delete(self._failed_model).where(
|
||||||
|
self._failed_model.source_id == source_id,
|
||||||
|
self._failed_model.filehash.in_(unique),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
session.commit()
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
"""Extract off-platform file-host links from a post body.
|
||||||
|
|
||||||
|
Pure (no I/O) so it's unit-testable and reusable by EVERY in-house downloader's
|
||||||
|
import path — every platform stores its body in `Post.description`, so running
|
||||||
|
this there covers them all. Finds links to the supported external file hosts in
|
||||||
|
a post's HTML body (both `<a href="...">` anchors and bare URLs in text),
|
||||||
|
unwraps a Patreon outbound-redirect wrapper, and preserves the FULL url
|
||||||
|
including the `#fragment` (mega.nz puts the decryption key there) and the query
|
||||||
|
string — without those a mega download is impossible.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from html import unescape
|
||||||
|
from urllib.parse import parse_qs, urlsplit
|
||||||
|
|
||||||
|
# Supported file-host enum values (kept in sync with the external_link CHECK).
|
||||||
|
SUPPORTED_HOSTS = ("mega", "gdrive", "mediafire", "dropbox", "pixeldrain")
|
||||||
|
|
||||||
|
# Bare-domain suffix → canonical host label. Matched against the URL netloc
|
||||||
|
# (exact or as a dotted suffix, so www./dl. subdomains resolve too).
|
||||||
|
_HOST_MAP = {
|
||||||
|
"mega.nz": "mega",
|
||||||
|
"mega.co.nz": "mega",
|
||||||
|
"drive.google.com": "gdrive",
|
||||||
|
"mediafire.com": "mediafire",
|
||||||
|
"dropbox.com": "dropbox",
|
||||||
|
"pixeldrain.com": "pixeldrain",
|
||||||
|
}
|
||||||
|
|
||||||
|
# `<a href="...">label</a>` — DOTALL so a label spanning tags/newlines is caught.
|
||||||
|
_HREF_RE = re.compile(
|
||||||
|
r"""<a\b[^>]*?\bhref=["']([^"']+)["'][^>]*>(.*?)</a>""",
|
||||||
|
re.IGNORECASE | re.DOTALL,
|
||||||
|
)
|
||||||
|
_TAG_RE = re.compile(r"<[^>]+>")
|
||||||
|
# Bare http(s) URL in text. Stops at whitespace, quotes, angle brackets, and
|
||||||
|
# closing brackets — but KEEPS `#`, `&`, `?`, `=` so fragments/queries survive.
|
||||||
|
_URL_RE = re.compile(r"""https?://[^\s"'<>)\]}]+""", re.IGNORECASE)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ExtractedLink:
|
||||||
|
host: str # one of SUPPORTED_HOSTS
|
||||||
|
url: str # full url incl. query + #fragment
|
||||||
|
label: str | None # visible anchor text, when present
|
||||||
|
|
||||||
|
|
||||||
|
def _netloc(url: str) -> str:
|
||||||
|
# Lowercase host without credentials or port.
|
||||||
|
return urlsplit(url).netloc.lower().split("@")[-1].split(":")[0]
|
||||||
|
|
||||||
|
|
||||||
|
def host_for(url: str) -> str | None:
|
||||||
|
"""Canonical host label for a url, or None if it's not a supported host."""
|
||||||
|
netloc = _netloc(url)
|
||||||
|
for suffix, host in _HOST_MAP.items():
|
||||||
|
if netloc == suffix or netloc.endswith("." + suffix):
|
||||||
|
return host
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _unwrap(url: str) -> str:
|
||||||
|
"""Unwrap a Patreon outbound-redirect wrapper to its inner target, so a
|
||||||
|
wrapped mega/gdrive link resolves to the real host. Patreon has used both
|
||||||
|
`www.patreon.com/...?url=<encoded>` and the `l.patreon.com` shim; check the
|
||||||
|
common target-param names. Non-Patreon urls pass through untouched."""
|
||||||
|
netloc = _netloc(url)
|
||||||
|
if not (netloc == "patreon.com" or netloc.endswith(".patreon.com")):
|
||||||
|
return url
|
||||||
|
qs = parse_qs(urlsplit(url).query)
|
||||||
|
for key in ("url", "u", "ext_url", "redirect", "target"):
|
||||||
|
vals = qs.get(key)
|
||||||
|
if vals and vals[0]:
|
||||||
|
return unescape(vals[0]).strip()
|
||||||
|
return url
|
||||||
|
|
||||||
|
|
||||||
|
def extract_external_links(html: str | None) -> list[ExtractedLink]:
|
||||||
|
"""All supported-host links in `html`, de-duplicated by url (first wins, so
|
||||||
|
an anchor's label is kept over a later bare sighting of the same url)."""
|
||||||
|
if not html:
|
||||||
|
return []
|
||||||
|
found: dict[str, ExtractedLink] = {}
|
||||||
|
|
||||||
|
# 1) Anchors first — they carry a human label ("Mega - Streamable").
|
||||||
|
for raw_href, inner in _HREF_RE.findall(html):
|
||||||
|
url = _unwrap(unescape(raw_href).strip())
|
||||||
|
host = host_for(url)
|
||||||
|
if host is None:
|
||||||
|
continue
|
||||||
|
label = unescape(_TAG_RE.sub("", inner)).strip() or None
|
||||||
|
found.setdefault(url, ExtractedLink(host=host, url=url, label=label))
|
||||||
|
|
||||||
|
# 2) Bare URLs pasted in text (no label). Trailing prose punctuation is
|
||||||
|
# trimmed; the href values already captured above de-dup away here.
|
||||||
|
for raw in _URL_RE.findall(html):
|
||||||
|
url = _unwrap(unescape(raw).strip().rstrip(".,;"))
|
||||||
|
host = host_for(url)
|
||||||
|
if host is None:
|
||||||
|
continue
|
||||||
|
found.setdefault(url, ExtractedLink(host=host, url=url, label=None))
|
||||||
|
|
||||||
|
return list(found.values())
|
||||||
@@ -1,10 +0,0 @@
|
|||||||
"""FC-5 migration tooling.
|
|
||||||
|
|
||||||
One module per concern (gs/ir/overlap/ml_queue/verify/cleanup).
|
|
||||||
Each migrator returns a counts dict; the run_migration task wires
|
|
||||||
that dict into MigrationRun.counts so the UI polling shows progress.
|
|
||||||
|
|
||||||
backup + rollback were retired in FC-3h (2026-05-24); first-class
|
|
||||||
backup lives at backend/app/services/backup_service.py and exposes
|
|
||||||
its own /api/system/backup/* surface.
|
|
||||||
"""
|
|
||||||
@@ -1,182 +0,0 @@
|
|||||||
"""Targeted cleanup migrator: delete every image attributed to one Artist.
|
|
||||||
|
|
||||||
Built for the IR-migration rescue case where the filesystem scan derived
|
|
||||||
a bogus 'imagerepo' artist from a mismatched bind-mount layout. Every
|
|
||||||
image attributed to that artist (40k+ rows) needs to be removed — DB
|
|
||||||
rows, original files under `/images/<bucket>/...`, and thumbnails under
|
|
||||||
`/images/thumbs/...` — before the operator remounts and re-scans.
|
|
||||||
|
|
||||||
CASCADE handles image_tag, image_provenance, series_page, and
|
|
||||||
tag_suggestion_rejection child rows; import_task.result_image_id is
|
|
||||||
SET NULL by FK. We also delete ImportTask rows whose source_path starts
|
|
||||||
with the (still-existing) IR scan prefix so the next scan isn't fooled
|
|
||||||
by them.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from sqlalchemy import delete, func, select
|
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
|
||||||
|
|
||||||
from ...models import Artist, ImageRecord, ImportBatch, ImportTask
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
_BATCH_SIZE = 500
|
|
||||||
|
|
||||||
|
|
||||||
def _zero_counts() -> dict:
|
|
||||||
return {
|
|
||||||
"rows_processed": 0, "rows_inserted": 0, "rows_skipped": 0,
|
|
||||||
"files_copied": 0, "bytes_copied": 0, "conflicts": 0,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def _thumb_path(images_root: Path, sha256_hex: str) -> tuple[Path, Path]:
|
|
||||||
"""Return both possible thumbnail paths (.jpg and .png). We try both
|
|
||||||
because the extension is chosen at generate-time based on the source
|
|
||||||
image's mode (alpha → .png, otherwise → .jpg)."""
|
|
||||||
bucket = sha256_hex[:3]
|
|
||||||
base = images_root / "thumbs" / bucket / sha256_hex
|
|
||||||
return base.with_suffix(".jpg"), base.with_suffix(".png")
|
|
||||||
|
|
||||||
|
|
||||||
def _delete_file(path: Path) -> bool:
|
|
||||||
"""Best-effort unlink; True if the file was actually removed."""
|
|
||||||
try:
|
|
||||||
path.unlink(missing_ok=True)
|
|
||||||
return True
|
|
||||||
except OSError as exc:
|
|
||||||
log.warning("cleanup: failed to unlink %s: %s", path, exc)
|
|
||||||
return False
|
|
||||||
|
|
||||||
|
|
||||||
async def cleanup_artist_async(
|
|
||||||
db: AsyncSession,
|
|
||||||
*,
|
|
||||||
slug: str,
|
|
||||||
images_root: Path | None = None,
|
|
||||||
dry_run: bool = False,
|
|
||||||
source_path_prefix: str | None = None,
|
|
||||||
) -> dict:
|
|
||||||
"""Delete every image attributed to the Artist with this slug,
|
|
||||||
along with the artist row itself and any associated import tasks.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
slug: artist.slug to target (e.g. 'imagerepo').
|
|
||||||
images_root: defaults to /images.
|
|
||||||
dry_run: skip filesystem + DB writes; still walk rows for counts.
|
|
||||||
source_path_prefix: if set, ImportTask rows whose source_path
|
|
||||||
starts with this string are deleted too (use the IR scan
|
|
||||||
mount prefix, e.g. '/import/imagerepo').
|
|
||||||
"""
|
|
||||||
root = images_root if images_root is not None else Path("/images")
|
|
||||||
|
|
||||||
artist = (await db.execute(
|
|
||||||
select(Artist).where(Artist.slug == slug)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if artist is None:
|
|
||||||
raise ValueError(f"no Artist with slug={slug!r}")
|
|
||||||
|
|
||||||
artist_id = artist.id
|
|
||||||
artist_name = artist.name
|
|
||||||
|
|
||||||
total_images = (await db.execute(
|
|
||||||
select(func.count(ImageRecord.id)).where(ImageRecord.artist_id == artist_id)
|
|
||||||
)).scalar_one()
|
|
||||||
|
|
||||||
counts = _zero_counts()
|
|
||||||
files_deleted = 0
|
|
||||||
thumbs_deleted = 0
|
|
||||||
images_deleted = 0
|
|
||||||
|
|
||||||
# Batched delete loop. CASCADE handles image_tag, image_provenance,
|
|
||||||
# series_page, tag_suggestion_rejection. import_task.result_image_id
|
|
||||||
# is SET NULL by FK.
|
|
||||||
while True:
|
|
||||||
rows = (await db.execute(
|
|
||||||
select(ImageRecord.id, ImageRecord.path, ImageRecord.sha256)
|
|
||||||
.where(ImageRecord.artist_id == artist_id)
|
|
||||||
.limit(_BATCH_SIZE)
|
|
||||||
)).all()
|
|
||||||
if not rows:
|
|
||||||
break
|
|
||||||
|
|
||||||
ids = [r.id for r in rows]
|
|
||||||
counts["rows_processed"] += len(ids)
|
|
||||||
|
|
||||||
if not dry_run:
|
|
||||||
for r in rows:
|
|
||||||
if r.path:
|
|
||||||
if _delete_file(Path(r.path)):
|
|
||||||
files_deleted += 1
|
|
||||||
if r.sha256:
|
|
||||||
jpg, png = _thumb_path(root, r.sha256)
|
|
||||||
if _delete_file(jpg):
|
|
||||||
thumbs_deleted += 1
|
|
||||||
if _delete_file(png):
|
|
||||||
thumbs_deleted += 1
|
|
||||||
|
|
||||||
await db.execute(
|
|
||||||
delete(ImageRecord).where(ImageRecord.id.in_(ids))
|
|
||||||
)
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
images_deleted += len(ids)
|
|
||||||
|
|
||||||
if dry_run:
|
|
||||||
# Nothing was actually deleted from the DB; bail after one
|
|
||||||
# pass so we don't loop forever.
|
|
||||||
break
|
|
||||||
|
|
||||||
import_tasks_deleted = 0
|
|
||||||
if source_path_prefix and not dry_run:
|
|
||||||
# Delete ImportTask rows whose source_path is under the bad mount
|
|
||||||
# prefix. These are mostly orphaned now (result_image_id was set
|
|
||||||
# NULL by CASCADE) but their presence still blocks the
|
|
||||||
# idempotency check in scan_directory if the operator remounts
|
|
||||||
# the same prefix.
|
|
||||||
like_pattern = source_path_prefix.rstrip("/") + "/%"
|
|
||||||
result = await db.execute(
|
|
||||||
delete(ImportTask).where(ImportTask.source_path.like(like_pattern))
|
|
||||||
)
|
|
||||||
import_tasks_deleted = result.rowcount or 0
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
# Sweep ImportBatch rows that are now empty.
|
|
||||||
empty_batches_deleted = 0
|
|
||||||
if not dry_run:
|
|
||||||
empty_batch_ids = (await db.execute(
|
|
||||||
select(ImportBatch.id).where(
|
|
||||||
~select(ImportTask.id)
|
|
||||||
.where(ImportTask.batch_id == ImportBatch.id)
|
|
||||||
.exists()
|
|
||||||
)
|
|
||||||
)).scalars().all()
|
|
||||||
if empty_batch_ids:
|
|
||||||
result = await db.execute(
|
|
||||||
delete(ImportBatch).where(ImportBatch.id.in_(empty_batch_ids))
|
|
||||||
)
|
|
||||||
empty_batches_deleted = result.rowcount or 0
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
# Finally, the artist row.
|
|
||||||
if not dry_run:
|
|
||||||
await db.execute(delete(Artist).where(Artist.id == artist_id))
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
return {
|
|
||||||
"counts": counts,
|
|
||||||
"artist": {"id": artist_id, "name": artist_name, "slug": slug},
|
|
||||||
"summary": {
|
|
||||||
"images_targeted": total_images,
|
|
||||||
"images_deleted": images_deleted,
|
|
||||||
"files_deleted": files_deleted,
|
|
||||||
"thumbs_deleted": thumbs_deleted,
|
|
||||||
"import_tasks_deleted": import_tasks_deleted,
|
|
||||||
"empty_batches_deleted": empty_batches_deleted,
|
|
||||||
"dry_run": dry_run,
|
|
||||||
},
|
|
||||||
}
|
|
||||||
@@ -1,126 +0,0 @@
|
|||||||
"""GallerySubscriber export → FabledCurator ingest.
|
|
||||||
|
|
||||||
Reads a parsed gallerysubscriber-export-v1.json dict (no DB connection
|
|
||||||
to GS). Creates Artist (from subscriptions) + Source (nested under each
|
|
||||||
subscription) + Credential (re-encrypted with FC's key). Idempotent on
|
|
||||||
natural keys: Artist.slug, (artist_id, platform, url), Credential.platform.
|
|
||||||
|
|
||||||
Credentials arrive plaintext in the export — GS's export script
|
|
||||||
decrypts using GS's Fernet key in GS's own process. FC re-encrypts
|
|
||||||
with FC's CredentialCrypto.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
|
|
||||||
from sqlalchemy import select
|
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
|
||||||
|
|
||||||
from ...models import Artist, Credential, Source
|
|
||||||
from ...utils.slug import slugify
|
|
||||||
from ..credential_crypto import CredentialCrypto
|
|
||||||
|
|
||||||
|
|
||||||
def _zero_counts() -> dict:
|
|
||||||
return {
|
|
||||||
"rows_processed": 0, "rows_inserted": 0, "rows_skipped": 0,
|
|
||||||
"files_copied": 0, "bytes_copied": 0, "conflicts": 0,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
async def migrate_async(
|
|
||||||
db: AsyncSession,
|
|
||||||
*,
|
|
||||||
data: dict,
|
|
||||||
fc_crypto: CredentialCrypto | None = None,
|
|
||||||
dry_run: bool = False,
|
|
||||||
) -> dict:
|
|
||||||
"""Ingest a parsed gallerysubscriber-export-v1.json dict."""
|
|
||||||
if data.get("source_app") != "gallerysubscriber":
|
|
||||||
raise ValueError("export source_app must be 'gallerysubscriber'")
|
|
||||||
if data.get("schema_version") != 1:
|
|
||||||
raise ValueError(f"unsupported schema_version: {data.get('schema_version')}")
|
|
||||||
|
|
||||||
counts = _zero_counts()
|
|
||||||
|
|
||||||
# Phase 1: subscriptions → Artist; nested sources within each.
|
|
||||||
for sub in data.get("subscriptions", []):
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
slug = slugify(sub["name"])
|
|
||||||
artist = (await db.execute(
|
|
||||||
select(Artist).where(Artist.slug == slug)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if artist is None:
|
|
||||||
if dry_run:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
# Continue to nested sources, but they can't link without an artist row.
|
|
||||||
continue
|
|
||||||
notes = json.dumps(sub.get("metadata"), indent=2) if sub.get("metadata") else None
|
|
||||||
artist = Artist(
|
|
||||||
name=sub["name"], slug=slug,
|
|
||||||
is_subscription=True,
|
|
||||||
auto_check=bool(sub.get("enabled", True)),
|
|
||||||
notes=notes,
|
|
||||||
)
|
|
||||||
db.add(artist)
|
|
||||||
await db.flush()
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
else:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
|
|
||||||
# Nested sources under this subscription.
|
|
||||||
for src in sub.get("sources", []):
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(Source).where(
|
|
||||||
Source.artist_id == artist.id,
|
|
||||||
Source.platform == src["platform"],
|
|
||||||
Source.url == src["url"],
|
|
||||||
)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
if dry_run:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
continue
|
|
||||||
db.add(Source(
|
|
||||||
artist_id=artist.id,
|
|
||||||
platform=src["platform"],
|
|
||||||
url=src["url"],
|
|
||||||
enabled=bool(src.get("enabled", True)),
|
|
||||||
check_interval_override=src.get("check_interval"),
|
|
||||||
config_overrides=src.get("metadata") or {},
|
|
||||||
))
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
|
|
||||||
# Phase 2: credentials.
|
|
||||||
for cred in data.get("credentials", []):
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(Credential).where(Credential.platform == cred["platform"])
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
if dry_run:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
continue
|
|
||||||
if fc_crypto is None:
|
|
||||||
# Without a crypto helper we can't encrypt — skip rather than
|
|
||||||
# store plaintext.
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
counts["conflicts"] += 1
|
|
||||||
continue
|
|
||||||
encrypted = fc_crypto.encrypt(cred["plaintext"])
|
|
||||||
db.add(Credential(
|
|
||||||
platform=cred["platform"],
|
|
||||||
credential_type=cred.get("credential_type") or "cookies",
|
|
||||||
encrypted_blob=encrypted,
|
|
||||||
expires_at=cred.get("expires_at"),
|
|
||||||
))
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
|
|
||||||
if not dry_run:
|
|
||||||
await db.commit()
|
|
||||||
return counts
|
|
||||||
@@ -1,146 +0,0 @@
|
|||||||
"""ImageRepo export → FabledCurator ingest.
|
|
||||||
|
|
||||||
Reads a parsed imagerepo-export-v1.json dict (no DB connection to IR).
|
|
||||||
Creates Tag rows (skipping artist/post kinds, resolving fandom_name to
|
|
||||||
FK). Writes the per-image-sha256 artist assignments + tag associations
|
|
||||||
+ series page assignments to /images/_migration_state/ir_tag_manifest.json
|
|
||||||
so tag_apply.py can join them to ImageRecord rows AFTER the operator
|
|
||||||
runs FC's filesystem scan.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from sqlalchemy import select
|
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
|
||||||
|
|
||||||
from ...models import Tag, TagKind
|
|
||||||
|
|
||||||
_SKIP_KINDS = frozenset({"artist", "post"})
|
|
||||||
_MIGRATION_STATE_DIRNAME = "_migration_state"
|
|
||||||
_IR_MANIFEST_FILENAME = "ir_tag_manifest.json"
|
|
||||||
|
|
||||||
|
|
||||||
def _zero_counts() -> dict:
|
|
||||||
return {
|
|
||||||
"rows_processed": 0, "rows_inserted": 0, "rows_skipped": 0,
|
|
||||||
"files_copied": 0, "bytes_copied": 0, "conflicts": 0,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def manifest_path(images_root: Path | None = None) -> Path:
|
|
||||||
root = images_root if images_root is not None else Path("/images")
|
|
||||||
p = root / _MIGRATION_STATE_DIRNAME
|
|
||||||
p.mkdir(parents=True, exist_ok=True)
|
|
||||||
return p / _IR_MANIFEST_FILENAME
|
|
||||||
|
|
||||||
|
|
||||||
async def _resolve_fandom_id(
|
|
||||||
db: AsyncSession, fandom_name: str | None, dry_run: bool,
|
|
||||||
) -> int | None:
|
|
||||||
"""Find-or-create a fandom-kind Tag by name."""
|
|
||||||
if not fandom_name:
|
|
||||||
return None
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(Tag).where(Tag.name == fandom_name, Tag.kind == "fandom")
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
return existing.id
|
|
||||||
if dry_run:
|
|
||||||
return None
|
|
||||||
t = Tag(name=fandom_name, kind=TagKind.fandom)
|
|
||||||
db.add(t)
|
|
||||||
await db.flush()
|
|
||||||
return t.id
|
|
||||||
|
|
||||||
|
|
||||||
async def migrate_async(
|
|
||||||
db: AsyncSession,
|
|
||||||
*,
|
|
||||||
data: dict,
|
|
||||||
images_root: Path | None = None,
|
|
||||||
dry_run: bool = False,
|
|
||||||
) -> dict:
|
|
||||||
"""Ingest a parsed imagerepo-export-v1.json dict.
|
|
||||||
|
|
||||||
Creates Tag rows + writes the IR tag manifest file. Tag-to-image
|
|
||||||
binding happens later in tag_apply.py (after FC's filesystem scan
|
|
||||||
populates image_record.sha256 → id).
|
|
||||||
"""
|
|
||||||
if data.get("source_app") != "imagerepo":
|
|
||||||
raise ValueError("export source_app must be 'imagerepo'")
|
|
||||||
if data.get("schema_version") not in (1, 2):
|
|
||||||
raise ValueError(f"unsupported schema_version: {data.get('schema_version')}")
|
|
||||||
|
|
||||||
counts = _zero_counts()
|
|
||||||
|
|
||||||
# Phase 1: tags (skip artist + post kinds; resolve fandom_name → fandom_id).
|
|
||||||
# First pass: create all fandom-kind tags so they're available for FK resolution.
|
|
||||||
for tag in data.get("tags", []):
|
|
||||||
kind = tag.get("kind") or "general"
|
|
||||||
if kind != "fandom":
|
|
||||||
continue
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(Tag).where(Tag.name == tag["name"], Tag.kind == "fandom")
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
if dry_run:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
continue
|
|
||||||
db.add(Tag(name=tag["name"], kind=TagKind.fandom))
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
if not dry_run:
|
|
||||||
await db.flush()
|
|
||||||
|
|
||||||
# Second pass: every other kind.
|
|
||||||
for tag in data.get("tags", []):
|
|
||||||
kind_str = tag.get("kind") or "general"
|
|
||||||
if kind_str in _SKIP_KINDS:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
if kind_str == "fandom":
|
|
||||||
continue # handled above
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
try:
|
|
||||||
kind = TagKind(kind_str)
|
|
||||||
except ValueError:
|
|
||||||
kind = TagKind.general
|
|
||||||
fandom_id = await _resolve_fandom_id(db, tag.get("fandom_name"), dry_run)
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(Tag).where(Tag.name == tag["name"], Tag.kind == kind)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
if dry_run:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
continue
|
|
||||||
db.add(Tag(name=tag["name"], kind=kind, fandom_id=fandom_id))
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
|
|
||||||
if not dry_run:
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
# Phase 2: write the per-image manifest for tag_apply.py to consume later.
|
|
||||||
# schema_version 2 (added 2026-05-24) carries `image_posts` for
|
|
||||||
# Post + Source + ImageProvenance restore; schema 1 manifests
|
|
||||||
# without it stay valid (tag_apply treats the missing field as []).
|
|
||||||
manifest = {
|
|
||||||
"schema_version": data.get("schema_version", 1),
|
|
||||||
"image_artist_assignments": data.get("image_artist_assignments", []),
|
|
||||||
"image_tag_associations": data.get("image_tag_associations", []),
|
|
||||||
"series_pages": data.get("series_pages", []),
|
|
||||||
"image_posts": data.get("image_posts", []),
|
|
||||||
}
|
|
||||||
counts["rows_processed"] += len(manifest["image_artist_assignments"])
|
|
||||||
counts["rows_processed"] += len(manifest["image_tag_associations"])
|
|
||||||
counts["rows_processed"] += len(manifest["series_pages"])
|
|
||||||
counts["rows_processed"] += len(manifest["image_posts"])
|
|
||||||
if not dry_run:
|
|
||||||
manifest_path(images_root).write_text(json.dumps(manifest, indent=2))
|
|
||||||
|
|
||||||
return counts
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
"""Queue every migrated image_record with no embedding for ML re-processing."""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from sqlalchemy import select
|
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
|
||||||
|
|
||||||
from ...models import ImageRecord
|
|
||||||
|
|
||||||
|
|
||||||
async def queue_all_unprocessed_async(db: AsyncSession) -> int:
|
|
||||||
"""Find every ImageRecord with siglip_embedding IS NULL, fire
|
|
||||||
tag_and_embed.delay(id) for each. Returns count queued.
|
|
||||||
"""
|
|
||||||
from ...tasks.ml import tag_and_embed
|
|
||||||
|
|
||||||
rows = (await db.execute(
|
|
||||||
select(ImageRecord.id).where(ImageRecord.siglip_embedding.is_(None))
|
|
||||||
)).scalars().all()
|
|
||||||
for image_id in rows:
|
|
||||||
tag_and_embed.delay(image_id)
|
|
||||||
return len(rows)
|
|
||||||
@@ -1,368 +0,0 @@
|
|||||||
"""Apply the IR tag manifest after FC's filesystem scan.
|
|
||||||
|
|
||||||
Reads /images/_migration_state/ir_tag_manifest.json and joins each entry
|
|
||||||
to an ImageRecord row by sha256 (which exists after the operator runs
|
|
||||||
FC's filesystem scan over the mounted IR images dir).
|
|
||||||
|
|
||||||
- image_artist_assignments → ImageRecord.artist_id (find_or_create Artist by slug).
|
|
||||||
- image_tag_associations → image_tag insert (idempotent).
|
|
||||||
- series_pages → series_page insert (idempotent on image_id unique).
|
|
||||||
- image_posts (schema v2) → Source + Post + ImageProvenance restore.
|
|
||||||
|
|
||||||
Unmatched sha256s are logged into the result's `unmatched` list so the
|
|
||||||
Celery task can drop them into MigrationRun.metadata for the operator
|
|
||||||
to inspect.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
from datetime import datetime
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from sqlalchemy import select
|
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
|
||||||
|
|
||||||
from ...models import (
|
|
||||||
Artist,
|
|
||||||
ImageProvenance,
|
|
||||||
ImageRecord,
|
|
||||||
Post,
|
|
||||||
SeriesPage,
|
|
||||||
Source,
|
|
||||||
Tag,
|
|
||||||
TagKind,
|
|
||||||
image_tag,
|
|
||||||
)
|
|
||||||
from ...utils.slug import slugify
|
|
||||||
from .ir_ingest import manifest_path
|
|
||||||
|
|
||||||
# Per-platform artist-profile URL — used as Source.url when restoring
|
|
||||||
# IR PostMetadata into FC. Must cover every platform that
|
|
||||||
# backend/app/services/extension_service.py:_PLATFORM_PATTERNS
|
|
||||||
# recognizes; an entry missing here silently drops ALL PostMetadata for
|
|
||||||
# that platform during phase 4 (operator hit this 2026-05-25:
|
|
||||||
# DeviantArt + Pixiv posts in the IR migration produced empty
|
|
||||||
# ImageProvenance because they fell through this table).
|
|
||||||
#
|
|
||||||
# Pixiv caveat: the real profile URL takes a numeric user_id
|
|
||||||
# (https://www.pixiv.net/users/12345), but IR's PostMetadata.artist
|
|
||||||
# stores the display name not the id. We use the slugified name here
|
|
||||||
# so we preserve the artist→post→image linkage; the resulting Source.url
|
|
||||||
# won't resolve in a browser and the operator may want to manually fix
|
|
||||||
# it via Settings → Subscriptions once the migration lands.
|
|
||||||
_PLATFORM_PROFILE_URL = {
|
|
||||||
"patreon": "https://www.patreon.com/{slug}",
|
|
||||||
"subscribestar": "https://www.subscribestar.com/{slug}",
|
|
||||||
"hentaifoundry": "https://www.hentai-foundry.com/user/{slug}",
|
|
||||||
"deviantart": "https://www.deviantart.com/{slug}",
|
|
||||||
"pixiv": "https://www.pixiv.net/users/{slug}",
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def _profile_url(platform: str, artist_slug: str) -> str | None:
|
|
||||||
fmt = _PLATFORM_PROFILE_URL.get(platform)
|
|
||||||
return fmt.format(slug=artist_slug) if fmt else None
|
|
||||||
|
|
||||||
|
|
||||||
async def _find_or_create_source(
|
|
||||||
db: AsyncSession, *, artist_id: int, platform: str, url: str, dry_run: bool,
|
|
||||||
) -> int | None:
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(Source.id).where(
|
|
||||||
Source.artist_id == artist_id,
|
|
||||||
Source.platform == platform,
|
|
||||||
Source.url == url,
|
|
||||||
)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
return existing
|
|
||||||
if dry_run:
|
|
||||||
return None
|
|
||||||
s = Source(artist_id=artist_id, platform=platform, url=url, enabled=False)
|
|
||||||
db.add(s)
|
|
||||||
await db.flush()
|
|
||||||
return s.id
|
|
||||||
|
|
||||||
|
|
||||||
async def _find_or_create_post(
|
|
||||||
db: AsyncSession, *,
|
|
||||||
source_id: int, external_post_id: str,
|
|
||||||
title: str | None, description: str | None, post_url: str | None,
|
|
||||||
post_date_iso: str | None, attachment_count: int, dry_run: bool,
|
|
||||||
) -> int | None:
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(Post.id).where(
|
|
||||||
Post.source_id == source_id,
|
|
||||||
Post.external_post_id == external_post_id,
|
|
||||||
)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
return existing
|
|
||||||
if dry_run:
|
|
||||||
return None
|
|
||||||
post_date = None
|
|
||||||
if post_date_iso:
|
|
||||||
post_date = datetime.fromisoformat(post_date_iso)
|
|
||||||
p = Post(
|
|
||||||
source_id=source_id,
|
|
||||||
external_post_id=external_post_id,
|
|
||||||
post_title=title,
|
|
||||||
description=description,
|
|
||||||
post_url=post_url,
|
|
||||||
post_date=post_date,
|
|
||||||
attachment_count=attachment_count,
|
|
||||||
raw_metadata={"migrated_from": "imagerepo"},
|
|
||||||
)
|
|
||||||
db.add(p)
|
|
||||||
await db.flush()
|
|
||||||
return p.id
|
|
||||||
|
|
||||||
|
|
||||||
async def _ensure_provenance(
|
|
||||||
db: AsyncSession, *,
|
|
||||||
image_id: int, post_id: int, source_id: int, dry_run: bool,
|
|
||||||
) -> bool:
|
|
||||||
"""Returns True if a new ImageProvenance row was inserted.
|
|
||||||
|
|
||||||
Also sets ImageRecord.primary_post_id to this post if the image
|
|
||||||
doesn't already have one — preserves any primary_post_id already
|
|
||||||
assigned at download time by the importer (don't clobber). This is
|
|
||||||
the linkage gallery_service.py uses to surface Post.post_date as
|
|
||||||
the image's effective date for sort/group/jump/neighbor nav.
|
|
||||||
"""
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(ImageProvenance.id).where(
|
|
||||||
ImageProvenance.image_record_id == image_id,
|
|
||||||
ImageProvenance.post_id == post_id,
|
|
||||||
ImageProvenance.source_id == source_id,
|
|
||||||
)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
|
|
||||||
# Whether-or-not the provenance row already exists, ensure the
|
|
||||||
# image's primary_post_id is set so the gallery date-coalesce works.
|
|
||||||
# Idempotent: only writes when currently NULL.
|
|
||||||
if not dry_run:
|
|
||||||
await db.execute(
|
|
||||||
ImageRecord.__table__.update()
|
|
||||||
.where(ImageRecord.id == image_id)
|
|
||||||
.where(ImageRecord.primary_post_id.is_(None))
|
|
||||||
.values(primary_post_id=post_id)
|
|
||||||
)
|
|
||||||
|
|
||||||
if existing is not None:
|
|
||||||
return False
|
|
||||||
if dry_run:
|
|
||||||
return True
|
|
||||||
db.add(ImageProvenance(
|
|
||||||
image_record_id=image_id, post_id=post_id, source_id=source_id,
|
|
||||||
))
|
|
||||||
await db.flush()
|
|
||||||
return True
|
|
||||||
|
|
||||||
|
|
||||||
def _zero_counts() -> dict:
|
|
||||||
return {
|
|
||||||
"rows_processed": 0, "rows_inserted": 0, "rows_skipped": 0,
|
|
||||||
"files_copied": 0, "bytes_copied": 0, "conflicts": 0,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
async def _ensure_artist_id(
|
|
||||||
db: AsyncSession, artist_name: str, dry_run: bool,
|
|
||||||
) -> int | None:
|
|
||||||
if not artist_name or not artist_name.strip():
|
|
||||||
return None
|
|
||||||
slug = slugify(artist_name)
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(Artist).where(Artist.slug == slug)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
return existing.id
|
|
||||||
if dry_run:
|
|
||||||
return None
|
|
||||||
a = Artist(name=artist_name, slug=slug, is_subscription=False)
|
|
||||||
db.add(a)
|
|
||||||
await db.flush()
|
|
||||||
return a.id
|
|
||||||
|
|
||||||
|
|
||||||
async def _resolve_tag_id(
|
|
||||||
db: AsyncSession, tag_name: str, tag_kind: str,
|
|
||||||
) -> int | None:
|
|
||||||
try:
|
|
||||||
kind = TagKind(tag_kind)
|
|
||||||
except ValueError:
|
|
||||||
kind = TagKind.general
|
|
||||||
row = (await db.execute(
|
|
||||||
select(Tag.id).where(Tag.name == tag_name, Tag.kind == kind)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
return row
|
|
||||||
|
|
||||||
|
|
||||||
async def _sha_to_image_id(db: AsyncSession, sha: str) -> int | None:
|
|
||||||
return (await db.execute(
|
|
||||||
select(ImageRecord.id).where(ImageRecord.sha256 == sha)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
|
|
||||||
|
|
||||||
async def apply_async(
|
|
||||||
db: AsyncSession,
|
|
||||||
*,
|
|
||||||
images_root: Path | None = None,
|
|
||||||
dry_run: bool = False,
|
|
||||||
) -> dict:
|
|
||||||
"""Apply the manifest. Returns counts + an `unmatched` list of sha256s."""
|
|
||||||
mf_path = manifest_path(images_root)
|
|
||||||
if not mf_path.exists():
|
|
||||||
raise FileNotFoundError(f"no IR tag manifest at {mf_path}")
|
|
||||||
|
|
||||||
manifest = json.loads(mf_path.read_text())
|
|
||||||
counts = _zero_counts()
|
|
||||||
unmatched: list[dict] = []
|
|
||||||
|
|
||||||
# 1. Artist assignments.
|
|
||||||
for entry in manifest.get("image_artist_assignments", []):
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
img_id = await _sha_to_image_id(db, entry["sha256"])
|
|
||||||
if img_id is None:
|
|
||||||
unmatched.append({"kind": "artist", **entry})
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
aid = await _ensure_artist_id(db, entry["artist_name"], dry_run)
|
|
||||||
if aid is None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
if dry_run:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
continue
|
|
||||||
img = await db.get(ImageRecord, img_id)
|
|
||||||
if img is not None and img.artist_id != aid:
|
|
||||||
img.artist_id = aid
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
else:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
|
|
||||||
# 2. Tag associations.
|
|
||||||
for entry in manifest.get("image_tag_associations", []):
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
img_id = await _sha_to_image_id(db, entry["sha256"])
|
|
||||||
if img_id is None:
|
|
||||||
unmatched.append({"kind": "tag", **entry})
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
tag_id = await _resolve_tag_id(
|
|
||||||
db, entry["tag_name"], entry.get("tag_kind") or "general",
|
|
||||||
)
|
|
||||||
if tag_id is None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
# Skip if association already exists.
|
|
||||||
already = (await db.execute(
|
|
||||||
select(image_tag.c.image_record_id).where(
|
|
||||||
image_tag.c.image_record_id == img_id,
|
|
||||||
image_tag.c.tag_id == tag_id,
|
|
||||||
)
|
|
||||||
)).first()
|
|
||||||
if already is not None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
if dry_run:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
continue
|
|
||||||
await db.execute(image_tag.insert().values(
|
|
||||||
image_record_id=img_id, tag_id=tag_id, source="manual",
|
|
||||||
))
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
|
|
||||||
# 3. Series pages.
|
|
||||||
for entry in manifest.get("series_pages", []):
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
img_id = await _sha_to_image_id(db, entry["sha256"])
|
|
||||||
if img_id is None:
|
|
||||||
unmatched.append({"kind": "series", **entry})
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
series_tag_id = await _resolve_tag_id(db, entry["series_tag_name"], "series")
|
|
||||||
if series_tag_id is None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
existing = (await db.execute(
|
|
||||||
select(SeriesPage).where(SeriesPage.image_id == img_id)
|
|
||||||
)).scalar_one_or_none()
|
|
||||||
if existing is not None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
if dry_run:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
continue
|
|
||||||
db.add(SeriesPage(
|
|
||||||
series_tag_id=series_tag_id,
|
|
||||||
image_id=img_id,
|
|
||||||
page_number=entry["page_number"],
|
|
||||||
))
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
|
|
||||||
# 4. Image posts (schema v2) → Source + Post + ImageProvenance.
|
|
||||||
# Restores IR PostMetadata as FC's downloader-track provenance,
|
|
||||||
# so the modal's ProvenancePanel surfaces title/description/
|
|
||||||
# source URL/publish date the same way it does for live
|
|
||||||
# gallery-dl downloads.
|
|
||||||
for entry in manifest.get("image_posts", []):
|
|
||||||
counts["rows_processed"] += 1
|
|
||||||
platform = entry.get("platform")
|
|
||||||
artist_name = entry.get("artist")
|
|
||||||
if not platform or not artist_name:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
|
|
||||||
aid = await _ensure_artist_id(db, artist_name, dry_run)
|
|
||||||
if aid is None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
|
|
||||||
url = _profile_url(platform, slugify(artist_name))
|
|
||||||
if url is None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
|
|
||||||
source_id = await _find_or_create_source(
|
|
||||||
db, artist_id=aid, platform=platform, url=url, dry_run=dry_run,
|
|
||||||
)
|
|
||||||
if source_id is None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
|
|
||||||
post_id = await _find_or_create_post(
|
|
||||||
db, source_id=source_id,
|
|
||||||
external_post_id=entry.get("post_id") or "",
|
|
||||||
title=entry.get("title"),
|
|
||||||
description=entry.get("description"),
|
|
||||||
post_url=entry.get("source_url"),
|
|
||||||
post_date_iso=entry.get("published_at"),
|
|
||||||
attachment_count=entry.get("attachment_count") or 0,
|
|
||||||
dry_run=dry_run,
|
|
||||||
)
|
|
||||||
if post_id is None:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
continue
|
|
||||||
|
|
||||||
for sha in entry.get("image_sha256s", []):
|
|
||||||
img_id = await _sha_to_image_id(db, sha)
|
|
||||||
if img_id is None:
|
|
||||||
unmatched.append({
|
|
||||||
"kind": "post", "sha256": sha,
|
|
||||||
"post_id": entry.get("post_id"),
|
|
||||||
})
|
|
||||||
continue
|
|
||||||
inserted = await _ensure_provenance(
|
|
||||||
db, image_id=img_id, post_id=post_id,
|
|
||||||
source_id=source_id, dry_run=dry_run,
|
|
||||||
)
|
|
||||||
if inserted:
|
|
||||||
counts["rows_inserted"] += 1
|
|
||||||
else:
|
|
||||||
counts["rows_skipped"] += 1
|
|
||||||
|
|
||||||
if not dry_run:
|
|
||||||
await db.commit()
|
|
||||||
return {"counts": counts, "unmatched": unmatched}
|
|
||||||
@@ -1,80 +0,0 @@
|
|||||||
"""Post-migration verification: row counts + sha256 sampling."""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import hashlib
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from sqlalchemy import func, select
|
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
|
||||||
|
|
||||||
from ...models import Artist, Credential, ImageRecord, Source, Tag
|
|
||||||
|
|
||||||
|
|
||||||
async def verify_async(db: AsyncSession, *, expected: dict | None = None) -> dict:
|
|
||||||
"""Return per-check status dicts. `expected` is optional row-count
|
|
||||||
assertions; checks default to status='ok' when no expected provided."""
|
|
||||||
expected = expected or {}
|
|
||||||
results: dict[str, dict] = {}
|
|
||||||
|
|
||||||
checks = {
|
|
||||||
"artist_subscriptions": (
|
|
||||||
select(func.count(Artist.id)).where(Artist.is_subscription.is_(True))
|
|
||||||
),
|
|
||||||
"source_count": select(func.count(Source.id)),
|
|
||||||
"credential_count": select(func.count(Credential.id)),
|
|
||||||
"tag_count": select(func.count(Tag.id)),
|
|
||||||
"image_record_imported_or_downloaded": (
|
|
||||||
select(func.count(ImageRecord.id))
|
|
||||||
.where(ImageRecord.origin.in_(["imported_filesystem", "downloaded"]))
|
|
||||||
),
|
|
||||||
}
|
|
||||||
for name, stmt in checks.items():
|
|
||||||
actual = (await db.execute(stmt)).scalar_one()
|
|
||||||
exp = expected.get(name)
|
|
||||||
status = "ok" if exp is None or exp == actual else "mismatch"
|
|
||||||
results[name] = {"status": status, "actual": int(actual), "expected": exp}
|
|
||||||
return results
|
|
||||||
|
|
||||||
|
|
||||||
async def verify_sha256_sample(
|
|
||||||
db: AsyncSession, *, sample_size: int = 20,
|
|
||||||
) -> dict:
|
|
||||||
"""Sample N image_records; verify file exists + sha256 matches."""
|
|
||||||
rows = (await db.execute(
|
|
||||||
select(ImageRecord.id, ImageRecord.path, ImageRecord.sha256)
|
|
||||||
.order_by(func.random()).limit(sample_size)
|
|
||||||
)).all()
|
|
||||||
|
|
||||||
matched = 0
|
|
||||||
mismatched = 0
|
|
||||||
missing = 0
|
|
||||||
samples: list[dict] = []
|
|
||||||
for img_id, path, expected_sha in rows:
|
|
||||||
p = Path(path)
|
|
||||||
# Sync stdlib filesystem ops are intentional: this verify pass runs
|
|
||||||
# inside a Celery task under asyncio.run; no other awaitables compete
|
|
||||||
# for the loop. Same pattern as download_service.py.
|
|
||||||
if not p.exists(): # noqa: ASYNC240
|
|
||||||
missing += 1
|
|
||||||
samples.append({"id": img_id, "path": path, "result": "missing"})
|
|
||||||
continue
|
|
||||||
h = hashlib.sha256()
|
|
||||||
with p.open("rb") as f: # noqa: ASYNC230
|
|
||||||
for chunk in iter(lambda: f.read(65536), b""):
|
|
||||||
h.update(chunk)
|
|
||||||
if h.hexdigest() == expected_sha:
|
|
||||||
matched += 1
|
|
||||||
samples.append({"id": img_id, "result": "ok"})
|
|
||||||
else:
|
|
||||||
mismatched += 1
|
|
||||||
samples.append({
|
|
||||||
"id": img_id, "path": path, "result": "mismatch",
|
|
||||||
"expected_sha": expected_sha, "actual_sha": h.hexdigest(),
|
|
||||||
})
|
|
||||||
return {
|
|
||||||
"sample_size": len(rows),
|
|
||||||
"matched": matched,
|
|
||||||
"mismatched": mismatched,
|
|
||||||
"missing": missing,
|
|
||||||
"samples": samples,
|
|
||||||
}
|
|
||||||
@@ -1,7 +1,7 @@
|
|||||||
"""Alias resolution + CRUD.
|
"""Alias resolution + CRUD.
|
||||||
|
|
||||||
A tag_alias maps (model_name, model_category) -> canonical Tag. Resolution
|
A tag_alias maps (model_name, model_category) -> canonical Tag. Resolution
|
||||||
happens at suggestion-read time so raw tagger_predictions stay unmolested.
|
happens at suggestion-read time so the raw image_prediction rows stay unmolested.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from collections.abc import Sequence
|
from collections.abc import Sequence
|
||||||
@@ -81,6 +81,31 @@ class AliasService:
|
|||||||
.where(TagAlias.alias_category == alias_category)
|
.where(TagAlias.alias_category == alias_category)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
async def list_for_tag(self, canonical_tag_id: int) -> Sequence[AliasRow]:
|
||||||
|
"""Aliases that resolve TO this tag — drives the tag-side 'Aliases'
|
||||||
|
view (see/remove the model keys that fold into a tag)."""
|
||||||
|
stmt = (
|
||||||
|
select(
|
||||||
|
TagAlias.alias_string,
|
||||||
|
TagAlias.alias_category,
|
||||||
|
TagAlias.canonical_tag_id,
|
||||||
|
Tag.name,
|
||||||
|
)
|
||||||
|
.join(Tag, Tag.id == TagAlias.canonical_tag_id)
|
||||||
|
.where(TagAlias.canonical_tag_id == canonical_tag_id)
|
||||||
|
.order_by(TagAlias.alias_string.asc())
|
||||||
|
)
|
||||||
|
rows = (await self.session.execute(stmt)).all()
|
||||||
|
return [
|
||||||
|
AliasRow(
|
||||||
|
alias_string=r[0],
|
||||||
|
alias_category=r[1],
|
||||||
|
canonical_tag_id=r[2],
|
||||||
|
canonical_tag_name=r[3],
|
||||||
|
)
|
||||||
|
for r in rows
|
||||||
|
]
|
||||||
|
|
||||||
async def list_all(self) -> Sequence[AliasRow]:
|
async def list_all(self) -> Sequence[AliasRow]:
|
||||||
stmt = (
|
stmt = (
|
||||||
select(
|
select(
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ from sqlalchemy import delete, select
|
|||||||
from sqlalchemy.dialects.postgresql import insert
|
from sqlalchemy.dialects.postgresql import insert
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from ...models import Tag, TagAllowlist, TagSuggestionRejection
|
from ...models import MLSettings, Tag, TagAllowlist, TagSuggestionRejection
|
||||||
from ...models.tag import image_tag
|
from ...models.tag import image_tag
|
||||||
from .aliases import AliasService
|
from .aliases import AliasService
|
||||||
|
|
||||||
@@ -91,12 +91,25 @@ class AllowlistService:
|
|||||||
)
|
)
|
||||||
await self.dismiss(image_id, tag_id)
|
await self.dismiss(image_id, tag_id)
|
||||||
|
|
||||||
|
async def _store_floor(self) -> float:
|
||||||
|
return (
|
||||||
|
await self.session.execute(
|
||||||
|
select(MLSettings.tagger_store_floor).where(MLSettings.id == 1)
|
||||||
|
)
|
||||||
|
).scalar_one()
|
||||||
|
|
||||||
async def update_threshold(
|
async def update_threshold(
|
||||||
self, tag_id: int, min_confidence: float
|
self, tag_id: int, min_confidence: float
|
||||||
) -> None:
|
) -> None:
|
||||||
row = await self.session.get(TagAllowlist, tag_id)
|
row = await self.session.get(TagAllowlist, tag_id)
|
||||||
if row is not None:
|
if row is not None:
|
||||||
row.min_confidence = min_confidence
|
# An allowlist tag can't auto-apply more permissively than the
|
||||||
|
# ingest store floor — predictions below tagger_store_floor aren't
|
||||||
|
# stored, so a lower min_confidence would behave identically to the
|
||||||
|
# floor. Clamp so the stored threshold matches actual behavior
|
||||||
|
# (#764).
|
||||||
|
floor = await self._store_floor()
|
||||||
|
row.min_confidence = max(min_confidence, floor)
|
||||||
|
|
||||||
async def remove(self, tag_id: int) -> None:
|
async def remove(self, tag_id: int) -> None:
|
||||||
await self.session.execute(
|
await self.session.execute(
|
||||||
|
|||||||
@@ -19,7 +19,6 @@ from ...models import (
|
|||||||
TagReferenceEmbedding,
|
TagReferenceEmbedding,
|
||||||
)
|
)
|
||||||
from ...models.tag import image_tag
|
from ...models.tag import image_tag
|
||||||
from .embedder import MODEL_VERSION as SIGLIP_VERSION
|
|
||||||
|
|
||||||
ELIGIBLE_KINDS = {
|
ELIGIBLE_KINDS = {
|
||||||
TagKind.character,
|
TagKind.character,
|
||||||
@@ -46,6 +45,21 @@ class CentroidService:
|
|||||||
)
|
)
|
||||||
).scalar_one()
|
).scalar_one()
|
||||||
|
|
||||||
|
async def _model_version(self) -> str:
|
||||||
|
"""Audit 2026-06-02: SigLIP model-version stamp comes from the
|
||||||
|
DB row, not the env constant. tag_and_embed (tasks/ml.py:110)
|
||||||
|
already reads from MLSettings.embedder_model_version, so by
|
||||||
|
sourcing centroid stamps + drift checks from the same row, we
|
||||||
|
eliminate the silent-drift case the audit flagged. env
|
||||||
|
SIGLIP_MODEL_VERSION still drives which model embedder.py
|
||||||
|
loads at runtime; the version stamp is purely the operator-
|
||||||
|
controlled identifier."""
|
||||||
|
return (
|
||||||
|
await self.session.execute(
|
||||||
|
select(MLSettings.embedder_model_version).where(MLSettings.id == 1)
|
||||||
|
)
|
||||||
|
).scalar_one()
|
||||||
|
|
||||||
async def recompute_for_tag(self, tag_id: int) -> bool:
|
async def recompute_for_tag(self, tag_id: int) -> bool:
|
||||||
"""Recompute one tag's centroid. Returns True if a centroid was
|
"""Recompute one tag's centroid. Returns True if a centroid was
|
||||||
written, False if skipped (ineligible kind or too few members)."""
|
written, False if skipped (ineligible kind or too few members)."""
|
||||||
@@ -69,19 +83,20 @@ class CentroidService:
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
centroid = np.mean(np.stack(embeddings), axis=0).astype(np.float32)
|
centroid = np.mean(np.stack(embeddings), axis=0).astype(np.float32)
|
||||||
|
model_version = await self._model_version()
|
||||||
|
|
||||||
stmt = insert(TagReferenceEmbedding).values(
|
stmt = insert(TagReferenceEmbedding).values(
|
||||||
tag_id=tag_id,
|
tag_id=tag_id,
|
||||||
embedding=centroid.tolist(),
|
embedding=centroid.tolist(),
|
||||||
reference_count=len(embeddings),
|
reference_count=len(embeddings),
|
||||||
model_version=SIGLIP_VERSION,
|
model_version=model_version,
|
||||||
)
|
)
|
||||||
stmt = stmt.on_conflict_do_update(
|
stmt = stmt.on_conflict_do_update(
|
||||||
index_elements=["tag_id"],
|
index_elements=["tag_id"],
|
||||||
set_={
|
set_={
|
||||||
"embedding": centroid.tolist(),
|
"embedding": centroid.tolist(),
|
||||||
"reference_count": len(embeddings),
|
"reference_count": len(embeddings),
|
||||||
"model_version": SIGLIP_VERSION,
|
"model_version": model_version,
|
||||||
"updated_at": func.now(),
|
"updated_at": func.now(),
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
@@ -92,6 +107,7 @@ class CentroidService:
|
|||||||
"""Tag ids whose centroid is stale: member count != reference_count,
|
"""Tag ids whose centroid is stale: member count != reference_count,
|
||||||
OR no centroid row, OR centroid built on a different SigLIP version.
|
OR no centroid row, OR centroid built on a different SigLIP version.
|
||||||
Only considers eligible-kind tags with embeddings present."""
|
Only considers eligible-kind tags with embeddings present."""
|
||||||
|
current_model_version = await self._model_version()
|
||||||
member_counts = (
|
member_counts = (
|
||||||
select(
|
select(
|
||||||
image_tag.c.tag_id.label("tag_id"),
|
image_tag.c.tag_id.label("tag_id"),
|
||||||
@@ -116,7 +132,7 @@ class CentroidService:
|
|||||||
TagReferenceEmbedding.reference_count
|
TagReferenceEmbedding.reference_count
|
||||||
!= member_counts.c.members
|
!= member_counts.c.members
|
||||||
)
|
)
|
||||||
| (TagReferenceEmbedding.model_version != SIGLIP_VERSION)
|
| (TagReferenceEmbedding.model_version != current_model_version)
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
return list((await self.session.execute(stmt)).scalars().all())
|
return list((await self.session.execute(stmt)).scalars().all())
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user