pixiv retired, placement reconciler removed, build race fixed, mirror pinned #257
+175
-13
@@ -64,6 +64,43 @@ on:
|
|||||||
schedule:
|
schedule:
|
||||||
- cron: '0 6 * * 0'
|
- cron: '0 6 * * 0'
|
||||||
|
|
||||||
|
# One build.yml run per branch at a time (#4290).
|
||||||
|
#
|
||||||
|
# Without this, two pushes to one branch run in full parallel. Both read
|
||||||
|
# `fc.revision` off the channel tag before either has pushed, so both miss the
|
||||||
|
# reuse check and both build, and whichever finishes LAST owns the tag — so a
|
||||||
|
# slower older build can leave `:dev` carrying content older than the commit
|
||||||
|
# that moved it. Family rule 146 says a rolling channel refreshes itself; that
|
||||||
|
# is the case where it quietly does not.
|
||||||
|
#
|
||||||
|
# `cancel-in-progress: false` — QUEUE, never cancel. Cancelling could kill
|
||||||
|
# sign-extension mid-AMO-upload, leaving the version registered at AMO with no
|
||||||
|
# cached asset: exactly the stuck state the rollback trap in that job exists to
|
||||||
|
# prevent, reached through a different door. AMO will not release a burned
|
||||||
|
# version, so that state is unrecoverable rather than merely annoying. Waiting
|
||||||
|
# a few minutes is the cheaper end of that trade by a wide margin.
|
||||||
|
#
|
||||||
|
# Keyed on `github.ref`, so `dev` and `main` never block each other. Not on
|
||||||
|
# BUILD_REF: the group is evaluated before any job starts and cannot read the
|
||||||
|
# `env` context (the same restriction that makes a job-level `if:` unable to
|
||||||
|
# see it — see build-web's `outputs.candidate` note). The one consequence is
|
||||||
|
# that a scheduled refresh, whose ref is the default branch, shares dev's
|
||||||
|
# queue while publishing main's channel. It only ever waits, and it fires
|
||||||
|
# 06:00 Sunday precisely because nothing else is running then.
|
||||||
|
#
|
||||||
|
# UNVERIFIED AT THE TIME OF WRITING, and this file has been burned by exactly
|
||||||
|
# that before: the `format()` note below records `true == 'true'` evaluating
|
||||||
|
# FALSE on run 5270 with no symptom whatsoever — every lane green, the feature
|
||||||
|
# simply not happening. A `concurrency:` key this Gitea ignored would look
|
||||||
|
# identical: runs still overlapping, nothing red. So this is a belt, and the
|
||||||
|
# digest-pinned repoint in each build job is the braces — that one makes the
|
||||||
|
# :c-<sha> correctness property hold whether or not this key is honoured.
|
||||||
|
# Confirm by pushing twice in quick succession and reading the RUN LIST for a
|
||||||
|
# queued second run, never by reading this comment.
|
||||||
|
concurrency:
|
||||||
|
group: build-${{ github.ref }}
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
# Which branch a run BUILDS, as opposed to which one triggered it.
|
# Which branch a run BUILDS, as opposed to which one triggered it.
|
||||||
#
|
#
|
||||||
# They are the same thing on every trigger but `schedule`. Forgejo registers a
|
# They are the same thing on every trigger but `schedule`. Forgejo registers a
|
||||||
@@ -873,6 +910,10 @@ jobs:
|
|||||||
ls -la frontend/public/extension/
|
ls -la frontend/public/extension/
|
||||||
|
|
||||||
- name: Build and push web image
|
- name: Build and push web image
|
||||||
|
# `id:` so the repoint step below can read `outputs.digest` — the
|
||||||
|
# manifest THIS run published, as opposed to whatever the channel tag
|
||||||
|
# happens to name by the time that step runs (#4290).
|
||||||
|
id: build
|
||||||
if: steps.reuse.outputs.hit != 'true'
|
if: steps.reuse.outputs.hit != 'true'
|
||||||
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||||
# it normalises the image config's `created` field and the history
|
# it normalises the image config's `created` field and the history
|
||||||
@@ -1002,10 +1043,42 @@ jobs:
|
|||||||
- name: Write the remaining tags from the published image
|
- name: Write the remaining tags from the published image
|
||||||
env:
|
env:
|
||||||
IMAGE: git.fabledsword.com/bvandeusen/fabledcurator
|
IMAGE: git.fabledsword.com/bvandeusen/fabledcurator
|
||||||
SOURCE: ${{ steps.reuse.outputs.channel_ref }}
|
CHANNEL_REF: ${{ steps.reuse.outputs.channel_ref }}
|
||||||
|
# Empty when no build ran this job (a reuse hit, or the step's `if:`
|
||||||
|
# skipped it). Non-empty means THIS run pushed that manifest.
|
||||||
|
BUILT_DIGEST: ${{ steps.build.outputs.digest }}
|
||||||
TAGS: ${{ steps.tag.outputs.tags }}
|
TAGS: ${{ steps.tag.outputs.tags }}
|
||||||
run: |
|
run: |
|
||||||
set -euf
|
set -euf
|
||||||
|
# WHAT WE COPY FROM, which is not what we EXCLUDE (#4290).
|
||||||
|
#
|
||||||
|
# This step used to copy from the channel tag by NAME. Nothing
|
||||||
|
# serialises builds — there is no `concurrency:` key anywhere in
|
||||||
|
# .forgejo/workflows/ — so two pushes to one branch run in full
|
||||||
|
# parallel, both miss the reuse check, and both build. If the OLDER
|
||||||
|
# one finishes last it wins the channel tag; and then its repoint
|
||||||
|
# step, reading that tag by name, wrote :c-<sha> from whatever the
|
||||||
|
# other run had just published. An immutable rollback tag (rule 145)
|
||||||
|
# naming a different commit's bytes, wrong from birth — and
|
||||||
|
# immutability then guarantees nobody ever corrects it. Nothing goes
|
||||||
|
# red; it surfaces the day someone needs to roll back.
|
||||||
|
#
|
||||||
|
# So when this job built, copy from the DIGEST it pushed. Correct
|
||||||
|
# whatever a concurrent run does to the tag, and it does not depend
|
||||||
|
# on the runner honouring a `concurrency:` key — which this file has
|
||||||
|
# already been burned by once (the `format()` note at the top: an
|
||||||
|
# expression that evaluated false with no symptom at all).
|
||||||
|
#
|
||||||
|
# On a reuse hit there is no digest, and the channel tag is still the
|
||||||
|
# right source: "hit" MEANS that tag already carries this commit's
|
||||||
|
# fc.revision, which the reuse step verified by reading it.
|
||||||
|
if [ -n "${BUILT_DIGEST:-}" ]; then
|
||||||
|
SOURCE="$IMAGE@$BUILT_DIGEST"
|
||||||
|
echo "repoint: copying the digest this run published: $SOURCE"
|
||||||
|
else
|
||||||
|
SOURCE="$CHANNEL_REF"
|
||||||
|
echo "repoint: no build this run (reuse hit) — copying from $SOURCE"
|
||||||
|
fi
|
||||||
# The source tag is EXCLUDED from the targets, and that is load-
|
# The source tag is EXCLUDED from the targets, and that is load-
|
||||||
# bearing rather than an optimisation.
|
# bearing rather than an optimisation.
|
||||||
#
|
#
|
||||||
@@ -1031,18 +1104,25 @@ jobs:
|
|||||||
ARGS=""
|
ARGS=""
|
||||||
IFS=,
|
IFS=,
|
||||||
for t in $TAGS; do
|
for t in $TAGS; do
|
||||||
[ "$t" = "$SOURCE" ] && continue
|
# Keyed on CHANNEL_REF, never on SOURCE. SOURCE may now be a digest
|
||||||
|
# ref, which never equals a tag string — testing against it would
|
||||||
|
# stop excluding the channel tag, imagetools would index-wrap it,
|
||||||
|
# and `.Image.Config.Labels` would stop resolving through it. That
|
||||||
|
# kills the reuse label permanently (see the note just below).
|
||||||
|
[ "$t" = "$CHANNEL_REF" ] && continue
|
||||||
ARGS="$ARGS -t $t"
|
ARGS="$ARGS -t $t"
|
||||||
done
|
done
|
||||||
unset IFS
|
unset IFS
|
||||||
#
|
#
|
||||||
# This is also the whole of the scheduled refresh's tag handling
|
# This is also the whole of the scheduled refresh's tag handling
|
||||||
# (#3154): a refresh's tag list is the channel tag alone, so SOURCE
|
# (#3154): a refresh's tag list is the channel tag alone, so
|
||||||
# is the only entry, it gets excluded, and this step correctly does
|
# CHANNEL_REF is the only entry, it gets excluded, and this step
|
||||||
# nothing. No `if:` on the step and no schedule special-case —
|
# correctly does nothing. No `if:` on the step and no schedule
|
||||||
# excluding the source was already the right rule.
|
# special-case — excluding the channel tag was already the right
|
||||||
|
# rule. (A refresh builds to :refresh-candidate, so BUILT_DIGEST is
|
||||||
|
# set here and simply goes unused; `promote` moves :latest later.)
|
||||||
if [ -z "$ARGS" ]; then
|
if [ -z "$ARGS" ]; then
|
||||||
echo "repoint: $SOURCE is the only tag for this channel and"
|
echo "repoint: $CHANNEL_REF is the only tag for this channel and"
|
||||||
echo "repoint: already holds this revision — nothing to write."
|
echo "repoint: already holds this revision — nothing to write."
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
@@ -1591,6 +1671,10 @@ jobs:
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
- name: Build and push ml image
|
- name: Build and push ml image
|
||||||
|
# `id:` so the repoint step below can read `outputs.digest` — the
|
||||||
|
# manifest THIS run published, as opposed to whatever the channel tag
|
||||||
|
# happens to name by the time that step runs (#4290).
|
||||||
|
id: build
|
||||||
if: steps.reuse.outputs.hit != 'true'
|
if: steps.reuse.outputs.hit != 'true'
|
||||||
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||||
# it normalises the image config's `created` field and the history
|
# it normalises the image config's `created` field and the history
|
||||||
@@ -1714,10 +1798,42 @@ jobs:
|
|||||||
- name: Write the remaining tags from the published image
|
- name: Write the remaining tags from the published image
|
||||||
env:
|
env:
|
||||||
IMAGE: git.fabledsword.com/bvandeusen/fabledcurator-ml
|
IMAGE: git.fabledsword.com/bvandeusen/fabledcurator-ml
|
||||||
SOURCE: ${{ steps.reuse.outputs.channel_ref }}
|
CHANNEL_REF: ${{ steps.reuse.outputs.channel_ref }}
|
||||||
|
# Empty when no build ran this job (a reuse hit, or the step's `if:`
|
||||||
|
# skipped it). Non-empty means THIS run pushed that manifest.
|
||||||
|
BUILT_DIGEST: ${{ steps.build.outputs.digest }}
|
||||||
TAGS: ${{ steps.tag.outputs.tags }}
|
TAGS: ${{ steps.tag.outputs.tags }}
|
||||||
run: |
|
run: |
|
||||||
set -euf
|
set -euf
|
||||||
|
# WHAT WE COPY FROM, which is not what we EXCLUDE (#4290).
|
||||||
|
#
|
||||||
|
# This step used to copy from the channel tag by NAME. Nothing
|
||||||
|
# serialises builds — there is no `concurrency:` key anywhere in
|
||||||
|
# .forgejo/workflows/ — so two pushes to one branch run in full
|
||||||
|
# parallel, both miss the reuse check, and both build. If the OLDER
|
||||||
|
# one finishes last it wins the channel tag; and then its repoint
|
||||||
|
# step, reading that tag by name, wrote :c-<sha> from whatever the
|
||||||
|
# other run had just published. An immutable rollback tag (rule 145)
|
||||||
|
# naming a different commit's bytes, wrong from birth — and
|
||||||
|
# immutability then guarantees nobody ever corrects it. Nothing goes
|
||||||
|
# red; it surfaces the day someone needs to roll back.
|
||||||
|
#
|
||||||
|
# So when this job built, copy from the DIGEST it pushed. Correct
|
||||||
|
# whatever a concurrent run does to the tag, and it does not depend
|
||||||
|
# on the runner honouring a `concurrency:` key — which this file has
|
||||||
|
# already been burned by once (the `format()` note at the top: an
|
||||||
|
# expression that evaluated false with no symptom at all).
|
||||||
|
#
|
||||||
|
# On a reuse hit there is no digest, and the channel tag is still the
|
||||||
|
# right source: "hit" MEANS that tag already carries this commit's
|
||||||
|
# fc.revision, which the reuse step verified by reading it.
|
||||||
|
if [ -n "${BUILT_DIGEST:-}" ]; then
|
||||||
|
SOURCE="$IMAGE@$BUILT_DIGEST"
|
||||||
|
echo "repoint: copying the digest this run published: $SOURCE"
|
||||||
|
else
|
||||||
|
SOURCE="$CHANNEL_REF"
|
||||||
|
echo "repoint: no build this run (reuse hit) — copying from $SOURCE"
|
||||||
|
fi
|
||||||
# The source tag is EXCLUDED from the targets, and that is load-
|
# The source tag is EXCLUDED from the targets, and that is load-
|
||||||
# bearing rather than an optimisation.
|
# bearing rather than an optimisation.
|
||||||
#
|
#
|
||||||
@@ -1743,12 +1859,17 @@ jobs:
|
|||||||
ARGS=""
|
ARGS=""
|
||||||
IFS=,
|
IFS=,
|
||||||
for t in $TAGS; do
|
for t in $TAGS; do
|
||||||
[ "$t" = "$SOURCE" ] && continue
|
# Keyed on CHANNEL_REF, never on SOURCE. SOURCE may now be a digest
|
||||||
|
# ref, which never equals a tag string — testing against it would
|
||||||
|
# stop excluding the channel tag, imagetools would index-wrap it,
|
||||||
|
# and `.Image.Config.Labels` would stop resolving through it. That
|
||||||
|
# kills the reuse label permanently (see the note just below).
|
||||||
|
[ "$t" = "$CHANNEL_REF" ] && continue
|
||||||
ARGS="$ARGS -t $t"
|
ARGS="$ARGS -t $t"
|
||||||
done
|
done
|
||||||
unset IFS
|
unset IFS
|
||||||
if [ -z "$ARGS" ]; then
|
if [ -z "$ARGS" ]; then
|
||||||
echo "repoint: $SOURCE is the only tag for this channel and"
|
echo "repoint: $CHANNEL_REF is the only tag for this channel and"
|
||||||
echo "repoint: already holds this revision — nothing to write."
|
echo "repoint: already holds this revision — nothing to write."
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
@@ -2019,6 +2140,10 @@ jobs:
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
- name: Build and push agent image
|
- name: Build and push agent image
|
||||||
|
# `id:` so the repoint step below can read `outputs.digest` — the
|
||||||
|
# manifest THIS run published, as opposed to whatever the channel tag
|
||||||
|
# happens to name by the time that step runs (#4290).
|
||||||
|
id: build
|
||||||
if: steps.reuse.outputs.hit != 'true'
|
if: steps.reuse.outputs.hit != 'true'
|
||||||
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||||
# it normalises the image config's `created` field and the history
|
# it normalises the image config's `created` field and the history
|
||||||
@@ -2142,10 +2267,42 @@ jobs:
|
|||||||
- name: Write the remaining tags from the published image
|
- name: Write the remaining tags from the published image
|
||||||
env:
|
env:
|
||||||
IMAGE: git.fabledsword.com/bvandeusen/fabledcurator-agent
|
IMAGE: git.fabledsword.com/bvandeusen/fabledcurator-agent
|
||||||
SOURCE: ${{ steps.reuse.outputs.channel_ref }}
|
CHANNEL_REF: ${{ steps.reuse.outputs.channel_ref }}
|
||||||
|
# Empty when no build ran this job (a reuse hit, or the step's `if:`
|
||||||
|
# skipped it). Non-empty means THIS run pushed that manifest.
|
||||||
|
BUILT_DIGEST: ${{ steps.build.outputs.digest }}
|
||||||
TAGS: ${{ steps.tag.outputs.tags }}
|
TAGS: ${{ steps.tag.outputs.tags }}
|
||||||
run: |
|
run: |
|
||||||
set -euf
|
set -euf
|
||||||
|
# WHAT WE COPY FROM, which is not what we EXCLUDE (#4290).
|
||||||
|
#
|
||||||
|
# This step used to copy from the channel tag by NAME. Nothing
|
||||||
|
# serialises builds — there is no `concurrency:` key anywhere in
|
||||||
|
# .forgejo/workflows/ — so two pushes to one branch run in full
|
||||||
|
# parallel, both miss the reuse check, and both build. If the OLDER
|
||||||
|
# one finishes last it wins the channel tag; and then its repoint
|
||||||
|
# step, reading that tag by name, wrote :c-<sha> from whatever the
|
||||||
|
# other run had just published. An immutable rollback tag (rule 145)
|
||||||
|
# naming a different commit's bytes, wrong from birth — and
|
||||||
|
# immutability then guarantees nobody ever corrects it. Nothing goes
|
||||||
|
# red; it surfaces the day someone needs to roll back.
|
||||||
|
#
|
||||||
|
# So when this job built, copy from the DIGEST it pushed. Correct
|
||||||
|
# whatever a concurrent run does to the tag, and it does not depend
|
||||||
|
# on the runner honouring a `concurrency:` key — which this file has
|
||||||
|
# already been burned by once (the `format()` note at the top: an
|
||||||
|
# expression that evaluated false with no symptom at all).
|
||||||
|
#
|
||||||
|
# On a reuse hit there is no digest, and the channel tag is still the
|
||||||
|
# right source: "hit" MEANS that tag already carries this commit's
|
||||||
|
# fc.revision, which the reuse step verified by reading it.
|
||||||
|
if [ -n "${BUILT_DIGEST:-}" ]; then
|
||||||
|
SOURCE="$IMAGE@$BUILT_DIGEST"
|
||||||
|
echo "repoint: copying the digest this run published: $SOURCE"
|
||||||
|
else
|
||||||
|
SOURCE="$CHANNEL_REF"
|
||||||
|
echo "repoint: no build this run (reuse hit) — copying from $SOURCE"
|
||||||
|
fi
|
||||||
# The source tag is EXCLUDED from the targets, and that is load-
|
# The source tag is EXCLUDED from the targets, and that is load-
|
||||||
# bearing rather than an optimisation.
|
# bearing rather than an optimisation.
|
||||||
#
|
#
|
||||||
@@ -2171,12 +2328,17 @@ jobs:
|
|||||||
ARGS=""
|
ARGS=""
|
||||||
IFS=,
|
IFS=,
|
||||||
for t in $TAGS; do
|
for t in $TAGS; do
|
||||||
[ "$t" = "$SOURCE" ] && continue
|
# Keyed on CHANNEL_REF, never on SOURCE. SOURCE may now be a digest
|
||||||
|
# ref, which never equals a tag string — testing against it would
|
||||||
|
# stop excluding the channel tag, imagetools would index-wrap it,
|
||||||
|
# and `.Image.Config.Labels` would stop resolving through it. That
|
||||||
|
# kills the reuse label permanently (see the note just below).
|
||||||
|
[ "$t" = "$CHANNEL_REF" ] && continue
|
||||||
ARGS="$ARGS -t $t"
|
ARGS="$ARGS -t $t"
|
||||||
done
|
done
|
||||||
unset IFS
|
unset IFS
|
||||||
if [ -z "$ARGS" ]; then
|
if [ -z "$ARGS" ]; then
|
||||||
echo "repoint: $SOURCE is the only tag for this channel and"
|
echo "repoint: $CHANNEL_REF is the only tag for this channel and"
|
||||||
echo "repoint: already holds this revision — nothing to write."
|
echo "repoint: already holds this revision — nothing to write."
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
|
|||||||
@@ -0,0 +1,109 @@
|
|||||||
|
"""Drop library_placement_run — the placement reconciler is removed.
|
||||||
|
|
||||||
|
Milestone #421 built a sweep that compared each image's `artist_id` to the
|
||||||
|
name of the directory its file sat in, and called every mismatch a misplaced
|
||||||
|
file. On the operator's library that reported 33,789 of 63,605 images as
|
||||||
|
wrongly filed.
|
||||||
|
|
||||||
|
That number was an artefact of the comparison, not a fact about the library:
|
||||||
|
|
||||||
|
- **97.1%** of it was one artist's own folder spelled differently —
|
||||||
|
`Telepurte/` versus `telepurte/`. Same artist, same art, nothing wrong.
|
||||||
|
- Of the 1% that sat in a differently-named folder, querying `ImageProvenance`
|
||||||
|
— which records the post and source each file was actually downloaded from —
|
||||||
|
showed 87 where provenance agreed with the FOLDER and not the record, and 40
|
||||||
|
genuinely posted by several creators. The sweep would have misfiled or
|
||||||
|
arbitrarily picked for roughly 41% of that set.
|
||||||
|
|
||||||
|
The system already knows where every file came from. The reconciler inferred
|
||||||
|
it from a column and a directory name instead, and manufactured work out of a
|
||||||
|
naming convention. Operator's call, 2026-09-21: *"the current system
|
||||||
|
consistently records where items are and where they came from this is just
|
||||||
|
complicating something works and doesn't need fixing."*
|
||||||
|
|
||||||
|
Rule #22 — no legacy to preserve. The table goes with the code.
|
||||||
|
|
||||||
|
## What is deliberately kept
|
||||||
|
|
||||||
|
`utils.paths.canonical_subdir` stays: new filesystem imports derive their
|
||||||
|
directory from the artist's slug, matching what the downloader has always
|
||||||
|
done. It is not part of this tool and removing it would be churn for no fix.
|
||||||
|
Run 1's 327 moved files (`InsoUwu/` -> `insouwu/`) also stay where they are —
|
||||||
|
same artist either way, and the gallery renders them correctly.
|
||||||
|
|
||||||
|
Revision ID: 0100
|
||||||
|
Revises: 0099
|
||||||
|
Create Date: 2026-09-21
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from sqlalchemy.dialects import postgresql
|
||||||
|
|
||||||
|
revision: str = "0100"
|
||||||
|
down_revision: Union[str, None] = "0099"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.drop_index(
|
||||||
|
"ix_library_placement_run_artist_id",
|
||||||
|
table_name="library_placement_run",
|
||||||
|
)
|
||||||
|
op.drop_index(
|
||||||
|
"ix_library_placement_run_status", table_name="library_placement_run",
|
||||||
|
)
|
||||||
|
op.drop_table("library_placement_run")
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Recreates the table only. The three runs it held (one applied, two
|
||||||
|
# planned-and-never-run) are not restored and are not worth restoring —
|
||||||
|
# the code that reads them is gone.
|
||||||
|
op.create_table(
|
||||||
|
"library_placement_run",
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column(
|
||||||
|
"status", sa.String(length=16), server_default="running",
|
||||||
|
nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("artist_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"started_at", sa.DateTime(timezone=True),
|
||||||
|
server_default=sa.text("now()"), nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"planned_count", sa.Integer(), server_default="0", nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"moved_count", sa.Integer(), server_default="0", nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"refused_count", sa.Integer(), server_default="0", nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"moves", postgresql.JSONB(astext_type=sa.Text()),
|
||||||
|
server_default=sa.text("'[]'::jsonb"), nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"refusals", postgresql.JSONB(astext_type=sa.Text()),
|
||||||
|
server_default=sa.text("'[]'::jsonb"), nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("error", sa.Text(), nullable=True),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["artist_id"], ["artist.id"],
|
||||||
|
name="fk_library_placement_run_artist_id", ondelete="SET NULL",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id"),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_library_placement_run_status", "library_placement_run", ["status"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_library_placement_run_artist_id", "library_placement_run",
|
||||||
|
["artist_id"],
|
||||||
|
)
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
"""Clear failure state on sources that are disabled (#4279).
|
||||||
|
|
||||||
|
`failing_sources_clause()` now means "enabled AND erroring", so a disabled
|
||||||
|
source no longer counts as failing. That fixes what the surfaces REPORT; it
|
||||||
|
does not touch what the rows already CARRY, and the rows are the reason the
|
||||||
|
operator saw a banner for six days with no way to act on it (lesson #4202 —
|
||||||
|
a guard does not undo the value already stored).
|
||||||
|
|
||||||
|
## The row this exists for
|
||||||
|
|
||||||
|
Ebi77 (source 19): the membership sweep stopped it as `former_patron` at
|
||||||
|
02:50 on 2026-09-15 and correctly cleared its failure state. A deep scan was
|
||||||
|
armed twenty minutes later — `/backfill` had no `enabled` guard, which this
|
||||||
|
release also fixes — and could not complete without access, so the recovery
|
||||||
|
sweep stranded it:
|
||||||
|
|
||||||
|
consecutive_failures = 1
|
||||||
|
last_error = "stranded by recovery sweep (no terminal status after time_limit)"
|
||||||
|
|
||||||
|
Nothing could clear that. A disabled source is never scheduled, so no
|
||||||
|
successful run resets the counter; `SourceService.update` clears failure
|
||||||
|
state only on an explicit disable, and the source was already disabled; and
|
||||||
|
the card's Retry routes to `/check`, which refuses a disabled source.
|
||||||
|
|
||||||
|
## Why every disabled source, not just that one
|
||||||
|
|
||||||
|
The clear matches what `SourceService.update` already does when a source is
|
||||||
|
disabled through the app — "disable the subs you're not paying for without
|
||||||
|
them lingering as failing" — so this brings rows disabled by any OTHER path
|
||||||
|
(the membership sweep, a retired platform in 0097) into line with the rows
|
||||||
|
disabled by hand. Same shape as 0097: a repair migration reaches the live
|
||||||
|
instance on deploy rather than waiting for someone to find the row.
|
||||||
|
|
||||||
|
Enabled sources are untouched — a real failure on a live source must keep
|
||||||
|
showing.
|
||||||
|
|
||||||
|
Revision ID: 0101
|
||||||
|
Revises: 0100
|
||||||
|
Create Date: 2026-09-21
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0101"
|
||||||
|
down_revision: Union[str, None] = "0100"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.execute(
|
||||||
|
"UPDATE source SET last_error = NULL, error_type = NULL, "
|
||||||
|
"consecutive_failures = 0 "
|
||||||
|
"WHERE NOT enabled "
|
||||||
|
"AND (last_error IS NOT NULL OR error_type IS NOT NULL "
|
||||||
|
" OR consecutive_failures <> 0)"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Irreversible by design: the cleared strings and counts are not recorded
|
||||||
|
# anywhere, and restoring a failure state nobody can act on would only
|
||||||
|
# re-create the banner this removes. Rule #22 owes no story backwards.
|
||||||
|
pass
|
||||||
@@ -0,0 +1,138 @@
|
|||||||
|
"""Drop pixiv's ledgers, and delete credentials for platforms FC no longer has.
|
||||||
|
|
||||||
|
Milestone #406 step 6 (with issue #3980 folded in). Phase 1 unregistered pixiv
|
||||||
|
and the commit alongside this one deleted its client, downloader, ingester and
|
||||||
|
models. This removes the data those models described, and the stored secrets of
|
||||||
|
every platform that has been retired.
|
||||||
|
|
||||||
|
## The two ledger tables
|
||||||
|
|
||||||
|
`pixiv_seen_media` and `pixiv_failed_media` are the per-source seen / dead-letter
|
||||||
|
ledgers for a downloader that no longer exists. They were created in
|
||||||
|
`0089_baseline.py`, so dropping them needs a new revision rather than an edit
|
||||||
|
there.
|
||||||
|
|
||||||
|
## The credentials
|
||||||
|
|
||||||
|
Written as *delete every credential whose platform is not registered* rather
|
||||||
|
than as `platform = 'pixiv'`, at the explicit ask in this step's plan. That is
|
||||||
|
what makes one migration cover two retirements:
|
||||||
|
|
||||||
|
- **pixiv** — a live OAuth refresh token for a service FC no longer talks to.
|
||||||
|
- **deviantart** — issue #3980. #3069 retired DeviantArt in code on 2026-08-27
|
||||||
|
and left its stored session behind; seven weeks later it was still there.
|
||||||
|
|
||||||
|
And it is the only way either row can go. The credentials UI
|
||||||
|
(`subscriptions/SettingsTab.vue`) renders one card per platform returned by
|
||||||
|
`/api/platforms`, then looks the credential up by key — so a row whose platform
|
||||||
|
is unregistered has no card, no Remove button, and no way for the operator to
|
||||||
|
reach it. `CredentialService.list()` would return it; nothing asks.
|
||||||
|
|
||||||
|
The registered set is written out literally instead of importing
|
||||||
|
`known_platform_keys()`. A migration is a statement about one moment in the
|
||||||
|
schema's history: if it imported the live registry, retiring a fifth platform
|
||||||
|
in 2027 would silently change what this 2026 revision did on a fresh database.
|
||||||
|
The list below is the registry as of 2026-09-21.
|
||||||
|
|
||||||
|
## What is deliberately kept
|
||||||
|
|
||||||
|
**Every pixiv `Source` row.** The original plan deleted them; the operator's
|
||||||
|
call on 2026-09-21 was to keep them, and the reason is that `platform` is
|
||||||
|
stored ONLY on `Source` — neither `Post` nor `ImageRecord` carries it. Both
|
||||||
|
FKs are `ON DELETE SET NULL`, so a delete would not lose the art, but it would
|
||||||
|
drop every pixiv image into the gallery's `__unsourced__` bucket and strip the
|
||||||
|
platform chip off every pixiv post. The rows stay disabled (0097) and their
|
||||||
|
platform is unregistered, so nothing schedules them, nothing downloads through
|
||||||
|
them, and `POST /api/sources` will not make another. Keeping them costs
|
||||||
|
nothing and keeps the attribution the milestone's goal — *"the art already
|
||||||
|
downloaded from pixiv stays"* — is actually about.
|
||||||
|
|
||||||
|
Every pixiv `Post` and `ImageRecord` is likewise untouched.
|
||||||
|
|
||||||
|
Revision ID: 0102
|
||||||
|
Revises: 0101
|
||||||
|
Create Date: 2026-09-21
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0102"
|
||||||
|
down_revision: Union[str, None] = "0101"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
# services/platforms/__init__.py's PLATFORMS as of this revision. See the
|
||||||
|
# docstring for why this is a literal and not an import.
|
||||||
|
_REGISTERED_PLATFORMS = ("patreon", "subscribestar", "hentaifoundry", "discord")
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.execute(
|
||||||
|
sa.text(
|
||||||
|
"DELETE FROM credential WHERE platform NOT IN :registered"
|
||||||
|
).bindparams(
|
||||||
|
sa.bindparam("registered", value=_REGISTERED_PLATFORMS, expanding=True)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
op.drop_index("ix_pixiv_failed_media_source_id", table_name="pixiv_failed_media")
|
||||||
|
op.drop_table("pixiv_failed_media")
|
||||||
|
op.drop_index("ix_pixiv_seen_media_source_id", table_name="pixiv_seen_media")
|
||||||
|
op.drop_table("pixiv_seen_media")
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# The tables come back empty, and the credentials do not come back at all:
|
||||||
|
# they were encrypted blobs, this migration does not copy them anywhere,
|
||||||
|
# and restoring a live token for a platform FC cannot talk to would only
|
||||||
|
# re-create the liability. Rule #22 owes no story backwards.
|
||||||
|
op.create_table(
|
||||||
|
"pixiv_seen_media",
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("filehash", sa.String(length=128), nullable=False),
|
||||||
|
sa.Column("url", sa.Text(), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"created_at", sa.DateTime(timezone=True),
|
||||||
|
server_default=sa.text("now()"), nullable=False,
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["source_id"], ["source.id"],
|
||||||
|
name=op.f("fk_pixiv_seen_media_source_id_source"), ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_pixiv_seen_media")),
|
||||||
|
sa.UniqueConstraint(
|
||||||
|
"source_id", "filehash", name="uq_pixiv_seen_media_source_id",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
op.f("ix_pixiv_seen_media_source_id"), "pixiv_seen_media", ["source_id"],
|
||||||
|
unique=False,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"pixiv_failed_media",
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("filehash", sa.String(length=128), nullable=False),
|
||||||
|
sa.Column("url", sa.Text(), nullable=True),
|
||||||
|
sa.Column("error", sa.Text(), nullable=True),
|
||||||
|
sa.Column("attempts", sa.Integer(), server_default="1", nullable=False),
|
||||||
|
sa.Column(
|
||||||
|
"created_at", sa.DateTime(timezone=True),
|
||||||
|
server_default=sa.text("now()"), nullable=False,
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["source_id"], ["source.id"],
|
||||||
|
name=op.f("fk_pixiv_failed_media_source_id_source"), ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_pixiv_failed_media")),
|
||||||
|
sa.UniqueConstraint(
|
||||||
|
"source_id", "filehash", name="uq_pixiv_failed_media_source_id",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
op.f("ix_pixiv_failed_media_source_id"), "pixiv_failed_media", ["source_id"],
|
||||||
|
unique=False,
|
||||||
|
)
|
||||||
+2
-112
@@ -29,8 +29,8 @@ from quart import Blueprint, jsonify, request
|
|||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
|
|
||||||
from ..extensions import get_session
|
from ..extensions import get_session
|
||||||
from ..models import LibraryAuditRun, LibraryPlacementRun
|
from ..models import LibraryAuditRun
|
||||||
from ..services import cleanup_service, library_layout
|
from ..services import cleanup_service
|
||||||
from ._responses import error_response as _bad
|
from ._responses import error_response as _bad
|
||||||
|
|
||||||
cleanup_bp = Blueprint("cleanup", __name__, url_prefix="/api/cleanup")
|
cleanup_bp = Blueprint("cleanup", __name__, url_prefix="/api/cleanup")
|
||||||
@@ -196,113 +196,3 @@ async def audit_cancel(audit_id: int):
|
|||||||
)
|
)
|
||||||
await session.commit()
|
await session.commit()
|
||||||
return jsonify({"cancelled": True})
|
return jsonify({"cancelled": True})
|
||||||
|
|
||||||
|
|
||||||
@cleanup_bp.route("/layout", methods=["GET"])
|
|
||||||
async def layout_survey():
|
|
||||||
"""Milestone #421 blast radius: which ImageRecord rows sit outside their
|
|
||||||
artist's canonical slug directory, per artist.
|
|
||||||
|
|
||||||
Read-only. `?check_disk=1` additionally stats every destination to find
|
|
||||||
collisions with a file already there and sources that have gone missing —
|
|
||||||
the numbers the apply refuses on, at the cost of one stat per misplaced
|
|
||||||
row over NFS. It is OFF by default because a count-only pass answers "how
|
|
||||||
big is this" in seconds where the disk pass can run for minutes and time
|
|
||||||
the request out.
|
|
||||||
"""
|
|
||||||
check_disk = request.args.get("check_disk", "").lower() in ("1", "true", "yes")
|
|
||||||
async with get_session() as session:
|
|
||||||
report = await session.run_sync(
|
|
||||||
lambda s: library_layout.survey_layout(
|
|
||||||
s, IMAGES_ROOT, check_disk=check_disk,
|
|
||||||
)
|
|
||||||
)
|
|
||||||
return jsonify({**report.as_dict(), "checked_disk": check_disk})
|
|
||||||
|
|
||||||
|
|
||||||
def _serialize_placement_run(run: LibraryPlacementRun, *, moves: bool = False) -> dict:
|
|
||||||
"""`moves` is opt-in: an applied whole-library run carries tens of
|
|
||||||
thousands of entries, which is a fine thing to hold in Postgres and a
|
|
||||||
poor thing to put in every list response."""
|
|
||||||
out = {
|
|
||||||
"id": run.id,
|
|
||||||
"status": run.status,
|
|
||||||
"artist_id": run.artist_id,
|
|
||||||
"started_at": run.started_at.isoformat() if run.started_at else None,
|
|
||||||
"finished_at": run.finished_at.isoformat() if run.finished_at else None,
|
|
||||||
"planned_count": run.planned_count,
|
|
||||||
"moved_count": run.moved_count,
|
|
||||||
"refused_count": run.refused_count,
|
|
||||||
"refusals": run.refusals or [],
|
|
||||||
"error": run.error,
|
|
||||||
}
|
|
||||||
if moves:
|
|
||||||
out["moves"] = run.moves or []
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
@cleanup_bp.route("/placement/runs", methods=["GET"])
|
|
||||||
async def placement_runs():
|
|
||||||
"""Newest first. Without `moves`, so the list stays small."""
|
|
||||||
try:
|
|
||||||
limit = min(int(request.args.get("limit", "25")), 100)
|
|
||||||
except ValueError:
|
|
||||||
return _bad("invalid_limit")
|
|
||||||
async with get_session() as session:
|
|
||||||
rows = (await session.execute(
|
|
||||||
select(LibraryPlacementRun)
|
|
||||||
.order_by(LibraryPlacementRun.id.desc()).limit(limit)
|
|
||||||
)).scalars().all()
|
|
||||||
return jsonify({"runs": [_serialize_placement_run(r) for r in rows]})
|
|
||||||
|
|
||||||
|
|
||||||
@cleanup_bp.route("/placement/runs/<int:run_id>", methods=["GET"])
|
|
||||||
async def placement_run(run_id: int):
|
|
||||||
"""One run WITH its moves — this is the preview the operator reads before
|
|
||||||
agreeing, and the record of what happened afterwards."""
|
|
||||||
async with get_session() as session:
|
|
||||||
run = await session.get(LibraryPlacementRun, run_id)
|
|
||||||
if run is None:
|
|
||||||
return _bad("not_found", status=404)
|
|
||||||
return jsonify(_serialize_placement_run(run, moves=True))
|
|
||||||
|
|
||||||
|
|
||||||
@cleanup_bp.route("/placement/plan", methods=["POST"])
|
|
||||||
async def placement_plan():
|
|
||||||
"""Queue a planning run. `artist_id` scopes it to one artist, which is the
|
|
||||||
intended use: do one, look at the gallery, then continue or revert."""
|
|
||||||
body = await request.get_json(silent=True) or {}
|
|
||||||
artist_id = body.get("artist_id")
|
|
||||||
if artist_id is not None and not isinstance(artist_id, int):
|
|
||||||
return _bad("invalid_artist_id")
|
|
||||||
from ..tasks.library_placement import plan_placement
|
|
||||||
plan_placement.delay(artist_id)
|
|
||||||
return jsonify({"status": "dispatched"}), 202
|
|
||||||
|
|
||||||
|
|
||||||
@cleanup_bp.route("/placement/runs/<int:run_id>/apply", methods=["POST"])
|
|
||||||
async def placement_apply(run_id: int):
|
|
||||||
"""Execute a ready run. This renames files and rewrites rows."""
|
|
||||||
async with get_session() as session:
|
|
||||||
run = await session.get(LibraryPlacementRun, run_id)
|
|
||||||
if run is None:
|
|
||||||
return _bad("not_found", status=404)
|
|
||||||
if run.status != "ready":
|
|
||||||
return _bad("not_ready", detail=f"run is {run.status}")
|
|
||||||
from ..tasks.library_placement import apply_placement
|
|
||||||
apply_placement.delay(run_id)
|
|
||||||
return jsonify({"status": "dispatched"}), 202
|
|
||||||
|
|
||||||
|
|
||||||
@cleanup_bp.route("/placement/runs/<int:run_id>/revert", methods=["POST"])
|
|
||||||
async def placement_revert(run_id: int):
|
|
||||||
"""Put an applied run's files back. The reason the ledger is kept."""
|
|
||||||
async with get_session() as session:
|
|
||||||
run = await session.get(LibraryPlacementRun, run_id)
|
|
||||||
if run is None:
|
|
||||||
return _bad("not_found", status=404)
|
|
||||||
if run.status != "applied":
|
|
||||||
return _bad("not_applied", detail=f"run is {run.status}")
|
|
||||||
from ..tasks.library_placement import revert_placement
|
|
||||||
revert_placement.delay(run_id)
|
|
||||||
return jsonify({"status": "dispatched"}), 202
|
|
||||||
|
|||||||
@@ -109,8 +109,8 @@ async def quick_add_source():
|
|||||||
if not await _ext_key_required(session):
|
if not await _ext_key_required(session):
|
||||||
return _bad("unauthorized", status=401)
|
return _bad("unauthorized", status=401)
|
||||||
try:
|
try:
|
||||||
# crypto lets a pixiv add resolve the artist's display name via the
|
# crypto lets an add resolve the artist's display name via the
|
||||||
# stored OAuth token (else it falls back to the numeric id). #130.
|
# stored credential (else it falls back to the URL handle). #130.
|
||||||
result = await ExtensionService(session, _get_crypto()).quick_add_source(url)
|
result = await ExtensionService(session, _get_crypto()).quick_add_source(url)
|
||||||
except UnknownPlatformError as exc:
|
except UnknownPlatformError as exc:
|
||||||
return _bad(
|
return _bad(
|
||||||
|
|||||||
@@ -201,6 +201,16 @@ async def set_backfill(source_id: int):
|
|||||||
rec = await SourceService(session).get(source_id)
|
rec = await SourceService(session).get(source_id)
|
||||||
if rec is None:
|
if rec is None:
|
||||||
return _bad("not_found", status=404)
|
return _bad("not_found", status=404)
|
||||||
|
# A disabled source must not be armable for a deep walk — the same
|
||||||
|
# rule /check has carried all along (see `source_disabled` below).
|
||||||
|
# Arming one anyway is how #4279 happened: the membership sweep had
|
||||||
|
# stopped Ebi77 as `former_patron`, a deep scan was armed twenty
|
||||||
|
# minutes later, the walk could not complete without access, and
|
||||||
|
# the recovery sweep stranded it with a failure count no surface
|
||||||
|
# could clear — a disabled source is never scheduled again, and
|
||||||
|
# Retry routes to /check, which refuses it.
|
||||||
|
if not rec.enabled:
|
||||||
|
return _bad("source_disabled", detail="enable the source first")
|
||||||
native = uses_native_ingester(rec.platform)
|
native = uses_native_ingester(rec.platform)
|
||||||
if native:
|
if native:
|
||||||
cred = CredentialService(session, _get_crypto())
|
cred = CredentialService(session, _get_crypto())
|
||||||
|
|||||||
@@ -35,7 +35,6 @@ def make_celery() -> Celery:
|
|||||||
"backend.app.tasks.backup",
|
"backend.app.tasks.backup",
|
||||||
"backend.app.tasks.admin",
|
"backend.app.tasks.admin",
|
||||||
"backend.app.tasks.library_audit",
|
"backend.app.tasks.library_audit",
|
||||||
"backend.app.tasks.library_placement",
|
|
||||||
"backend.app.tasks.translation",
|
"backend.app.tasks.translation",
|
||||||
],
|
],
|
||||||
)
|
)
|
||||||
@@ -63,8 +62,6 @@ def make_celery() -> Celery:
|
|||||||
# 2026-06-07: a 2h audit blocked vacuum/backup/normalize for hours).
|
# 2026-06-07: a 2h audit blocked vacuum/backup/normalize for hours).
|
||||||
"backend.app.tasks.maintenance.*": {"queue": "maintenance"},
|
"backend.app.tasks.maintenance.*": {"queue": "maintenance"},
|
||||||
"backend.app.tasks.backup.*": {"queue": "maintenance_long"},
|
"backend.app.tasks.backup.*": {"queue": "maintenance_long"},
|
||||||
# 33k renames on NFS: long lane, same as backups.
|
|
||||||
"backend.app.tasks.library_placement.*": {"queue": "maintenance_long"},
|
|
||||||
"backend.app.tasks.admin.*": {"queue": "maintenance_long"},
|
"backend.app.tasks.admin.*": {"queue": "maintenance_long"},
|
||||||
"backend.app.tasks.library_audit.*": {"queue": "maintenance_long"},
|
"backend.app.tasks.library_audit.*": {"queue": "maintenance_long"},
|
||||||
# Translation backfill hits the LLM (~1–6s/item) → the long lane so it
|
# Translation backfill hits the LLM (~1–6s/item) → the long lane so it
|
||||||
|
|||||||
@@ -22,13 +22,10 @@ from .import_batch import ImportBatch
|
|||||||
from .import_settings import ImportSettings
|
from .import_settings import ImportSettings
|
||||||
from .import_task import ImportTask
|
from .import_task import ImportTask
|
||||||
from .library_audit_run import LibraryAuditRun
|
from .library_audit_run import LibraryAuditRun
|
||||||
from .library_placement_run import LibraryPlacementRun
|
|
||||||
from .membership_sync import MembershipSync
|
from .membership_sync import MembershipSync
|
||||||
from .ml_settings import MLSettings
|
from .ml_settings import MLSettings
|
||||||
from .patreon_failed_media import PatreonFailedMedia
|
from .patreon_failed_media import PatreonFailedMedia
|
||||||
from .patreon_seen_media import PatreonSeenMedia
|
from .patreon_seen_media import PatreonSeenMedia
|
||||||
from .pixiv_failed_media import PixivFailedMedia
|
|
||||||
from .pixiv_seen_media import PixivSeenMedia
|
|
||||||
from .platform_membership import PlatformMembership
|
from .platform_membership import PlatformMembership
|
||||||
from .post import Post
|
from .post import Post
|
||||||
from .post_association import PostAssociation
|
from .post_association import PostAssociation
|
||||||
@@ -59,8 +56,6 @@ __all__ = [
|
|||||||
"Credential",
|
"Credential",
|
||||||
"PatreonFailedMedia",
|
"PatreonFailedMedia",
|
||||||
"PatreonSeenMedia",
|
"PatreonSeenMedia",
|
||||||
"PixivFailedMedia",
|
|
||||||
"PixivSeenMedia",
|
|
||||||
"SubscribeStarFailedMedia",
|
"SubscribeStarFailedMedia",
|
||||||
"SubscribeStarSeenMedia",
|
"SubscribeStarSeenMedia",
|
||||||
"Post",
|
"Post",
|
||||||
@@ -86,7 +81,6 @@ __all__ = [
|
|||||||
"ImportTask",
|
"ImportTask",
|
||||||
"ImportSettings",
|
"ImportSettings",
|
||||||
"LibraryAuditRun",
|
"LibraryAuditRun",
|
||||||
"LibraryPlacementRun",
|
|
||||||
"MembershipSync",
|
"MembershipSync",
|
||||||
"MLSettings",
|
"MLSettings",
|
||||||
"HeadAutoApplyRun",
|
"HeadAutoApplyRun",
|
||||||
|
|||||||
@@ -1,99 +0,0 @@
|
|||||||
"""LibraryPlacementRun — one run of the placement reconciler (milestone #421).
|
|
||||||
|
|
||||||
The library is keyed on the Artist row's `slug`, one directory per artist.
|
|
||||||
Every writer agrees on that now (`utils.paths.canonical_subdir`, task #4244),
|
|
||||||
but ~33,789 rows were written under older rules and sit in some other
|
|
||||||
artist's directory. This row is a run of the sweep that trues them up.
|
|
||||||
|
|
||||||
State machine, mirroring LibraryAuditRun:
|
|
||||||
|
|
||||||
running -> ready -> applied -> reverted
|
|
||||||
\\-> cancelled
|
|
||||||
(any) -> error
|
|
||||||
|
|
||||||
## The `moves` column does three jobs
|
|
||||||
|
|
||||||
`moves` is the plan: `[{"image_id": 1, "from": "...", "to": "..."}, ...]`.
|
|
||||||
|
|
||||||
1. **Preview.** It is what the operator reads before agreeing.
|
|
||||||
2. **Apply.** The apply executes THIS list rather than re-deriving the set,
|
|
||||||
so the preview cannot describe a different set from the apply. That is
|
|
||||||
rule 93's guarantee reached the way LibraryAuditRun reaches it — the
|
|
||||||
plan is materialised, not recomputed.
|
|
||||||
3. **Revert.** `from` is retained, so a batch that looks wrong in the
|
|
||||||
gallery goes back where it came from.
|
|
||||||
|
|
||||||
## An applied run IS the undo ledger — it must never be pruned
|
|
||||||
|
|
||||||
This is the trap lesson #4226 names: a record that answers both "what is the
|
|
||||||
current plan" and "what happened" gets deleted by whatever forgets the first.
|
|
||||||
A `ready` run is disposable state. An `applied` run is HISTORY, and it is the
|
|
||||||
only record of where 33,789 files used to be — delete it and the moves become
|
|
||||||
irreversible.
|
|
||||||
|
|
||||||
No pruning exists for this table today, and that is deliberate. If retention
|
|
||||||
is ever added here, it may prune `ready`, `cancelled` and `error` runs; an
|
|
||||||
`applied` run is only safe to drop once someone decides the moves are settled
|
|
||||||
and undo is no longer wanted, which is an operator decision and not a
|
|
||||||
timer's.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from datetime import datetime
|
|
||||||
from typing import Any
|
|
||||||
|
|
||||||
from sqlalchemy import DateTime, ForeignKey, Integer, String, Text, func, text
|
|
||||||
from sqlalchemy.dialects.postgresql import JSONB
|
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
|
||||||
|
|
||||||
from .base import Base
|
|
||||||
|
|
||||||
|
|
||||||
class LibraryPlacementRun(Base):
|
|
||||||
__tablename__ = "library_placement_run"
|
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
|
||||||
status: Mapped[str] = mapped_column(
|
|
||||||
String(16), nullable=False, default="running", index=True,
|
|
||||||
server_default="running",
|
|
||||||
)
|
|
||||||
# running | ready | applied | reverted | cancelled | error
|
|
||||||
|
|
||||||
# Scope. NULL = the whole library; set = one artist, which is how this is
|
|
||||||
# meant to be used — do one artist, look at it in the gallery, continue or
|
|
||||||
# revert. ondelete SET NULL rather than CASCADE: deleting an artist must
|
|
||||||
# not destroy the record of where their files were moved.
|
|
||||||
artist_id: Mapped[int | None] = mapped_column(
|
|
||||||
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True,
|
|
||||||
)
|
|
||||||
|
|
||||||
started_at: Mapped[datetime] = mapped_column(
|
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now(),
|
|
||||||
)
|
|
||||||
finished_at: Mapped[datetime | None] = mapped_column(
|
|
||||||
DateTime(timezone=True), nullable=True,
|
|
||||||
)
|
|
||||||
|
|
||||||
planned_count: Mapped[int] = mapped_column(
|
|
||||||
Integer, nullable=False, default=0, server_default="0",
|
|
||||||
)
|
|
||||||
moved_count: Mapped[int] = mapped_column(
|
|
||||||
Integer, nullable=False, default=0, server_default="0",
|
|
||||||
)
|
|
||||||
refused_count: Mapped[int] = mapped_column(
|
|
||||||
Integer, nullable=False, default=0, server_default="0",
|
|
||||||
)
|
|
||||||
|
|
||||||
# [{"image_id": int, "from": str, "to": str}, ...] — see the module
|
|
||||||
# docstring. This is the plan, the audit trail and the undo, in that order
|
|
||||||
# of appearance and in one place.
|
|
||||||
moves: Mapped[list[dict[str, Any]]] = mapped_column(
|
|
||||||
JSONB, nullable=False, default=list, server_default=text("'[]'::jsonb"),
|
|
||||||
)
|
|
||||||
# [{"image_id": int, "reason": str}, ...] — rows the apply declined to
|
|
||||||
# touch, with why. A refusal is an expected outcome, not an error: the
|
|
||||||
# world moves between plan and apply, and every gate fails closed.
|
|
||||||
refusals: Mapped[list[dict[str, Any]]] = mapped_column(
|
|
||||||
JSONB, nullable=False, default=list, server_default=text("'[]'::jsonb"),
|
|
||||||
)
|
|
||||||
|
|
||||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
|
||||||
@@ -1,45 +0,0 @@
|
|||||||
"""PixivFailedMedia — per-source dead-letter ledger of Pixiv media that keeps
|
|
||||||
failing to download/validate.
|
|
||||||
|
|
||||||
Mirror of PatreonFailedMedia/SubscribeStarFailedMedia. Media that fails every
|
|
||||||
walk (404'd pximg URL, deleted work, persistently-corrupt bytes) would
|
|
||||||
otherwise re-error forever and re-burn backfill chunks. After ``attempts``
|
|
||||||
reaches the dead-letter threshold the ingester skips it on routine
|
|
||||||
tick/backfill walks (recovery still re-attempts). A later clean download
|
|
||||||
clears the row.
|
|
||||||
|
|
||||||
`filehash` is the same synthesized ``<illust_id>:p<num>`` /
|
|
||||||
``<illust_id>:ugoira`` key the seen-ledger uses. UNIQUE (source_id, filehash)
|
|
||||||
is the upsert key.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from datetime import datetime
|
|
||||||
|
|
||||||
from sqlalchemy import ForeignKey, Integer, String, Text, UniqueConstraint, func
|
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
|
||||||
from sqlalchemy.types import DateTime
|
|
||||||
|
|
||||||
from .base import Base
|
|
||||||
|
|
||||||
|
|
||||||
class PixivFailedMedia(Base):
|
|
||||||
__tablename__ = "pixiv_failed_media"
|
|
||||||
__table_args__ = (
|
|
||||||
UniqueConstraint(
|
|
||||||
"source_id", "filehash", name="uq_pixiv_failed_media_source_id"
|
|
||||||
),
|
|
||||||
)
|
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
|
||||||
source_id: Mapped[int] = mapped_column(
|
|
||||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
|
||||||
)
|
|
||||||
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
|
||||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1")
|
|
||||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
|
||||||
first_failed_at: Mapped[datetime] = mapped_column(
|
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
|
||||||
)
|
|
||||||
last_failed_at: Mapped[datetime] = mapped_column(
|
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
|
||||||
)
|
|
||||||
@@ -1,42 +0,0 @@
|
|||||||
"""PixivSeenMedia — per-source ledger of Pixiv media already
|
|
||||||
downloaded+processed.
|
|
||||||
|
|
||||||
Mirror of PatreonSeenMedia/SubscribeStarSeenMedia for the Pixiv native
|
|
||||||
ingester (replacing gallery-dl). One queryable row per (source, media) so
|
|
||||||
routine walks skip media we've already ingested; recovery mode bypasses the
|
|
||||||
ledger to re-walk.
|
|
||||||
|
|
||||||
Pixiv original URLs carry no content hash, so `filehash` is always the
|
|
||||||
synthesized ``<illust_id>:p<num>`` (page) / ``<illust_id>:ugoira`` (frame
|
|
||||||
zip) key — stable across any URL-shape drift. String(128) matches the sibling
|
|
||||||
ledgers.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from datetime import datetime
|
|
||||||
|
|
||||||
from sqlalchemy import ForeignKey, Integer, String, UniqueConstraint, func
|
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
|
||||||
from sqlalchemy.types import DateTime
|
|
||||||
|
|
||||||
from .base import Base
|
|
||||||
|
|
||||||
|
|
||||||
class PixivSeenMedia(Base):
|
|
||||||
__tablename__ = "pixiv_seen_media"
|
|
||||||
__table_args__ = (
|
|
||||||
# Dedup key the downloader upserts against: one ledger row per
|
|
||||||
# (source, media). A second sighting of the same media is a no-op.
|
|
||||||
UniqueConstraint(
|
|
||||||
"source_id", "filehash", name="uq_pixiv_seen_media_source_id"
|
|
||||||
),
|
|
||||||
)
|
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
|
||||||
source_id: Mapped[int] = mapped_column(
|
|
||||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
|
||||||
)
|
|
||||||
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
|
||||||
post_id: Mapped[str | None] = mapped_column(String(64), nullable=True)
|
|
||||||
seen_at: Mapped[datetime] = mapped_column(
|
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
|
||||||
)
|
|
||||||
@@ -16,7 +16,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
from collections.abc import Awaitable, Callable
|
from collections.abc import Awaitable, Callable
|
||||||
|
|
||||||
from sqlalchemy import Select
|
from sqlalchemy import Select, and_
|
||||||
from sqlalchemy.exc import IntegrityError
|
from sqlalchemy.exc import IntegrityError
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
@@ -67,12 +67,26 @@ async def get_or_create[T](
|
|||||||
|
|
||||||
|
|
||||||
def failing_sources_clause():
|
def failing_sources_clause():
|
||||||
"""A source is FAILING when its runs are actually erroring.
|
"""A source is FAILING when it is ENABLED and its runs are erroring.
|
||||||
|
|
||||||
Deliberately not `last_error IS NOT NULL` — a tier-limited source clears
|
Deliberately not `last_error IS NOT NULL` — a tier-limited source clears
|
||||||
last_error and keeps a chip, and must never be counted as broken.
|
last_error and keeps a chip, and must never be counted as broken.
|
||||||
|
|
||||||
|
The `enabled` half was folded in 2026-09-21 (#4279). A disabled source is
|
||||||
|
one FC deliberately stopped — most often because the membership sweep saw
|
||||||
|
`former_patron` — and "stopped because you no longer subscribe" is not
|
||||||
|
"failing". Worse, it is a failure nobody can clear: a disabled source is
|
||||||
|
never scheduled, so no successful run ever resets the counter, and the
|
||||||
|
card's Retry button routes to `/check`, which refuses a disabled source
|
||||||
|
outright. Ebi77 sat in the banner for six days with no action available.
|
||||||
|
|
||||||
|
This also settles a disagreement the two callers already had. The
|
||||||
|
scheduler's status count paired this clause with `enabled.is_(True)`;
|
||||||
|
`SourceService.list(failing=True)` did not. One counted Ebi77, the other
|
||||||
|
did not — the exact drift the note above this function warns about, which
|
||||||
|
is why the `enabled` test belongs IN the predicate rather than beside it.
|
||||||
"""
|
"""
|
||||||
return Source.consecutive_failures > 0
|
return and_(Source.enabled.is_(True), Source.consecutive_failures > 0)
|
||||||
|
|
||||||
|
|
||||||
def no_access_sources_clause():
|
def no_access_sources_clause():
|
||||||
|
|||||||
@@ -26,14 +26,12 @@ from pathlib import Path
|
|||||||
from .gallery_dl import DownloadResult, ErrorType
|
from .gallery_dl import DownloadResult, ErrorType
|
||||||
from .patreon_ingester import PatreonIngester
|
from .patreon_ingester import PatreonIngester
|
||||||
from .patreon_resolver import extract_vanity, resolve_campaign_id_for_source
|
from .patreon_resolver import extract_vanity, resolve_campaign_id_for_source
|
||||||
from .pixiv_client import user_id_from_url
|
|
||||||
from .pixiv_ingester import PixivIngester
|
|
||||||
from .platforms import known_platform_keys
|
from .platforms import known_platform_keys
|
||||||
from .subscribestar_ingester import SubscribeStarIngester
|
from .subscribestar_ingester import SubscribeStarIngester
|
||||||
|
|
||||||
# Platforms whose download + verify go through the native ingester rather than
|
# Platforms whose download + verify go through the native ingester rather than
|
||||||
# gallery-dl. gallery-dl still serves the rest (hentaifoundry, discord) until
|
# gallery-dl. gallery-dl still serves the rest (hentaifoundry, discord) until
|
||||||
# they migrate too. pixiv left this set when it was retired (milestone #406).
|
# they migrate too.
|
||||||
NATIVE_INGESTER_PLATFORMS = frozenset({"patreon", "subscribestar"})
|
NATIVE_INGESTER_PLATFORMS = frozenset({"patreon", "subscribestar"})
|
||||||
|
|
||||||
|
|
||||||
@@ -67,7 +65,6 @@ def _native_ingester_cls(platform: str):
|
|||||||
dispatch pick up the replacement."""
|
dispatch pick up the replacement."""
|
||||||
return {
|
return {
|
||||||
"patreon": PatreonIngester,
|
"patreon": PatreonIngester,
|
||||||
"pixiv": PixivIngester,
|
|
||||||
"subscribestar": SubscribeStarIngester,
|
"subscribestar": SubscribeStarIngester,
|
||||||
}[platform]
|
}[platform]
|
||||||
|
|
||||||
@@ -127,25 +124,17 @@ async def _resolve_native_campaign_id(
|
|||||||
platform: str, url: str, cookies_path: str | None, overrides: dict,
|
platform: str, url: str, cookies_path: str | None, overrides: dict,
|
||||||
) -> tuple[str | None, str | None]:
|
) -> tuple[str | None, str | None]:
|
||||||
"""`(campaign_id, resolved_campaign_id)` for a native source. SubscribeStar's
|
"""`(campaign_id, resolved_campaign_id)` for a native source. SubscribeStar's
|
||||||
feed id IS the creator URL; Pixiv's is the numeric user id parsed straight
|
feed id IS the creator URL (no lookup → resolved None). Patreon resolves the
|
||||||
from it (no lookup → resolved None either way). Patreon resolves the
|
|
||||||
campaign id from the vanity URL (resolved non-None when a lookup actually ran,
|
campaign id from the vanity URL (resolved non-None when a lookup actually ran,
|
||||||
so phase 3 caches it)."""
|
so phase 3 caches it)."""
|
||||||
if platform == "subscribestar":
|
if platform == "subscribestar":
|
||||||
return url, None
|
return url, None
|
||||||
if platform == "pixiv":
|
|
||||||
return user_id_from_url(url), None
|
|
||||||
return await resolve_campaign_id_for_source(url, cookies_path, overrides)
|
return await resolve_campaign_id_for_source(url, cookies_path, overrides)
|
||||||
|
|
||||||
|
|
||||||
def _campaign_resolution_error(platform: str, url: str) -> str:
|
def _campaign_resolution_error(platform: str, url: str) -> str:
|
||||||
"""Operator-facing message for a native source whose campaign id could not
|
"""Operator-facing message for a native source whose campaign id could not
|
||||||
be resolved — names the platform's own lookup mechanism."""
|
be resolved — names the platform's own lookup mechanism."""
|
||||||
if platform == "pixiv":
|
|
||||||
return (
|
|
||||||
f"Could not extract a pixiv user id. source_url={url!r} — expected "
|
|
||||||
"a URL like https://www.pixiv.net/users/<id>."
|
|
||||||
)
|
|
||||||
vanity = extract_vanity(url)
|
vanity = extract_vanity(url)
|
||||||
return (
|
return (
|
||||||
f"Could not resolve Patreon campaign id. source_url={url!r}; "
|
f"Could not resolve Patreon campaign id. source_url={url!r}; "
|
||||||
@@ -172,8 +161,8 @@ async def _run_native_ingester(
|
|||||||
platform, ctx["url"], ctx["cookies_path"], overrides
|
platform, ctx["url"], ctx["cookies_path"], overrides
|
||||||
)
|
)
|
||||||
if not campaign_id:
|
if not campaign_id:
|
||||||
# Patreon: vanity lookup failed. Pixiv: no numeric user id in the URL.
|
# Patreon: vanity lookup failed. (SubscribeStar's campaign id is the
|
||||||
# (SubscribeStar's campaign id is the URL itself — never lands here.)
|
# URL itself — never lands here.)
|
||||||
url = ctx["url"]
|
url = ctx["url"]
|
||||||
return (
|
return (
|
||||||
DownloadResult(
|
DownloadResult(
|
||||||
@@ -205,7 +194,7 @@ async def _run_native_ingester(
|
|||||||
validate=gdl._validate_files,
|
validate=gdl._validate_files,
|
||||||
rate_limit=rate_limit,
|
rate_limit=rate_limit,
|
||||||
request_sleep=request_sleep,
|
request_sleep=request_sleep,
|
||||||
# Uniform across adapters: token platforms (pixiv) authenticate with
|
# Uniform across adapters: a token platform would authenticate with
|
||||||
# it, cookie platforms accept-and-ignore — so this construction stays
|
# it, cookie platforms accept-and-ignore — so this construction stays
|
||||||
# platform-agnostic.
|
# platform-agnostic.
|
||||||
auth_token=ctx["auth_token"],
|
auth_token=ctx["auth_token"],
|
||||||
@@ -252,16 +241,11 @@ async def verify_source_credential(
|
|||||||
if uses_native_ingester(platform):
|
if uses_native_ingester(platform):
|
||||||
# Native ingester platforms verify via their own lightweight auth probe.
|
# Native ingester platforms verify via their own lightweight auth probe.
|
||||||
# SubscribeStar's probe takes the creator URL directly; Patreon's
|
# SubscribeStar's probe takes the creator URL directly; Patreon's
|
||||||
# resolves the campaign id first; Pixiv's is one OAuth refresh (the
|
# resolves the campaign id first.
|
||||||
# exact call that fails when the token is bad — no feed walk).
|
|
||||||
if platform == "subscribestar":
|
if platform == "subscribestar":
|
||||||
from .subscribestar_ingester import verify_subscribestar_credential
|
from .subscribestar_ingester import verify_subscribestar_credential
|
||||||
|
|
||||||
return await verify_subscribestar_credential(url, cookies_path, config_overrides)
|
return await verify_subscribestar_credential(url, cookies_path, config_overrides)
|
||||||
if platform == "pixiv":
|
|
||||||
from .pixiv_ingester import verify_pixiv_credential
|
|
||||||
|
|
||||||
return await verify_pixiv_credential(auth_token)
|
|
||||||
from .patreon_ingester import verify_patreon_credential
|
from .patreon_ingester import verify_patreon_credential
|
||||||
|
|
||||||
return await verify_patreon_credential(url, cookies_path, config_overrides)
|
return await verify_patreon_credential(url, cookies_path, config_overrides)
|
||||||
|
|||||||
@@ -61,8 +61,8 @@ _PLATFORM_PATTERNS: list[tuple[str, re.Pattern[str]]] = [
|
|||||||
class ExtensionService:
|
class ExtensionService:
|
||||||
def __init__(self, session: AsyncSession, crypto=None) -> None:
|
def __init__(self, session: AsyncSession, crypto=None) -> None:
|
||||||
self.session = session
|
self.session = session
|
||||||
# Optional decryptor for resolving a token-auth platform's display name
|
# Optional decryptor for resolving a platform's display name at
|
||||||
# (pixiv) at add-time. None → skip resolution, fall back to the handle.
|
# add-time. None → skip resolution, fall back to the handle.
|
||||||
self._crypto = crypto
|
self._crypto = crypto
|
||||||
|
|
||||||
async def quick_add_source(self, url: str) -> dict:
|
async def quick_add_source(self, url: str) -> dict:
|
||||||
@@ -112,12 +112,12 @@ class ExtensionService:
|
|||||||
self, platform: str, raw_slug: str, url: str
|
self, platform: str, raw_slug: str, url: str
|
||||||
) -> str:
|
) -> str:
|
||||||
"""The real display name for a new artist, resolved from the platform at
|
"""The real display name for a new artist, resolved from the platform at
|
||||||
add-time (#130). Our native platforms each have a name source — pixiv the
|
add-time (#130). Our native platforms each have a name source — patreon
|
||||||
app API (token), patreon the campaigns API, subscribestar the profile
|
the campaigns API, subscribestar the profile page (both cookies). Other
|
||||||
page (both cookies). Other platforms (and any failure — no credential,
|
platforms (and any failure — no credential, network error) fall back to
|
||||||
network error) fall back to the URL handle, which is already readable.
|
the URL handle, which is already readable.
|
||||||
The resolvers are sync, so they run in an executor."""
|
The resolvers are sync, so they run in an executor."""
|
||||||
if self._crypto is None or platform not in ("pixiv", "patreon", "subscribestar"):
|
if self._crypto is None or platform not in ("patreon", "subscribestar"):
|
||||||
return raw_slug
|
return raw_slug
|
||||||
import asyncio
|
import asyncio
|
||||||
|
|
||||||
@@ -125,15 +125,7 @@ class ExtensionService:
|
|||||||
cred = CredentialService(self.session, self._crypto)
|
cred = CredentialService(self.session, self._crypto)
|
||||||
loop = asyncio.get_running_loop()
|
loop = asyncio.get_running_loop()
|
||||||
try:
|
try:
|
||||||
if platform == "pixiv":
|
if platform == "patreon":
|
||||||
token = await cred.get_token("pixiv")
|
|
||||||
if not token:
|
|
||||||
return raw_slug
|
|
||||||
from .pixiv_client import PixivClient
|
|
||||||
name = await loop.run_in_executor(
|
|
||||||
None, PixivClient(token).resolve_display_name, raw_slug
|
|
||||||
)
|
|
||||||
elif platform == "patreon":
|
|
||||||
cookies = await cred.get_cookies_path("patreon")
|
cookies = await cred.get_cookies_path("patreon")
|
||||||
from .patreon_resolver import resolve_display_name
|
from .patreon_resolver import resolve_display_name
|
||||||
name = await loop.run_in_executor(
|
name = await loop.run_in_executor(
|
||||||
|
|||||||
@@ -45,7 +45,7 @@ from .tag_query import (
|
|||||||
# provenance (filesystem imports). Returned by facets() as a null-valued
|
# provenance (filesystem imports). Returned by facets() as a null-valued
|
||||||
# bucket; the frontend maps that null back to this sentinel in the URL so the
|
# bucket; the frontend maps that null back to this sentinel in the URL so the
|
||||||
# bucket is selectable. Underscore-wrapped so it can't collide with a real
|
# bucket is selectable. Underscore-wrapped so it can't collide with a real
|
||||||
# gallery-dl platform name (patreon/pixiv/...).
|
# gallery-dl platform name (patreon/hentaifoundry/...).
|
||||||
UNSOURCED_PLATFORM = "__unsourced__"
|
UNSOURCED_PLATFORM = "__unsourced__"
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -113,9 +113,9 @@ class Ingester:
|
|||||||
# (e.g. "Patreon API", "SubscribeStar markup").
|
# (e.g. "Patreon API", "SubscribeStar markup").
|
||||||
self._drift_label = drift_label or platform
|
self._drift_label = drift_label or platform
|
||||||
# #862 canary opt-out: platforms whose posts legitimately have empty
|
# #862 canary opt-out: platforms whose posts legitimately have empty
|
||||||
# bodies across large samples (pixiv — caption-less artists are common)
|
# bodies across large samples would false-positive the
|
||||||
# would false-positive the zero-bodies-means-drift alarm; their clients
|
# zero-bodies-means-drift alarm; their clients catch drift structurally
|
||||||
# catch drift structurally (response-shape checks) instead. The
|
# (response-shape checks) instead. The
|
||||||
# "bodies X/N" summary line still surfaces the ratio either way.
|
# "bodies X/N" summary line still surfaces the ratio either way.
|
||||||
self._body_canary = body_canary
|
self._body_canary = body_canary
|
||||||
|
|
||||||
|
|||||||
@@ -1,434 +0,0 @@
|
|||||||
"""Milestone #421: where an image file BELONGS, and which rows are not there.
|
|
||||||
|
|
||||||
The library is keyed on the Artist row's `slug` — one directory per artist.
|
|
||||||
It grew a second (and third, and fourth) home for many of them because
|
|
||||||
`Importer._copy_to_library` used to name the destination after the IMPORT
|
|
||||||
folder while the downloader wrote under the slug. That writer is fixed
|
|
||||||
(`utils.paths.canonical_subdir`, task #4244); this module is the other half —
|
|
||||||
finding the rows whose files are still in the old places, and saying where
|
|
||||||
each one goes.
|
|
||||||
|
|
||||||
## Preview and apply share these predicates, they do not re-derive them
|
|
||||||
|
|
||||||
`_misplaced_conditions` and `destination_for` are the whole decision. The
|
|
||||||
report (task #4245) and the move (task #4246) both spread them rather than
|
|
||||||
writing their own — the house shape for rule 93, snippet #3087. A preview
|
|
||||||
that computes its set differently from the apply is a preview that can lie,
|
|
||||||
and here the apply RENAMES the operator's art.
|
|
||||||
|
|
||||||
## What writes and what does not
|
|
||||||
|
|
||||||
`survey_layout` and `plan_placement` are read-only — they report and they
|
|
||||||
record a plan. `apply_run` and `revert_run` are the only functions here that
|
|
||||||
rename a file or rewrite a row, and each does both for one row at a time,
|
|
||||||
updating the row only after its rename lands.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from dataclasses import dataclass, field
|
|
||||||
from datetime import UTC, datetime
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from sqlalchemy import func, select
|
|
||||||
from sqlalchemy.orm import Session
|
|
||||||
|
|
||||||
from ..models import Artist, ImageRecord, LibraryPlacementRun
|
|
||||||
from ..utils.paths import canonical_subdir
|
|
||||||
|
|
||||||
# Top-level directories under the images root that are STORES, not artists.
|
|
||||||
# A sweep that treats these as misplaced artwork would relocate the
|
|
||||||
# thumbnail cache, the attachment blobs, or the credential key.
|
|
||||||
# thumbs/ sha-addressed thumbnail cache (NOT path-keyed — see below)
|
|
||||||
# attachments/ sha-addressed non-media blobs
|
|
||||||
# cookies/, secrets/ credential material
|
|
||||||
# _backups/, _quarantine/ backup artifacts; files pulled out of the library
|
|
||||||
RESERVED_TOP_LEVEL = frozenset({
|
|
||||||
"thumbs", "attachments", "cookies", "secrets", "_backups", "_quarantine",
|
|
||||||
})
|
|
||||||
|
|
||||||
|
|
||||||
def canonical_dir(images_root: Path, slug: str) -> Path:
|
|
||||||
"""The one directory an artist's files belong under."""
|
|
||||||
return images_root / slug
|
|
||||||
|
|
||||||
|
|
||||||
def _misplaced_conditions(images_root: Path, artist_id: int, slug: str) -> list:
|
|
||||||
"""Rows of `artist_id` whose file is NOT under that artist's canonical
|
|
||||||
directory. Spread into both halves — never restated.
|
|
||||||
|
|
||||||
The prefix carries a trailing separator on purpose: without it, artist
|
|
||||||
`ara` would match every path under `arbuzbudesh/`, and the sweep would
|
|
||||||
report one artist's whole library as correctly placed while quietly
|
|
||||||
skipping another's.
|
|
||||||
|
|
||||||
`startswith` compiles to LIKE, where `_` and `%` are wildcards, and this
|
|
||||||
does not escape them. That is safe ONLY because `utils.slug.slugify`
|
|
||||||
reduces a slug to `[a-z0-9-]` — neither character can reach the pattern.
|
|
||||||
Widen that charset and this needs `autoescape=True`, or `poch4n_art`
|
|
||||||
starts matching `poch4nXart` too.
|
|
||||||
"""
|
|
||||||
prefix = f"{canonical_dir(images_root, slug)}/"
|
|
||||||
return [
|
|
||||||
ImageRecord.artist_id == artist_id,
|
|
||||||
ImageRecord.path.is_not(None),
|
|
||||||
~ImageRecord.path.startswith(prefix),
|
|
||||||
]
|
|
||||||
|
|
||||||
|
|
||||||
def destination_for(path: str, images_root: Path, slug: str) -> Path | None:
|
|
||||||
"""Where `path`'s file belongs, or None when this row must not be moved.
|
|
||||||
|
|
||||||
None means: the path is outside the images root, or its top-level segment
|
|
||||||
is a reserved store. Both are refusals rather than errors — a row pointing
|
|
||||||
somewhere unexpected is exactly what should NOT be relocated automatically.
|
|
||||||
|
|
||||||
## The one place this diverges from `canonical_subdir`
|
|
||||||
|
|
||||||
A file sitting at the images ROOT with a known artist moves under that
|
|
||||||
artist's directory here, where `canonical_subdir` would leave it alone.
|
|
||||||
The two answer different questions. At import time an empty subdir means
|
|
||||||
no artist was resolved, so there is nothing to canonicalise against. Here
|
|
||||||
the row already CARRIES an artist_id, so a file at the root is an anomaly
|
|
||||||
with a known correct home — which is the whole point of the sweep.
|
|
||||||
|
|
||||||
(The 660 unattributed files at the root have no artist_id at all and are
|
|
||||||
not reachable from these predicates; task #4247 decides those.)
|
|
||||||
"""
|
|
||||||
p = Path(path)
|
|
||||||
try:
|
|
||||||
rel_dir = p.parent.relative_to(images_root)
|
|
||||||
except ValueError:
|
|
||||||
return None
|
|
||||||
parts = rel_dir.parts
|
|
||||||
if parts and parts[0] in RESERVED_TOP_LEVEL:
|
|
||||||
return None
|
|
||||||
sub = canonical_subdir(str(rel_dir) if str(rel_dir) != "." else "", slug)
|
|
||||||
if not sub:
|
|
||||||
# At the root, with an artist — see the docstring above.
|
|
||||||
return canonical_dir(images_root, slug) / p.name
|
|
||||||
return images_root / sub / p.name
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class ArtistLayout:
|
|
||||||
"""One artist's verdict."""
|
|
||||||
|
|
||||||
artist_id: int
|
|
||||||
name: str
|
|
||||||
slug: str
|
|
||||||
canonical_rows: int = 0
|
|
||||||
misplaced_rows: int = 0
|
|
||||||
# The non-canonical top-level directories this artist's files sit in —
|
|
||||||
# "Conto", "StickySpoodge", … This is what makes the report readable as
|
|
||||||
# the family list the disk survey found.
|
|
||||||
stray_dirs: list[str] = field(default_factory=list)
|
|
||||||
collisions: list[str] = field(default_factory=list)
|
|
||||||
missing_files: int = 0
|
|
||||||
unmovable: int = 0
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class LayoutReport:
|
|
||||||
artists: list[ArtistLayout] = field(default_factory=list)
|
|
||||||
total_rows: int = 0
|
|
||||||
canonical_rows: int = 0
|
|
||||||
misplaced_rows: int = 0
|
|
||||||
collision_count: int = 0
|
|
||||||
missing_files: int = 0
|
|
||||||
unmovable: int = 0
|
|
||||||
unattributed_rows: int = 0
|
|
||||||
|
|
||||||
def as_dict(self) -> dict:
|
|
||||||
return {
|
|
||||||
"total_rows": self.total_rows,
|
|
||||||
"canonical_rows": self.canonical_rows,
|
|
||||||
"misplaced_rows": self.misplaced_rows,
|
|
||||||
"collision_count": self.collision_count,
|
|
||||||
"missing_files": self.missing_files,
|
|
||||||
"unmovable": self.unmovable,
|
|
||||||
"unattributed_rows": self.unattributed_rows,
|
|
||||||
"artists": [
|
|
||||||
{
|
|
||||||
"artist_id": a.artist_id,
|
|
||||||
"name": a.name,
|
|
||||||
"slug": a.slug,
|
|
||||||
"canonical_rows": a.canonical_rows,
|
|
||||||
"misplaced_rows": a.misplaced_rows,
|
|
||||||
"stray_dirs": a.stray_dirs,
|
|
||||||
"collisions": a.collisions,
|
|
||||||
"missing_files": a.missing_files,
|
|
||||||
"unmovable": a.unmovable,
|
|
||||||
}
|
|
||||||
for a in self.artists
|
|
||||||
if a.misplaced_rows or a.collisions
|
|
||||||
],
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def survey_layout(
|
|
||||||
session: Session, images_root: Path, *, check_disk: bool = True,
|
|
||||||
) -> LayoutReport:
|
|
||||||
"""Read-only blast radius for the consolidation.
|
|
||||||
|
|
||||||
`check_disk` stats every destination to find rows that would collide with
|
|
||||||
a file already there, and sources that have already gone missing. It is
|
|
||||||
the honest number and it is what the apply will refuse on, but it costs
|
|
||||||
one stat per misplaced row over NFS — turn it off when you only want
|
|
||||||
counts.
|
|
||||||
"""
|
|
||||||
report = LayoutReport()
|
|
||||||
report.total_rows = session.execute(
|
|
||||||
select(func.count(ImageRecord.id))
|
|
||||||
).scalar_one()
|
|
||||||
report.unattributed_rows = session.execute(
|
|
||||||
select(func.count(ImageRecord.id)).where(ImageRecord.artist_id.is_(None))
|
|
||||||
).scalar_one()
|
|
||||||
|
|
||||||
artists = session.execute(
|
|
||||||
select(Artist).order_by(Artist.slug)
|
|
||||||
).scalars().all()
|
|
||||||
|
|
||||||
for artist in artists:
|
|
||||||
layout = ArtistLayout(
|
|
||||||
artist_id=artist.id, name=artist.name, slug=artist.slug,
|
|
||||||
)
|
|
||||||
conds = _misplaced_conditions(images_root, artist.id, artist.slug)
|
|
||||||
owned = session.execute(
|
|
||||||
select(func.count(ImageRecord.id))
|
|
||||||
.where(ImageRecord.artist_id == artist.id)
|
|
||||||
).scalar_one()
|
|
||||||
rows = session.execute(
|
|
||||||
select(ImageRecord.id, ImageRecord.path).where(*conds)
|
|
||||||
).all()
|
|
||||||
layout.misplaced_rows = len(rows)
|
|
||||||
layout.canonical_rows = owned - len(rows)
|
|
||||||
|
|
||||||
strays: set[str] = set()
|
|
||||||
destinations: dict[str, int] = {}
|
|
||||||
for row_id, path in rows:
|
|
||||||
dest = destination_for(path, images_root, artist.slug)
|
|
||||||
if dest is None:
|
|
||||||
layout.unmovable += 1
|
|
||||||
continue
|
|
||||||
try:
|
|
||||||
top = Path(path).parent.relative_to(images_root).parts
|
|
||||||
strays.add(top[0] if top else "<root>")
|
|
||||||
except ValueError:
|
|
||||||
strays.add("<outside>")
|
|
||||||
key = str(dest)
|
|
||||||
if key in destinations:
|
|
||||||
layout.collisions.append(key)
|
|
||||||
else:
|
|
||||||
destinations[key] = row_id
|
|
||||||
if check_disk:
|
|
||||||
if not Path(path).exists():
|
|
||||||
layout.missing_files += 1
|
|
||||||
elif dest.exists():
|
|
||||||
layout.collisions.append(key)
|
|
||||||
|
|
||||||
layout.stray_dirs = sorted(strays)
|
|
||||||
report.artists.append(layout)
|
|
||||||
report.canonical_rows += layout.canonical_rows
|
|
||||||
report.misplaced_rows += layout.misplaced_rows
|
|
||||||
report.collision_count += len(layout.collisions)
|
|
||||||
report.missing_files += layout.missing_files
|
|
||||||
report.unmovable += layout.unmovable
|
|
||||||
|
|
||||||
return report
|
|
||||||
|
|
||||||
|
|
||||||
# --- the reconciler: plan -> apply -> revert (#4246) -------------------------
|
|
||||||
#
|
|
||||||
# The three verbs share one materialised plan rather than each deriving its
|
|
||||||
# own set. `plan_placement` writes `LibraryPlacementRun.moves`; `apply_run`
|
|
||||||
# executes THAT list; `revert_run` walks it backwards. A preview that can
|
|
||||||
# disagree with its apply is the failure this shape exists to prevent, and
|
|
||||||
# here the apply renames the operator's art.
|
|
||||||
#
|
|
||||||
# Every step fails CLOSED. The world moves between planning and applying —
|
|
||||||
# a download lands, a supersede rewrites a path, a file is deleted — so the
|
|
||||||
# apply re-checks each row against what the plan recorded and declines the
|
|
||||||
# ones that moved on, instead of trusting a plan that may be minutes old.
|
|
||||||
|
|
||||||
|
|
||||||
def plan_placement(
|
|
||||||
session: Session, images_root: Path, *, artist_id: int | None = None,
|
|
||||||
) -> LibraryPlacementRun:
|
|
||||||
"""Build (and persist) the move plan. Touches no files.
|
|
||||||
|
|
||||||
`artist_id` scopes the run to one artist, which is how this is meant to be
|
|
||||||
used: do one, look at the gallery, then continue or revert. None plans the
|
|
||||||
whole library.
|
|
||||||
"""
|
|
||||||
stmt = select(Artist).order_by(Artist.slug)
|
|
||||||
if artist_id is not None:
|
|
||||||
stmt = stmt.where(Artist.id == artist_id)
|
|
||||||
artists = session.execute(stmt).scalars().all()
|
|
||||||
|
|
||||||
candidates: list[dict] = []
|
|
||||||
wanted: dict[str, int] = {}
|
|
||||||
for artist in artists:
|
|
||||||
rows = session.execute(
|
|
||||||
select(ImageRecord.id, ImageRecord.path)
|
|
||||||
.where(*_misplaced_conditions(images_root, artist.id, artist.slug))
|
|
||||||
).all()
|
|
||||||
for row_id, path in rows:
|
|
||||||
dest = destination_for(path, images_root, artist.slug)
|
|
||||||
if dest is None or dest.exists():
|
|
||||||
continue
|
|
||||||
key = str(dest)
|
|
||||||
candidates.append({"image_id": row_id, "from": path, "to": key})
|
|
||||||
wanted[key] = wanted.get(key, 0) + 1
|
|
||||||
|
|
||||||
# Two rows wanting one destination: plan NEITHER. Which of them "wins" is
|
|
||||||
# not this sweep's call, and planning one of them would silently pick a
|
|
||||||
# winner by iteration order. Counting first and filtering after is what
|
|
||||||
# makes that true — claiming as we go would quietly keep whichever came
|
|
||||||
# first.
|
|
||||||
moves = [m for m in candidates if wanted[m["to"]] == 1]
|
|
||||||
|
|
||||||
run = LibraryPlacementRun(
|
|
||||||
status="ready", artist_id=artist_id, moves=moves,
|
|
||||||
planned_count=len(moves),
|
|
||||||
)
|
|
||||||
session.add(run)
|
|
||||||
session.flush()
|
|
||||||
return run
|
|
||||||
|
|
||||||
|
|
||||||
def _move_one(src: Path, dest: Path) -> str | None:
|
|
||||||
"""Rename `src` to `dest`. Returns a refusal reason, or None on success.
|
|
||||||
|
|
||||||
A rename within one filesystem, so no copy and no free space needed. The
|
|
||||||
destination check is not a race-free guarantee — nothing here is — but it
|
|
||||||
turns the common case of "something already landed there" into a refusal
|
|
||||||
instead of a silent overwrite.
|
|
||||||
"""
|
|
||||||
if not src.exists():
|
|
||||||
return "source missing"
|
|
||||||
if dest.exists():
|
|
||||||
return "destination occupied"
|
|
||||||
try:
|
|
||||||
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
src.rename(dest)
|
|
||||||
except OSError as exc:
|
|
||||||
return f"rename failed: {exc}"
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
def apply_run(
|
|
||||||
session: Session, run: LibraryPlacementRun, *, chunk: int = 0,
|
|
||||||
) -> LibraryPlacementRun:
|
|
||||||
"""Execute a `ready` run's stored plan: file and row together, per row.
|
|
||||||
|
|
||||||
The row is updated ONLY after its rename lands, so a refused or failed
|
|
||||||
move can never leave `ImageRecord.path` pointing at a file that is not
|
|
||||||
there. Refusals are recorded and the run continues — one row that moved
|
|
||||||
on since planning is not a reason to abandon the other 33,788.
|
|
||||||
|
|
||||||
`chunk` commits progress every N moves. Set it for any real run: the
|
|
||||||
ledger is the ONLY record of where a file came from, so a worker that
|
|
||||||
dies at row 30,000 of 33,789 must not take the undo information for the
|
|
||||||
first 29,999 with it. Left at 0 (tests, small runs) everything persists
|
|
||||||
in one go at the end.
|
|
||||||
|
|
||||||
Re-running a partially-applied plan is safe rather than clever: the rows
|
|
||||||
already moved no longer match their `from`, so they refuse as "row moved
|
|
||||||
since planning" instead of being moved twice.
|
|
||||||
"""
|
|
||||||
if run.status != "ready":
|
|
||||||
raise ValueError(f"run {run.id} is {run.status}, not ready")
|
|
||||||
|
|
||||||
refusals: list[dict] = []
|
|
||||||
moved: list[dict] = []
|
|
||||||
|
|
||||||
def _persist() -> None:
|
|
||||||
# Reassign rather than mutate: SQLAlchemy does not track in-place
|
|
||||||
# changes to a JSONB list, so an .append() alone would never reach
|
|
||||||
# the database and the ledger would silently stay empty.
|
|
||||||
run.moves = list(moved)
|
|
||||||
run.refusals = list(refusals)
|
|
||||||
run.moved_count = len(moved)
|
|
||||||
run.refused_count = len(refusals)
|
|
||||||
session.commit()
|
|
||||||
|
|
||||||
# Snapshot the plan before iterating: `_persist` reassigns `run.moves`,
|
|
||||||
# and iterating the attribute while rewriting it would walk a list that
|
|
||||||
# changes underneath the loop.
|
|
||||||
plan = list(run.moves)
|
|
||||||
for done, move in enumerate(plan, start=1):
|
|
||||||
record = session.get(ImageRecord, move["image_id"])
|
|
||||||
if record is None:
|
|
||||||
refusals.append({"image_id": move["image_id"], "reason": "row gone"})
|
|
||||||
continue
|
|
||||||
if record.path != move["from"]:
|
|
||||||
# Something rewrote this row since the plan was built — a
|
|
||||||
# supersede, or an earlier run. The plan is stale for it.
|
|
||||||
refusals.append({
|
|
||||||
"image_id": move["image_id"], "reason": "row moved since planning",
|
|
||||||
})
|
|
||||||
continue
|
|
||||||
reason = _move_one(Path(move["from"]), Path(move["to"]))
|
|
||||||
if reason is not None:
|
|
||||||
refusals.append({"image_id": move["image_id"], "reason": reason})
|
|
||||||
continue
|
|
||||||
record.path = move["to"]
|
|
||||||
moved.append(move)
|
|
||||||
if chunk and done % chunk == 0:
|
|
||||||
_persist()
|
|
||||||
|
|
||||||
run.moves = moved
|
|
||||||
run.refusals = refusals
|
|
||||||
run.moved_count = len(moved)
|
|
||||||
run.refused_count = len(refusals)
|
|
||||||
run.status = "applied"
|
|
||||||
run.finished_at = datetime.now(UTC)
|
|
||||||
session.flush()
|
|
||||||
return run
|
|
||||||
|
|
||||||
|
|
||||||
def revert_run(
|
|
||||||
session: Session, run: LibraryPlacementRun,
|
|
||||||
) -> LibraryPlacementRun:
|
|
||||||
"""Put an applied run's files back where they came from.
|
|
||||||
|
|
||||||
This is why `from` is retained. It is the answer to "do one artist, look
|
|
||||||
at it, and undo if it reads wrong" — which is a cheaper way to settle
|
|
||||||
whether artist_id or the folder held the truth (#4257) than arguing it
|
|
||||||
from a sample.
|
|
||||||
|
|
||||||
Refuses the same way the apply does: a file someone has since moved or
|
|
||||||
replaced stays where it is, and its row is left alone.
|
|
||||||
|
|
||||||
A revert interrupted half way is resumable by re-running it: the rows
|
|
||||||
already put back no longer sit at `to`, so they refuse rather than move
|
|
||||||
twice. Unlike the apply this needs no chunked persistence — it consumes
|
|
||||||
the ledger rather than producing it, so a crash costs progress, not
|
|
||||||
information.
|
|
||||||
"""
|
|
||||||
if run.status != "applied":
|
|
||||||
raise ValueError(f"run {run.id} is {run.status}, not applied")
|
|
||||||
|
|
||||||
refusals: list[dict] = []
|
|
||||||
reverted = 0
|
|
||||||
for move in run.moves:
|
|
||||||
record = session.get(ImageRecord, move["image_id"])
|
|
||||||
if record is None or record.path != move["to"]:
|
|
||||||
refusals.append({
|
|
||||||
"image_id": move["image_id"], "reason": "row changed since apply",
|
|
||||||
})
|
|
||||||
continue
|
|
||||||
reason = _move_one(Path(move["to"]), Path(move["from"]))
|
|
||||||
if reason is not None:
|
|
||||||
refusals.append({"image_id": move["image_id"], "reason": reason})
|
|
||||||
continue
|
|
||||||
record.path = move["from"]
|
|
||||||
reverted += 1
|
|
||||||
|
|
||||||
run.refusals = refusals
|
|
||||||
run.refused_count = len(refusals)
|
|
||||||
run.moved_count = run.moved_count - reverted
|
|
||||||
run.status = "reverted"
|
|
||||||
run.finished_at = datetime.now(UTC)
|
|
||||||
session.flush()
|
|
||||||
return run
|
|
||||||
@@ -1,579 +0,0 @@
|
|||||||
"""Native Pixiv client — the Pixiv adapter's read path.
|
|
||||||
|
|
||||||
Pixiv has a real (if unofficial) API: the mobile app API gallery-dl drives
|
|
||||||
(`PixivAppAPI`). Per the downloader ground rule — gallery-dl is the
|
|
||||||
known-working base — this client mirrors gallery-dl 1.32.5's request profile
|
|
||||||
EXACTLY: the same iOS app headers on every request, the same OAuth
|
|
||||||
refresh-token dance against oauth.secure.pixiv.net (X-Client-Time +
|
|
||||||
X-Client-Hash), and the same `/v1/user/illusts` walk paginated by `next_url`.
|
|
||||||
Deviating from that profile is how the SubscribeStar/Patreon spikes broke, so
|
|
||||||
any change here should be diffed against gallery-dl's extractor first.
|
|
||||||
|
|
||||||
Feed shape (characterized from gallery-dl 1.32.5, extractor/pixiv.py):
|
|
||||||
- `GET /v1/user/illusts?user_id=<id>` returns `{"illusts": [work...],
|
|
||||||
"next_url": "https://app-api...?user_id=..&offset=30" | null}`.
|
|
||||||
- Pagination: re-issue the SAME endpoint with `next_url`'s query params. The
|
|
||||||
query string doubles as our resumable page cursor (re-fetching it re-serves
|
|
||||||
the same page — the ingest-core resume contract).
|
|
||||||
- A work carries id/title/type(illust|manga|ugoira)/caption(HTML)/
|
|
||||||
create_date(ISO+09:00)/tags[{name,translated_name}]/user/page_count/
|
|
||||||
x_restrict/series/total_view/total_bookmarks/meta_single_page/meta_pages.
|
|
||||||
- Files: multi-page → meta_pages[].image_urls.original; single page →
|
|
||||||
meta_single_page.original_image_url; ugoira → `/v1/ugoira/metadata` zip
|
|
||||||
(600x600 → 1920x1080 URL swap, gallery-dl's default non-original mode).
|
|
||||||
|
|
||||||
`campaign_id` for Pixiv is the numeric user id (extracted from the source URL
|
|
||||||
by `user_id_from_url` — no network resolver needed).
|
|
||||||
|
|
||||||
Gated works: pixiv serves a `https://s.pximg.net/common/images/limit_*.png`
|
|
||||||
placeholder as the "original" when a work is blocked for this account
|
|
||||||
(sanity-level filter, my-pixiv lock, deleted). gallery-dl's fallback for those
|
|
||||||
is a web-AJAX scrape that needs PHPSESSID browser cookies — FC stores only the
|
|
||||||
OAuth refresh token, so (exactly like our previous gallery-dl configuration,
|
|
||||||
which warned "No PHPSESSID cookie set") those works are skipped, via the
|
|
||||||
post_is_gated seam. Auth failures are loud (rotate the refresh token); a
|
|
||||||
response missing the fields we depend on is DRIFT (update this client).
|
|
||||||
|
|
||||||
FC runs on a plain-HTTP homelab; nothing here uses a secure-context Web API.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import hashlib
|
|
||||||
import logging
|
|
||||||
import time
|
|
||||||
from collections.abc import Iterator
|
|
||||||
from dataclasses import dataclass
|
|
||||||
from datetime import UTC, datetime
|
|
||||||
from urllib.parse import parse_qsl, urlsplit
|
|
||||||
|
|
||||||
import requests
|
|
||||||
|
|
||||||
from ..utils.paths import safe_ext
|
|
||||||
from .native_ingest_common import (
|
|
||||||
_MAX_429_RETRIES,
|
|
||||||
NativeAuthError,
|
|
||||||
NativeDriftError,
|
|
||||||
NativeIngestError,
|
|
||||||
make_session,
|
|
||||||
retry_after_seconds,
|
|
||||||
)
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
_TIMEOUT_SECONDS = 30.0
|
|
||||||
_API_ROOT = "https://app-api.pixiv.net"
|
|
||||||
_OAUTH_URL = "https://oauth.secure.pixiv.net/auth/token"
|
|
||||||
|
|
||||||
# gallery-dl's public Pixiv-app credentials (PixivAppAPI, also pixivpy's) —
|
|
||||||
# these identify the official iOS app to the API, NOT the operator; the
|
|
||||||
# operator's identity is the OAuth refresh token.
|
|
||||||
_CLIENT_ID = "MOBrBDS8blbauoSck0ZfDbtuzpyT"
|
|
||||||
_CLIENT_SECRET = "lsACyCD94FhDUtGTXi3QzcFE2uU1hqtDaKeqrdwj"
|
|
||||||
_HASH_SECRET = (
|
|
||||||
"28c1fdd170a5204386cb1313c7077b34"
|
|
||||||
"f83e4aaf4aa829ce78c231e05b0bae2c"
|
|
||||||
)
|
|
||||||
|
|
||||||
# The exact header set gallery-dl 1.32.5 installs on its session — the proven
|
|
||||||
# app-API request profile. The Referer also unlocks i.pximg.net media GETs
|
|
||||||
# (403 without it), so the downloader reuses this constant.
|
|
||||||
PIXIV_APP_HEADERS = {
|
|
||||||
"App-OS": "ios",
|
|
||||||
"App-OS-Version": "16.7.2",
|
|
||||||
"App-Version": "7.19.1",
|
|
||||||
"User-Agent": "PixivIOSApp/7.19.1 (iOS 16.7.2; iPhone12,8)",
|
|
||||||
"Referer": "https://app-api.pixiv.net/",
|
|
||||||
}
|
|
||||||
|
|
||||||
# Placeholder image prefix pixiv serves instead of a blocked work's original
|
|
||||||
# (limit_sanity_level / limit_mypixiv / limit_unknown variants).
|
|
||||||
_LIMIT_URL = "https://s.pximg.net/common/images/limit_"
|
|
||||||
|
|
||||||
# The app API reports rate-limiting as an error MESSAGE (often on HTTP 403),
|
|
||||||
# not only as HTTP 429. gallery-dl sleeps 300s in-walk; sleeping that long
|
|
||||||
# inside our time-boxed chunk would eat the whole budget, so we surface it as
|
|
||||||
# a typed 429 and let download_service's cooldown machinery honor the wait.
|
|
||||||
_RATE_LIMIT_RETRY_AFTER = 300.0
|
|
||||||
|
|
||||||
_TITLE_MAX = 50 # gallery-dl pixiv filename template: {title[:50]}
|
|
||||||
|
|
||||||
_RATINGS = {0: "General", 1: "R-18", 2: "R-18G"}
|
|
||||||
|
|
||||||
|
|
||||||
class PixivAPIError(NativeIngestError):
|
|
||||||
"""Base for native Pixiv client failures. status_code / retry_after are
|
|
||||||
inherited from NativeIngestError."""
|
|
||||||
|
|
||||||
|
|
||||||
class PixivAuthError(PixivAPIError, NativeAuthError):
|
|
||||||
"""Auth failure — missing/expired/revoked OAuth refresh token. Fix =
|
|
||||||
rotate the credential (Settings → Credentials → Pixiv), not update the
|
|
||||||
client. Maps to error_type 'auth_error'."""
|
|
||||||
|
|
||||||
|
|
||||||
class PixivDriftError(PixivAPIError, NativeDriftError):
|
|
||||||
"""A response did not match the shape this client depends on (missing
|
|
||||||
`illusts`, un-parseable JSON where JSON was promised). Fail loud so the
|
|
||||||
run flags 'the Pixiv app API changed' instead of silently importing
|
|
||||||
nothing. Maps to API_DRIFT."""
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class MediaItem:
|
|
||||||
"""One resolved downloadable file belonging to a Pixiv work.
|
|
||||||
|
|
||||||
Fields mirror the other native clients' MediaItem so the downloader and
|
|
||||||
ledger are structurally the same. Pixiv original URLs carry no content
|
|
||||||
hash, so `filehash` is always None and the ledger keys on
|
|
||||||
`<post_id>:<media_id>` where media_id is `p<num>` (page) or `ugoira`
|
|
||||||
(the frame zip) — stable across URL-shape drift.
|
|
||||||
"""
|
|
||||||
|
|
||||||
url: str
|
|
||||||
filename: str
|
|
||||||
kind: str
|
|
||||||
filehash: str | None
|
|
||||||
post_id: str
|
|
||||||
media_id: str
|
|
||||||
|
|
||||||
|
|
||||||
def user_id_from_url(url: str) -> str | None:
|
|
||||||
"""The numeric pixiv user id from a source URL, or None.
|
|
||||||
|
|
||||||
Handles the modern forms FC accepts as sources
|
|
||||||
(https://www.pixiv.net/users/<id>, /en/users/<id>) plus the legacy
|
|
||||||
member.php?id=<id>. This IS the campaign id — no network resolver.
|
|
||||||
"""
|
|
||||||
parts = urlsplit(url or "")
|
|
||||||
if "pixiv.net" not in parts.netloc:
|
|
||||||
return None
|
|
||||||
segs = [s for s in parts.path.split("/") if s]
|
|
||||||
if segs and segs[0] == "en":
|
|
||||||
segs = segs[1:]
|
|
||||||
if len(segs) >= 2 and segs[0] == "users" and segs[1].isdigit():
|
|
||||||
return segs[1]
|
|
||||||
if segs and segs[0] == "member.php":
|
|
||||||
qid = dict(parse_qsl(parts.query)).get("id", "")
|
|
||||||
if qid.isdigit():
|
|
||||||
return qid
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
def _work_filename(work: dict, num: int, url: str) -> str:
|
|
||||||
"""gallery-dl layout parity: `{id}_{title[:50]}_{num:>02}.{extension}`
|
|
||||||
(the downloader sanitizes the final segment)."""
|
|
||||||
title = work.get("title")
|
|
||||||
title50 = (title if isinstance(title, str) else "")[:_TITLE_MAX]
|
|
||||||
ext = safe_ext(urlsplit(url).path.rsplit("/", 1)[-1])
|
|
||||||
return f"{work.get('id')}_{title50}_{num:02d}{ext}"
|
|
||||||
|
|
||||||
|
|
||||||
class PixivClient:
|
|
||||||
"""Synchronous Pixiv app-API read client. Construct with the operator's
|
|
||||||
OAuth refresh token (the same token-type Credential the gallery-dl path
|
|
||||||
consumed as `extractor.pixiv.refresh-token`)."""
|
|
||||||
|
|
||||||
def __init__(
|
|
||||||
self,
|
|
||||||
refresh_token: str | None,
|
|
||||||
*,
|
|
||||||
request_sleep: float = 0.0,
|
|
||||||
max_retries: int = _MAX_429_RETRIES,
|
|
||||||
session: requests.Session | None = None,
|
|
||||||
):
|
|
||||||
self.refresh_token = refresh_token
|
|
||||||
self._request_sleep = request_sleep or 0.0
|
|
||||||
self._max_retries = max_retries
|
|
||||||
# No cookies — the app API authenticates via the Bearer token _login
|
|
||||||
# installs. make_session still supplies the retry/UA plumbing; the
|
|
||||||
# extra_headers overwrite its browser UA with the app profile.
|
|
||||||
self._session = (
|
|
||||||
session if session is not None
|
|
||||||
else make_session(None, extra_headers=PIXIV_APP_HEADERS)
|
|
||||||
)
|
|
||||||
self._authed_user: dict = {}
|
|
||||||
# Monotonic deadline after which the access token must be refreshed;
|
|
||||||
# 0 forces a refresh on first use.
|
|
||||||
self._token_deadline = 0.0
|
|
||||||
|
|
||||||
# -- auth ----------------------------------------------------------------
|
|
||||||
|
|
||||||
def _login(self) -> None:
|
|
||||||
"""Exchange the refresh token for a Bearer access token (gallery-dl's
|
|
||||||
`_login_impl`, including the X-Client-Time/X-Client-Hash pair the
|
|
||||||
endpoint validates). No-op while the current token is still fresh."""
|
|
||||||
if time.monotonic() < self._token_deadline:
|
|
||||||
return
|
|
||||||
if not self.refresh_token:
|
|
||||||
raise PixivAuthError(
|
|
||||||
"No Pixiv refresh token configured — add the OAuth refresh "
|
|
||||||
"token as the Pixiv credential (token type)."
|
|
||||||
)
|
|
||||||
# gallery-dl stamps naive-UTC with a literal +00:00 suffix.
|
|
||||||
now = datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%S+00:00")
|
|
||||||
headers = {
|
|
||||||
"X-Client-Time": now,
|
|
||||||
"X-Client-Hash": hashlib.md5(
|
|
||||||
(now + _HASH_SECRET).encode()
|
|
||||||
).hexdigest(),
|
|
||||||
}
|
|
||||||
data = {
|
|
||||||
"client_id": _CLIENT_ID,
|
|
||||||
"client_secret": _CLIENT_SECRET,
|
|
||||||
"grant_type": "refresh_token",
|
|
||||||
"refresh_token": self.refresh_token,
|
|
||||||
"get_secure_url": "1",
|
|
||||||
}
|
|
||||||
try:
|
|
||||||
resp = self._session.post(
|
|
||||||
_OAUTH_URL, data=data, headers=headers,
|
|
||||||
timeout=_TIMEOUT_SECONDS,
|
|
||||||
)
|
|
||||||
except requests.RequestException as exc:
|
|
||||||
raise PixivAPIError(f"Pixiv OAuth request failed: {exc}") from exc
|
|
||||||
if resp.status_code >= 400:
|
|
||||||
raise PixivAuthError(
|
|
||||||
"Pixiv rejected the refresh token (HTTP "
|
|
||||||
f"{resp.status_code}) — rotate the Pixiv credential.",
|
|
||||||
status_code=resp.status_code,
|
|
||||||
)
|
|
||||||
try:
|
|
||||||
payload = resp.json()["response"]
|
|
||||||
access = payload["access_token"]
|
|
||||||
except (ValueError, KeyError, TypeError) as exc:
|
|
||||||
raise PixivDriftError(
|
|
||||||
f"Pixiv OAuth response shape changed: {exc}"
|
|
||||||
) from exc
|
|
||||||
self._authed_user = payload.get("user") or {}
|
|
||||||
self._session.headers["Authorization"] = f"Bearer {access}"
|
|
||||||
# expires_in is 3600 today; refresh 60s early so a long walk never
|
|
||||||
# rides an expiring token into a spurious 400.
|
|
||||||
expires_in = payload.get("expires_in")
|
|
||||||
lifetime = float(expires_in) if isinstance(expires_in, (int, float)) else 3600.0
|
|
||||||
self._token_deadline = time.monotonic() + max(60.0, lifetime - 60.0)
|
|
||||||
|
|
||||||
# -- request -------------------------------------------------------------
|
|
||||||
|
|
||||||
def _call(self, endpoint: str, params: dict) -> dict:
|
|
||||||
"""Authenticated app-API GET → parsed JSON body, with the shared 429
|
|
||||||
backoff and the loud auth/drift/rate-limit mapping."""
|
|
||||||
self._login()
|
|
||||||
if self._request_sleep > 0:
|
|
||||||
time.sleep(self._request_sleep)
|
|
||||||
url = _API_ROOT + endpoint
|
|
||||||
attempt = 0
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
resp = self._session.get(
|
|
||||||
url, params=params, timeout=_TIMEOUT_SECONDS
|
|
||||||
)
|
|
||||||
except requests.RequestException as exc:
|
|
||||||
raise PixivAPIError(
|
|
||||||
f"Pixiv request failed ({endpoint}): {exc}"
|
|
||||||
) from exc
|
|
||||||
if resp.status_code == 429 and attempt < self._max_retries:
|
|
||||||
attempt += 1
|
|
||||||
delay = retry_after_seconds(resp, attempt)
|
|
||||||
log.warning(
|
|
||||||
"Pixiv 429 (%s) — backing off %.1fs (retry %d/%d)",
|
|
||||||
endpoint, delay, attempt, self._max_retries,
|
|
||||||
)
|
|
||||||
time.sleep(delay)
|
|
||||||
continue
|
|
||||||
break
|
|
||||||
|
|
||||||
try:
|
|
||||||
body = resp.json()
|
|
||||||
except ValueError as exc:
|
|
||||||
raise PixivDriftError(
|
|
||||||
f"Pixiv returned non-JSON for {endpoint} "
|
|
||||||
f"(HTTP {resp.status_code})"
|
|
||||||
) from exc
|
|
||||||
|
|
||||||
error = body.get("error") if isinstance(body, dict) else None
|
|
||||||
message = ""
|
|
||||||
if isinstance(error, dict):
|
|
||||||
message = str(
|
|
||||||
error.get("user_message") or error.get("message") or ""
|
|
||||||
)
|
|
||||||
# Rate limiting first: the app API reports it as an error MESSAGE
|
|
||||||
# (often on HTTP 403), which must not be mistaken for an auth failure.
|
|
||||||
if resp.status_code == 429 or "rate limit" in message.lower():
|
|
||||||
raise PixivAPIError(
|
|
||||||
f"Pixiv rate limit hit ({endpoint}): {message or 'HTTP 429'}",
|
|
||||||
status_code=429,
|
|
||||||
retry_after=_RATE_LIMIT_RETRY_AFTER,
|
|
||||||
)
|
|
||||||
if resp.status_code in (400, 401, 403):
|
|
||||||
# Invalid/expired access token surfaces as 400 invalid_grant-style
|
|
||||||
# errors on the app API; 401/403 are straight auth rejections.
|
|
||||||
raise PixivAuthError(
|
|
||||||
f"Pixiv rejected the request ({endpoint}, HTTP "
|
|
||||||
f"{resp.status_code}): {message or 'auth rejected'} — "
|
|
||||||
"rotate the Pixiv refresh token.",
|
|
||||||
status_code=resp.status_code,
|
|
||||||
)
|
|
||||||
if resp.status_code >= 400:
|
|
||||||
raise PixivAPIError(
|
|
||||||
f"Pixiv API error ({endpoint}, HTTP {resp.status_code}): "
|
|
||||||
f"{message or 'unknown error'}",
|
|
||||||
status_code=resp.status_code,
|
|
||||||
)
|
|
||||||
if error:
|
|
||||||
# HTTP 200 carrying an error object — unexpected, but never
|
|
||||||
# silently treat it as data.
|
|
||||||
raise PixivAPIError(
|
|
||||||
f"Pixiv API error ({endpoint}): {message or error}",
|
|
||||||
status_code=resp.status_code,
|
|
||||||
)
|
|
||||||
return body
|
|
||||||
|
|
||||||
# -- normalization -------------------------------------------------------
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def _normalize(work: dict) -> dict:
|
|
||||||
"""Wrap an app-API work in the `{"id", "attributes", ...}` post shape
|
|
||||||
the platform-agnostic core and shared helpers read. The raw work rides
|
|
||||||
along under `_work` for extract_media / the post record."""
|
|
||||||
title = work.get("title")
|
|
||||||
caption = work.get("caption")
|
|
||||||
wtype = work.get("type")
|
|
||||||
return {
|
|
||||||
"id": work.get("id"),
|
|
||||||
"attributes": {
|
|
||||||
"title": title if isinstance(title, str) else "",
|
|
||||||
"content": caption if isinstance(caption, str) else "",
|
|
||||||
"published_at": work.get("create_date"),
|
|
||||||
"post_type": wtype if isinstance(wtype, str) else "illust",
|
|
||||||
},
|
|
||||||
"_work": work,
|
|
||||||
}
|
|
||||||
|
|
||||||
# -- post-first seams ----------------------------------------------------
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def post_record_key(post: dict) -> tuple[str, str] | None:
|
|
||||||
"""`(ledger_key, post_id)` gating post-record capture through the seen
|
|
||||||
ledger (`post:<id>` synthetic key), or None when the work has no id."""
|
|
||||||
pid = post.get("id")
|
|
||||||
pid = str(pid) if pid is not None else ""
|
|
||||||
if not pid:
|
|
||||||
return None
|
|
||||||
return (f"post:{pid}", pid)
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def post_meta(post: dict) -> dict:
|
|
||||||
attrs = post.get("attributes") or {}
|
|
||||||
return {"title": attrs.get("title") or None, "date": attrs.get("published_at")}
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def post_is_gated(post: dict) -> bool:
|
|
||||||
"""True when this account cannot fetch the work's real files: pixiv
|
|
||||||
substitutes a `limit_*` placeholder for the original (sanity-level
|
|
||||||
filter / my-pixiv lock / deleted), or zeroes the author (deleted
|
|
||||||
account). Mirrors #874 semantics: gated content leaves NO trace — a
|
|
||||||
placeholder thumbnail and an empty stub would only pollute the
|
|
||||||
archive. (gallery-dl's PHPSESSID web-scrape fallback for these is out
|
|
||||||
of scope: FC holds no pixiv browser cookies — module docstring.)"""
|
|
||||||
work = post.get("_work") or {}
|
|
||||||
user = work.get("user") or {}
|
|
||||||
if not user.get("id"):
|
|
||||||
return True
|
|
||||||
if work.get("meta_pages"):
|
|
||||||
return False
|
|
||||||
single = work.get("meta_single_page") or {}
|
|
||||||
original = single.get("original_image_url")
|
|
||||||
return isinstance(original, str) and original.startswith(_LIMIT_URL)
|
|
||||||
|
|
||||||
# -- media ---------------------------------------------------------------
|
|
||||||
|
|
||||||
def extract_media(self, post: dict, included_index: dict) -> list[MediaItem]:
|
|
||||||
"""Resolve a work's downloadable files (gallery-dl's `_extract_files`):
|
|
||||||
multi-page originals, the single-page original, or the ugoira frame
|
|
||||||
zip. `included_index` is unused (pixiv works are self-contained)."""
|
|
||||||
work = post.get("_work") or {}
|
|
||||||
pid = str(post.get("id") or "")
|
|
||||||
if not pid or self.post_is_gated(post):
|
|
||||||
return []
|
|
||||||
|
|
||||||
if work.get("type") == "ugoira":
|
|
||||||
return self._ugoira_media(work, pid)
|
|
||||||
|
|
||||||
meta_pages = work.get("meta_pages") or []
|
|
||||||
if meta_pages:
|
|
||||||
items = []
|
|
||||||
for num, page in enumerate(meta_pages):
|
|
||||||
urls = page.get("image_urls") or {}
|
|
||||||
url = urls.get("original")
|
|
||||||
if not isinstance(url, str) or not url:
|
|
||||||
continue
|
|
||||||
items.append(
|
|
||||||
MediaItem(
|
|
||||||
url=url,
|
|
||||||
filename=_work_filename(work, num, url),
|
|
||||||
kind="image",
|
|
||||||
filehash=None,
|
|
||||||
post_id=pid,
|
|
||||||
media_id=f"p{num}",
|
|
||||||
)
|
|
||||||
)
|
|
||||||
return items
|
|
||||||
|
|
||||||
single = work.get("meta_single_page") or {}
|
|
||||||
url = single.get("original_image_url")
|
|
||||||
if not isinstance(url, str) or not url or url.startswith(_LIMIT_URL):
|
|
||||||
return []
|
|
||||||
return [
|
|
||||||
MediaItem(
|
|
||||||
url=url,
|
|
||||||
filename=_work_filename(work, 0, url),
|
|
||||||
kind="image",
|
|
||||||
filehash=None,
|
|
||||||
post_id=pid,
|
|
||||||
media_id="p0",
|
|
||||||
)
|
|
||||||
]
|
|
||||||
|
|
||||||
def _ugoira_meta(self, work: dict, pid: str) -> dict | None:
|
|
||||||
"""Fetch + memoize the ugoira metadata (frames + zip urls) for a work.
|
|
||||||
|
|
||||||
Idempotent and cached on the work dict, so the post record and the
|
|
||||||
media extraction share ONE `/v1/ugoira/metadata` call regardless of
|
|
||||||
which runs first (the core writes the post record BEFORE it extracts
|
|
||||||
media). Returns None — and caches the miss — on a non-auth failure
|
|
||||||
(matching gallery-dl's downgrade); auth failures stay loud."""
|
|
||||||
if "_ugoira_meta" in work:
|
|
||||||
return work["_ugoira_meta"]
|
|
||||||
try:
|
|
||||||
body = self._call("/v1/ugoira/metadata", {"illust_id": pid})
|
|
||||||
meta = body["ugoira_metadata"]
|
|
||||||
except PixivAuthError:
|
|
||||||
raise
|
|
||||||
except (PixivAPIError, KeyError, TypeError) as exc:
|
|
||||||
log.warning("Pixiv ugoira metadata failed for %s: %s", pid, exc)
|
|
||||||
work["_ugoira_meta"] = None
|
|
||||||
return None
|
|
||||||
work["_ugoira_meta"] = meta
|
|
||||||
# Frame delays: a future ugoira→video conversion needs the timings (the
|
|
||||||
# zip alone has none), so the post record captures them.
|
|
||||||
work["_ugoira_frames"] = meta.get("frames") or []
|
|
||||||
return meta
|
|
||||||
|
|
||||||
def fetch_ugoira_frames(self, post: dict) -> None:
|
|
||||||
"""Populate `post['_work']['_ugoira_frames']` for an ugoira post (no-op
|
|
||||||
otherwise). The core writes the post record BEFORE extract_media, so
|
|
||||||
without this the frame timings would never reach the record; this
|
|
||||||
fetches (and memoizes, so extract_media reuses it) the metadata. Injected
|
|
||||||
into the downloader by the ingester, mirroring Patreon's content_fetcher.
|
|
||||||
Auth errors propagate; other failures leave frames unset."""
|
|
||||||
work = post.get("_work") or {}
|
|
||||||
if work.get("type") != "ugoira":
|
|
||||||
return
|
|
||||||
pid = str(post.get("id") or "")
|
|
||||||
if pid:
|
|
||||||
self._ugoira_meta(work, pid)
|
|
||||||
|
|
||||||
def _ugoira_media(self, work: dict, pid: str) -> list[MediaItem]:
|
|
||||||
"""The ugoira frame zip (gallery-dl's default non-original mode):
|
|
||||||
`/v1/ugoira/metadata` → zip_urls.medium with the 600x600→1920x1080
|
|
||||||
swap. A metadata failure downgrades to 'no media' with a warning
|
|
||||||
(matching gallery-dl) instead of failing the walk — except auth
|
|
||||||
failures, which stay loud."""
|
|
||||||
meta = self._ugoira_meta(work, pid)
|
|
||||||
if meta is None:
|
|
||||||
return []
|
|
||||||
try:
|
|
||||||
zip_url = meta["zip_urls"]["medium"]
|
|
||||||
except (KeyError, TypeError) as exc:
|
|
||||||
log.warning("Pixiv ugoira zip url missing for %s: %s", pid, exc)
|
|
||||||
return []
|
|
||||||
url = zip_url.replace("_ugoira600x600", "_ugoira1920x1080", 1)
|
|
||||||
return [
|
|
||||||
MediaItem(
|
|
||||||
url=url,
|
|
||||||
filename=_work_filename(work, 0, url),
|
|
||||||
kind="ugoira",
|
|
||||||
filehash=None,
|
|
||||||
post_id=pid,
|
|
||||||
media_id="ugoira",
|
|
||||||
)
|
|
||||||
]
|
|
||||||
|
|
||||||
# -- iteration -----------------------------------------------------------
|
|
||||||
|
|
||||||
def iter_posts(
|
|
||||||
self, campaign_id: str, cursor: str | None = None
|
|
||||||
) -> Iterator[tuple[dict, dict, str | None]]:
|
|
||||||
"""Yield (post, {}, page_cursor) for every work in the user's feed.
|
|
||||||
|
|
||||||
`campaign_id` is the numeric pixiv user id. `cursor` is the query
|
|
||||||
string of the app API's `next_url` (offset pagination); None fetches
|
|
||||||
page 1. The yielded `page_cursor` is the cursor that FETCHED this
|
|
||||||
work's page, so the core checkpoints a value that re-serves the same
|
|
||||||
page on resume (the shared cursor contract)."""
|
|
||||||
if not str(campaign_id or "").isdigit():
|
|
||||||
raise PixivDriftError(
|
|
||||||
f"Pixiv campaign id must be a numeric user id, got "
|
|
||||||
f"{campaign_id!r}"
|
|
||||||
)
|
|
||||||
current = cursor
|
|
||||||
while True:
|
|
||||||
page_cursor = current
|
|
||||||
if current is None:
|
|
||||||
params: dict = {"user_id": campaign_id}
|
|
||||||
else:
|
|
||||||
params = dict(parse_qsl(current))
|
|
||||||
data = self._call("/v1/user/illusts", params)
|
|
||||||
works = data.get("illusts")
|
|
||||||
if not isinstance(works, list):
|
|
||||||
raise PixivDriftError(
|
|
||||||
"Pixiv user-illusts response had no 'illusts' list "
|
|
||||||
f"(keys: {sorted(data)[:8]})"
|
|
||||||
)
|
|
||||||
for work in works:
|
|
||||||
if not isinstance(work, dict):
|
|
||||||
continue
|
|
||||||
yield self._normalize(work), {}, page_cursor
|
|
||||||
next_url = data.get("next_url")
|
|
||||||
if not next_url:
|
|
||||||
return
|
|
||||||
current = str(next_url).rpartition("?")[2]
|
|
||||||
|
|
||||||
# -- user detail ---------------------------------------------------------
|
|
||||||
|
|
||||||
def resolve_display_name(self, user_id: str) -> str | None:
|
|
||||||
"""The pixiv user's display name via `/v1/user/detail` (gallery-dl's
|
|
||||||
user_detail) — used to name the Artist when a source is added by numeric
|
|
||||||
id. None on any failure (the caller falls back to the id)."""
|
|
||||||
try:
|
|
||||||
body = self._call("/v1/user/detail", {"user_id": str(user_id)})
|
|
||||||
except PixivAPIError:
|
|
||||||
return None
|
|
||||||
name = (body.get("user") or {}).get("name") if isinstance(body, dict) else None
|
|
||||||
return name if isinstance(name, str) and name.strip() else None
|
|
||||||
|
|
||||||
# -- verify --------------------------------------------------------------
|
|
||||||
|
|
||||||
def verify_auth(self) -> tuple[bool | None, str]:
|
|
||||||
"""Cheap credential probe: run the OAuth refresh (the thing that fails
|
|
||||||
when the token is bad) without walking any feed."""
|
|
||||||
try:
|
|
||||||
self._token_deadline = 0.0 # force a real refresh
|
|
||||||
self._login()
|
|
||||||
except PixivAuthError as exc:
|
|
||||||
return False, f"Pixiv rejected the credential — {exc}"
|
|
||||||
except PixivAPIError as exc:
|
|
||||||
return None, f"Couldn't verify (network/HTTP issue): {exc}"
|
|
||||||
account = self._authed_user.get("account") or self._authed_user.get("name")
|
|
||||||
suffix = f" as {account}" if account else ""
|
|
||||||
return True, f"Credentials valid — Pixiv OAuth refresh succeeded{suffix}."
|
|
||||||
|
|
||||||
|
|
||||||
def rating_label(x_restrict) -> str | None:
|
|
||||||
"""Human rating from pixiv's x_restrict (0/1/2) — written into the post
|
|
||||||
record so the archive keeps the R-18 flag without the reader needing to
|
|
||||||
know pixiv's numeric scheme."""
|
|
||||||
if isinstance(x_restrict, bool) or not isinstance(x_restrict, int):
|
|
||||||
return None
|
|
||||||
return _RATINGS.get(x_restrict)
|
|
||||||
@@ -1,276 +0,0 @@
|
|||||||
"""Native Pixiv media downloader — the Pixiv counterpart to
|
|
||||||
patreon_downloader / subscribestar_downloader.
|
|
||||||
|
|
||||||
Given a normalized Pixiv work and its resolved `MediaItem`s
|
|
||||||
(pixiv_client.extract_media), download the originals to gallery-dl's on-disk
|
|
||||||
layout (so pre-cutover gallery-dl downloads are recognized on disk and not
|
|
||||||
re-fetched), write the post-first sidecars the importer consumes, and report
|
|
||||||
per-media outcomes.
|
|
||||||
|
|
||||||
On-disk layout (matches FC's gallery-dl pixiv config, PLATFORM_DEFAULTS:
|
|
||||||
base-directory `<images_root>/<artist_slug>/pixiv` + `directory:
|
|
||||||
["{category}"]` + filename `{id}_{title[:50]}_{num:>02}.{extension}`):
|
|
||||||
|
|
||||||
<images_root>/<artist_slug>/pixiv/pixiv/<id>_<title50>_<NN>.<ext>
|
|
||||||
|
|
||||||
— note the intentional DOUBLE `pixiv` segment: gallery-dl appended
|
|
||||||
`{category}` under a base-directory that already ended in the platform name,
|
|
||||||
and tier-2 disk-skip parity requires reproducing that exactly. The layout is
|
|
||||||
FLAT (no per-post directory), so the post-first record is `_post_<id>.json`
|
|
||||||
in the same directory (the id suffix prevents the collisions a bare
|
|
||||||
`_post.json` would have here; phase 3 receives explicit post_record_paths, so
|
|
||||||
the name is a convention, not a discovery key).
|
|
||||||
|
|
||||||
Simpler than Patreon (no Mux/yt-dlp video branch) — the one special file is
|
|
||||||
the ugoira frame zip, downloaded as-is; FC's archive-containment import
|
|
||||||
extracts the frames, and the frame DELAYS ride the post record (the zip
|
|
||||||
carries none — a future ugoira→video conversion needs them).
|
|
||||||
|
|
||||||
PURE: no DB; the seen-skip is an injected predicate. FC runs on a plain-HTTP
|
|
||||||
homelab; nothing here uses a secure-context Web API.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
import logging
|
|
||||||
import re
|
|
||||||
import time
|
|
||||||
from collections.abc import Callable
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import requests
|
|
||||||
|
|
||||||
from .native_ingest_common import (
|
|
||||||
BaseNativeDownloader,
|
|
||||||
MediaOutcome,
|
|
||||||
PostRecordOutcome,
|
|
||||||
)
|
|
||||||
from .pixiv_client import PIXIV_APP_HEADERS, rating_label
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
# Control chars (0x00–0x1f + 0x7f DEL) — gallery-dl's default `path-remove`.
|
|
||||||
_GDL_PATH_REMOVE_RE = re.compile(r"[\x00-\x1f\x7f]")
|
|
||||||
|
|
||||||
|
|
||||||
def gdl_clean_filename(name: str) -> str:
|
|
||||||
"""Reproduce gallery-dl's on-disk filename EXACTLY as it wrote it on this
|
|
||||||
Linux host, so the tier-2 disk-skip recognizes pre-cutover files instead of
|
|
||||||
re-downloading them.
|
|
||||||
|
|
||||||
gallery-dl's PathFormat.build_filename is `clean_path(clean_segment(name))`.
|
|
||||||
On Linux (verified against gallery-dl 1.32.5 path.py) the defaults resolve to:
|
|
||||||
- path-restrict "auto" → "/" → clean_segment replaces ONLY "/" → "_"
|
|
||||||
- path-remove "\\x00-\\x1f\\x7f" → clean_path DELETES control chars
|
|
||||||
- path-strip "auto" → "" → NO trailing dot/space stripping
|
|
||||||
Crucially it does NOT touch the Windows-forbidden set (<>:"|?*) — those stay
|
|
||||||
raw in titles on disk. A stricter sanitizer here would rename any such title,
|
|
||||||
miss the on-disk match, and re-pull the whole work. Order mirrors gallery-dl
|
|
||||||
(segment inner, path outer); for these disjoint char sets it's commutative.
|
|
||||||
"""
|
|
||||||
return _GDL_PATH_REMOVE_RE.sub("", name.replace("/", "_"))
|
|
||||||
|
|
||||||
# Enrichment keys copied verbatim from the app-API work dict into the post
|
|
||||||
# record (they're already JSON scalars/objects). Everything lands in
|
|
||||||
# Post.raw_metadata via the importer, so the archive keeps pixiv's stats and
|
|
||||||
# structure without a schema change.
|
|
||||||
_WORK_PASSTHROUGH_KEYS = (
|
|
||||||
"type",
|
|
||||||
"page_count",
|
|
||||||
"width",
|
|
||||||
"height",
|
|
||||||
"total_view",
|
|
||||||
"total_bookmarks",
|
|
||||||
"total_comments",
|
|
||||||
"is_bookmarked",
|
|
||||||
"illust_ai_type",
|
|
||||||
"series",
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
class PixivDownloader(BaseNativeDownloader):
|
|
||||||
"""Download resolved Pixiv media to gallery-dl's on-disk layout.
|
|
||||||
Subclasses BaseNativeDownloader for the shared streaming GET
|
|
||||||
(transient-retry + Range-resume) and validation/quarantine. PURE: no DB."""
|
|
||||||
|
|
||||||
def __init__(
|
|
||||||
self,
|
|
||||||
images_root: Path,
|
|
||||||
cookies_path: str | None = None,
|
|
||||||
*,
|
|
||||||
validate: bool = True,
|
|
||||||
rate_limit: float = 0.0,
|
|
||||||
session: requests.Session | None = None,
|
|
||||||
ugoira_frames_fetcher: Callable[[dict], None] | None = None,
|
|
||||||
):
|
|
||||||
super().__init__(
|
|
||||||
images_root, cookies_path, platform="pixiv",
|
|
||||||
validate=validate, rate_limit=rate_limit, session=session,
|
|
||||||
)
|
|
||||||
# Injected by the ingester (client.fetch_ugoira_frames) so write_post_record
|
|
||||||
# can populate frame timings — which extract_media memoizes, but the core
|
|
||||||
# writes the post record FIRST. Mirrors Patreon's content_fetcher.
|
|
||||||
self._ugoira_frames_fetcher = ugoira_frames_fetcher
|
|
||||||
if session is None:
|
|
||||||
# i.pximg.net 403s any GET without the app Referer; mirror the
|
|
||||||
# client's full app-header profile (gallery-dl serves media off
|
|
||||||
# the same session it drives the API with). An injected session
|
|
||||||
# (tests) owns its own headers.
|
|
||||||
self.session.headers.update(PIXIV_APP_HEADERS)
|
|
||||||
|
|
||||||
# -- public ------------------------------------------------------------
|
|
||||||
|
|
||||||
def download_post(
|
|
||||||
self,
|
|
||||||
post: dict,
|
|
||||||
media_items: list,
|
|
||||||
artist_slug: str,
|
|
||||||
*,
|
|
||||||
is_seen: Callable[[object], bool] = lambda m: False,
|
|
||||||
should_stop: Callable[[], bool] = lambda: False,
|
|
||||||
recapture: bool = False,
|
|
||||||
) -> list[MediaOutcome]:
|
|
||||||
"""Download every media item of one work; return per-item outcomes.
|
|
||||||
Mirrors SubscribeStarDownloader.download_post (two-tier skip, mid-post
|
|
||||||
time-box, recapture surfacing)."""
|
|
||||||
flat_dir = self._flat_dir(artist_slug)
|
|
||||||
outcomes: list[MediaOutcome] = []
|
|
||||||
for media in media_items:
|
|
||||||
if should_stop():
|
|
||||||
break
|
|
||||||
try:
|
|
||||||
outcomes.append(
|
|
||||||
self._download_one(
|
|
||||||
post, media, flat_dir, artist_slug, is_seen,
|
|
||||||
recapture=recapture,
|
|
||||||
)
|
|
||||||
)
|
|
||||||
except Exception as exc: # resilient: isolate one item's failure
|
|
||||||
log.warning(
|
|
||||||
"Pixiv media failed (work %s, %s): %s",
|
|
||||||
post.get("id"), getattr(media, "media_id", "?"), exc,
|
|
||||||
)
|
|
||||||
outcomes.append(
|
|
||||||
MediaOutcome(media=media, status="error", path=None, error=str(exc))
|
|
||||||
)
|
|
||||||
return outcomes
|
|
||||||
|
|
||||||
def _flat_dir(self, artist_slug: str) -> Path:
|
|
||||||
# Double platform segment — gallery-dl layout parity (module docstring).
|
|
||||||
return self.images_root / artist_slug / "pixiv" / "pixiv"
|
|
||||||
|
|
||||||
# -- per-item ----------------------------------------------------------
|
|
||||||
|
|
||||||
def _download_one(
|
|
||||||
self,
|
|
||||||
post: dict,
|
|
||||||
media,
|
|
||||||
flat_dir: Path,
|
|
||||||
artist_slug: str,
|
|
||||||
is_seen: Callable[[object], bool],
|
|
||||||
*,
|
|
||||||
recapture: bool = False,
|
|
||||||
) -> MediaOutcome:
|
|
||||||
seen = is_seen(media)
|
|
||||||
if seen and not recapture:
|
|
||||||
return MediaOutcome(media=media, status="skipped_seen", path=None, error=None)
|
|
||||||
|
|
||||||
# The client's filename already carries the {id}_{title50}_{NN} shape
|
|
||||||
# (raw title, gallery-dl-template order); clean it to the byte-exact
|
|
||||||
# name gallery-dl wrote on disk so tier-2 disk-skip matches (else a
|
|
||||||
# re-download of the whole work). See gdl_clean_filename.
|
|
||||||
media_path = flat_dir / gdl_clean_filename(media.filename)
|
|
||||||
|
|
||||||
if media_path.exists(): # tier-2: already on disk
|
|
||||||
return MediaOutcome(
|
|
||||||
media=media, status="skipped_disk", path=media_path, error=None
|
|
||||||
)
|
|
||||||
# recapture: a seen item not on disk is NOT re-downloaded (recovery's job).
|
|
||||||
if seen:
|
|
||||||
return MediaOutcome(media=media, status="skipped_seen", path=None, error=None)
|
|
||||||
|
|
||||||
flat_dir.mkdir(parents=True, exist_ok=True)
|
|
||||||
if self._rate_limit > 0:
|
|
||||||
time.sleep(self._rate_limit)
|
|
||||||
|
|
||||||
out_path = self._fetch_get(media.url, media_path)
|
|
||||||
reason, quarantine_dest = self._validate_path(out_path, artist_slug, media.url)
|
|
||||||
if reason is not None:
|
|
||||||
return MediaOutcome(
|
|
||||||
media=media, status="quarantined", path=quarantine_dest, error=reason,
|
|
||||||
)
|
|
||||||
self._write_minimal_sidecar(post, out_path, source_url=media.url)
|
|
||||||
return MediaOutcome(media=media, status="downloaded", path=out_path, error=None)
|
|
||||||
|
|
||||||
# -- post record ---------------------------------------------------------
|
|
||||||
|
|
||||||
def write_post_record(self, post: dict, artist_slug: str) -> PostRecordOutcome:
|
|
||||||
"""Write the post-first `_post_<id>.json` — the sole writer of the post
|
|
||||||
body/metadata on the native path. Beyond the standard body fields, the
|
|
||||||
record carries pixiv's own structure (tags + EN translations, rating,
|
|
||||||
series, view/bookmark counts, AI flag, dimensions, author, ugoira frame
|
|
||||||
delays) so the archive keeps what the platform knows about the work."""
|
|
||||||
attrs = post.get("attributes") or {}
|
|
||||||
work = post.get("_work") or {}
|
|
||||||
title = attrs.get("title") if isinstance(attrs.get("title"), str) else None
|
|
||||||
post_type = attrs.get("post_type") if isinstance(attrs.get("post_type"), str) else None
|
|
||||||
pid = str(post.get("id") or "")
|
|
||||||
if not pid:
|
|
||||||
return PostRecordOutcome(
|
|
||||||
path=None, post_type=post_type, title=title, body_chars=0,
|
|
||||||
)
|
|
||||||
|
|
||||||
content = attrs.get("content")
|
|
||||||
content = content if isinstance(content, str) else ""
|
|
||||||
data: dict = {
|
|
||||||
"category": "pixiv",
|
|
||||||
"id": pid,
|
|
||||||
"title": title or "",
|
|
||||||
"content": content,
|
|
||||||
"published_at": attrs.get("published_at"),
|
|
||||||
# The post permalink is synthesized by platforms/pixiv.py
|
|
||||||
# derive_post_url from `id` at parse time — no url key here.
|
|
||||||
"rating": rating_label(work.get("x_restrict")),
|
|
||||||
}
|
|
||||||
for key in _WORK_PASSTHROUGH_KEYS:
|
|
||||||
if key in work:
|
|
||||||
data[key] = work[key]
|
|
||||||
tags = work.get("tags")
|
|
||||||
if isinstance(tags, list):
|
|
||||||
data["tags"] = [
|
|
||||||
{
|
|
||||||
"name": t.get("name"),
|
|
||||||
"translated_name": t.get("translated_name"),
|
|
||||||
}
|
|
||||||
for t in tags
|
|
||||||
if isinstance(t, dict)
|
|
||||||
]
|
|
||||||
user = work.get("user")
|
|
||||||
if isinstance(user, dict):
|
|
||||||
data["user"] = {
|
|
||||||
"id": user.get("id"),
|
|
||||||
"account": user.get("account"),
|
|
||||||
"name": user.get("name"),
|
|
||||||
}
|
|
||||||
# Ugoira frame timings. extract_media memoizes these, but the core writes
|
|
||||||
# the post record BEFORE extracting media, so fetch them here (shared +
|
|
||||||
# idempotent via the client's memoization) so the record actually keeps
|
|
||||||
# them — the zip carries no timings.
|
|
||||||
if (
|
|
||||||
work.get("type") == "ugoira"
|
|
||||||
and not work.get("_ugoira_frames")
|
|
||||||
and self._ugoira_frames_fetcher is not None
|
|
||||||
):
|
|
||||||
self._ugoira_frames_fetcher(post)
|
|
||||||
frames = work.get("_ugoira_frames")
|
|
||||||
if frames:
|
|
||||||
data["ugoira_frames"] = frames
|
|
||||||
|
|
||||||
flat_dir = self._flat_dir(artist_slug)
|
|
||||||
flat_dir.mkdir(parents=True, exist_ok=True)
|
|
||||||
path = flat_dir / f"_post_{pid}.json"
|
|
||||||
path.write_text(json.dumps(data, indent=2, ensure_ascii=False))
|
|
||||||
return PostRecordOutcome(
|
|
||||||
path=path, post_type=post_type, title=title, body_chars=len(content),
|
|
||||||
)
|
|
||||||
@@ -1,121 +0,0 @@
|
|||||||
"""Native Pixiv ingester — the Pixiv ADAPTER over the platform-agnostic core
|
|
||||||
(`ingest_core.Ingester`).
|
|
||||||
|
|
||||||
Thin counterpart to patreon_ingester / subscribestar_ingester: wires the Pixiv
|
|
||||||
client/downloader/ledger models/constraints/key into the core and supplies the
|
|
||||||
Pixiv failure mapping. The modes (tick / backfill / recovery / recapture), the
|
|
||||||
seen + dead-letter ledgers, cursor checkpointing, and the post-first capture
|
|
||||||
all live in the core. `download_service.download_source` drives
|
|
||||||
`PixivIngester.run` exactly as it drives the other two.
|
|
||||||
|
|
||||||
`campaign_id` is the numeric pixiv user id (download_backends extracts it from
|
|
||||||
the source URL — no network resolver). Auth is the operator's OAuth refresh
|
|
||||||
token (the token-type Credential), passed as `auth_token` — pixiv is the first
|
|
||||||
native platform authenticating by token rather than cookies, so the uniform
|
|
||||||
constructor accepts both and ignores what it doesn't need.
|
|
||||||
|
|
||||||
FC runs on a plain-HTTP homelab; nothing here uses a secure-context Web API.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import asyncio
|
|
||||||
import logging
|
|
||||||
from collections.abc import Callable
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from ..models import PixivFailedMedia, PixivSeenMedia
|
|
||||||
from .ingest_core import DEAD_LETTER_THRESHOLD, Ingester
|
|
||||||
from .pixiv_client import MediaItem, PixivAPIError, PixivClient
|
|
||||||
from .pixiv_downloader import PixivDownloader
|
|
||||||
|
|
||||||
__all__ = [
|
|
||||||
"DEAD_LETTER_THRESHOLD",
|
|
||||||
"PixivIngester",
|
|
||||||
"_ledger_key",
|
|
||||||
"verify_pixiv_credential",
|
|
||||||
]
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
_LEDGER_KEY_MAX = 128
|
|
||||||
|
|
||||||
|
|
||||||
def _ledger_key(media: MediaItem) -> str:
|
|
||||||
"""Stable per-media identity for the cross-run seen-ledger. Pixiv original
|
|
||||||
URLs carry no content hash, so the key is the page/zip identity scoped to
|
|
||||||
its work: `<illust_id>:p<num>` / `<illust_id>:ugoira`. Bounded to the
|
|
||||||
column width."""
|
|
||||||
if media.filehash:
|
|
||||||
return media.filehash
|
|
||||||
return f"{media.post_id}:{media.media_id}"[:_LEDGER_KEY_MAX]
|
|
||||||
|
|
||||||
|
|
||||||
class PixivIngester(Ingester):
|
|
||||||
"""Walk a pixiv user's works, download unseen originals, return a
|
|
||||||
`DownloadResult`. A thin adapter over `ingest_core.Ingester`; `client` /
|
|
||||||
`downloader` are injectable seams so unit tests run without network."""
|
|
||||||
|
|
||||||
def __init__(
|
|
||||||
self,
|
|
||||||
images_root: Path,
|
|
||||||
cookies_path: str | None,
|
|
||||||
session_factory: Callable[[], object],
|
|
||||||
*,
|
|
||||||
validate: bool = True,
|
|
||||||
rate_limit: float = 0.0,
|
|
||||||
request_sleep: float = 0.0,
|
|
||||||
auth_token: str | None = None,
|
|
||||||
client: PixivClient | None = None,
|
|
||||||
downloader: PixivDownloader | None = None,
|
|
||||||
):
|
|
||||||
self.images_root = Path(images_root)
|
|
||||||
self.cookies_path = str(cookies_path) if cookies_path else None
|
|
||||||
resolved_client = (
|
|
||||||
client
|
|
||||||
if client is not None
|
|
||||||
else PixivClient(auth_token, request_sleep=request_sleep)
|
|
||||||
)
|
|
||||||
resolved_downloader = (
|
|
||||||
downloader
|
|
||||||
if downloader is not None
|
|
||||||
else PixivDownloader(
|
|
||||||
self.images_root, cookies_path, validate=validate, rate_limit=rate_limit,
|
|
||||||
# write_post_record runs before extract_media in the core, so it
|
|
||||||
# fetches ugoira frame timings via the SAME client (shared,
|
|
||||||
# memoized) — else the record's ugoira_frames stays empty.
|
|
||||||
ugoira_frames_fetcher=resolved_client.fetch_ugoira_frames,
|
|
||||||
)
|
|
||||||
)
|
|
||||||
super().__init__(
|
|
||||||
client=resolved_client,
|
|
||||||
downloader=resolved_downloader,
|
|
||||||
session_factory=session_factory,
|
|
||||||
seen_model=PixivSeenMedia,
|
|
||||||
failed_model=PixivFailedMedia,
|
|
||||||
seen_constraint="uq_pixiv_seen_media_source_id",
|
|
||||||
failed_constraint="uq_pixiv_failed_media_source_id",
|
|
||||||
ledger_key=_ledger_key,
|
|
||||||
platform="pixiv",
|
|
||||||
error_base=PixivAPIError,
|
|
||||||
# API_DRIFT message phrasing; the base Ingester._failure_result owns
|
|
||||||
# the auth/drift/HTTP→error_type mapping (shared across platforms).
|
|
||||||
drift_label="Pixiv app API",
|
|
||||||
# Captions are legitimately empty for many pixiv artists, so the
|
|
||||||
# zero-bodies #862 canary would false-positive here; the client's
|
|
||||||
# response-shape checks (missing `illusts` → drift) cover the same
|
|
||||||
# failure class structurally.
|
|
||||||
body_canary=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
async def verify_pixiv_credential(
|
|
||||||
auth_token: str | None,
|
|
||||||
) -> tuple[bool | None, str]:
|
|
||||||
"""Native Pixiv credential probe — one OAuth refresh via
|
|
||||||
PixivClient.verify_auth (the exact call that fails when the token is
|
|
||||||
bad; no feed walk). Returns the uniform `(ok, message)` contract so
|
|
||||||
download_backends.verify_source_credential treats it like the others."""
|
|
||||||
client = PixivClient(auth_token)
|
|
||||||
loop = asyncio.get_running_loop()
|
|
||||||
return await loop.run_in_executor(None, client.verify_auth)
|
|
||||||
@@ -13,9 +13,8 @@ URL patterns match GS exactly so the existing browser extension
|
|||||||
hits FC unmodified. deviantart was dropped at #3069 (2026-08-27) —
|
hits FC unmodified. deviantart was dropped at #3069 (2026-08-27) —
|
||||||
FC downloaders are art-dedicated services only. pixiv was retired at
|
FC downloaders are art-dedicated services only. pixiv was retired at
|
||||||
milestone #406 (2026-09-13, rule #171): unregistered here first, which
|
milestone #406 (2026-09-13, rule #171): unregistered here first, which
|
||||||
switches it off everywhere this registry is consulted; `pixiv.py` and the
|
switched it off everywhere this registry is consulted, then removed from
|
||||||
pixiv client/downloader/ingester stay in the tree, uncalled, until the
|
the tree entirely in the milestone's phase 2 (2026-09-21).
|
||||||
milestone's phase 2 deletes them.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from .base import (
|
from .base import (
|
||||||
|
|||||||
@@ -24,8 +24,8 @@ from typing import Literal
|
|||||||
# external_post_id chain: `post_id` MUST come before `id` because
|
# external_post_id chain: `post_id` MUST come before `id` because
|
||||||
# SubscribeStar gallery-dl puts the per-attachment id in `id` and the
|
# SubscribeStar gallery-dl puts the per-attachment id in `id` and the
|
||||||
# actual post id in `post_id`; picking `id` first fragments
|
# actual post id in `post_id`; picking `id` first fragments
|
||||||
# multi-image SubscribeStar posts into N Post rows. Patreon/Pixiv have
|
# multi-image SubscribeStar posts into N Post rows. Patreon has
|
||||||
# no `post_id` so `id` still wins for them; HF uses `index`, Discord
|
# no `post_id` so `id` still wins for it; HF uses `index`, Discord
|
||||||
# uses `message_id` — all reached via the remaining chain entries.
|
# uses `message_id` — all reached via the remaining chain entries.
|
||||||
# (Banked 2026-05-27 during the sidecar audit.)
|
# (Banked 2026-05-27 during the sidecar audit.)
|
||||||
DEFAULT_EXTERNAL_POST_ID_KEYS: tuple[str, ...] = (
|
DEFAULT_EXTERNAL_POST_ID_KEYS: tuple[str, ...] = (
|
||||||
@@ -62,7 +62,7 @@ class PlatformInfo:
|
|||||||
# --- Behavioral hooks ---
|
# --- Behavioral hooks ---
|
||||||
# Synthesize a post permalink from sidecar data. Required when
|
# Synthesize a post permalink from sidecar data. Required when
|
||||||
# gallery-dl's `url` field is the file/CDN URL rather than the post
|
# gallery-dl's `url` field is the file/CDN URL rather than the post
|
||||||
# permalink (subscribestar/pixiv/hf/discord). None = trust the bare
|
# permalink (subscribestar/hf/discord). None = trust the bare
|
||||||
# `url` field (patreon).
|
# `url` field (patreon).
|
||||||
derive_post_url: Callable[[dict], str | None] | None = None
|
derive_post_url: Callable[[dict], str | None] | None = None
|
||||||
|
|
||||||
|
|||||||
@@ -1,38 +0,0 @@
|
|||||||
"""Pixiv — one quirk.
|
|
||||||
|
|
||||||
post_url: the sidecar's `url` (legacy gallery-dl era) is the image URL
|
|
||||||
on `i.pximg.net`, and the native post record (#129) writes no url key
|
|
||||||
at all — the permalink is synthesized from `id` here either way:
|
|
||||||
/artworks/<id>. external_post_id (= `id`) was already correct, so no
|
|
||||||
override there.
|
|
||||||
|
|
||||||
Downloads run through the native ingester (pixiv_ingester.py), not
|
|
||||||
gallery-dl; this registry entry still owns URL validation, sidecar
|
|
||||||
parsing, and the credential surface (the OAuth refresh token).
|
|
||||||
"""
|
|
||||||
|
|
||||||
from .base import GD_DEFAULTS, PlatformInfo, str_id_value
|
|
||||||
|
|
||||||
|
|
||||||
def derive_post_url(data: dict) -> str | None:
|
|
||||||
pid = str_id_value(data.get("id"))
|
|
||||||
if pid:
|
|
||||||
return f"https://www.pixiv.net/artworks/{pid}"
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
INFO = PlatformInfo(
|
|
||||||
key="pixiv",
|
|
||||||
name="Pixiv",
|
|
||||||
description="Download artwork from Pixiv artists",
|
|
||||||
auth_type="token",
|
|
||||||
requires_auth=True,
|
|
||||||
url_pattern=r"^https?://(www\.)?pixiv\.net/",
|
|
||||||
url_examples=[
|
|
||||||
"https://www.pixiv.net/users/12345678",
|
|
||||||
"https://www.pixiv.net/en/users/12345678",
|
|
||||||
],
|
|
||||||
default_config={**GD_DEFAULTS, "content_types": ["all"]},
|
|
||||||
notes="Requires OAuth refresh token. Run `gallery-dl oauth:pixiv` to obtain one.",
|
|
||||||
derive_post_url=derive_post_url,
|
|
||||||
)
|
|
||||||
@@ -229,7 +229,7 @@ async def scheduler_status(session: AsyncSession) -> dict:
|
|||||||
# links to cannot disagree about what they are counting.
|
# links to cannot disagree about what they are counting.
|
||||||
failing_sources = (await session.execute(
|
failing_sources = (await session.execute(
|
||||||
select(func.count()).select_from(Source)
|
select(func.count()).select_from(Source)
|
||||||
.where(Source.enabled.is_(True), failing_sources_clause())
|
.where(failing_sources_clause())
|
||||||
)).scalar_one()
|
)).scalar_one()
|
||||||
no_access_sources = (await session.execute(
|
no_access_sources = (await session.execute(
|
||||||
select(func.count()).select_from(Source)
|
select(func.count()).select_from(Source)
|
||||||
|
|||||||
@@ -1,125 +0,0 @@
|
|||||||
"""Placement reconciler tasks — plan, apply, revert (milestone #421).
|
|
||||||
|
|
||||||
The service (`services.library_layout`) holds the decisions; this module is
|
|
||||||
only the async wrapper, matching `tasks.library_audit`: run on the
|
|
||||||
maintenance queue, mark the run `error` with a traceback if anything escapes,
|
|
||||||
and return a small summary dict so eager-mode tests can assert on it.
|
|
||||||
|
|
||||||
Applying is a long run — 33,789 renames on the operator's library at the time
|
|
||||||
of writing — so `apply_placement` persists its ledger in chunks rather than
|
|
||||||
at the end. That ledger is the only record of where each file came from, and
|
|
||||||
a worker that dies two thirds of the way through must not take the undo
|
|
||||||
information for the first two thirds with it.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import traceback
|
|
||||||
from datetime import UTC, datetime
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from sqlalchemy.exc import DBAPIError, OperationalError
|
|
||||||
|
|
||||||
from ..celery_app import celery
|
|
||||||
from ..models import LibraryPlacementRun
|
|
||||||
from ..services import library_layout
|
|
||||||
from ._sync_engine import sync_session_factory as _sync_session_factory
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
IMAGES_ROOT = Path("/images")
|
|
||||||
|
|
||||||
# Commit the ledger every this many moves. Small enough that a crash loses
|
|
||||||
# seconds of work, large enough not to make a COMMIT per rename.
|
|
||||||
_APPLY_CHUNK = 200
|
|
||||||
|
|
||||||
|
|
||||||
def _fail(session, run_id: int, message: str) -> None:
|
|
||||||
run = session.get(LibraryPlacementRun, run_id)
|
|
||||||
if run is not None:
|
|
||||||
run.status = "error"
|
|
||||||
run.error = message
|
|
||||||
run.finished_at = datetime.now(UTC)
|
|
||||||
session.commit()
|
|
||||||
|
|
||||||
|
|
||||||
@celery.task(
|
|
||||||
name="backend.app.tasks.library_placement.plan_placement",
|
|
||||||
autoretry_for=(OperationalError, DBAPIError),
|
|
||||||
retry_backoff=5, retry_backoff_max=60, retry_jitter=True, max_retries=3,
|
|
||||||
soft_time_limit=900, time_limit=1000,
|
|
||||||
)
|
|
||||||
def plan_placement(artist_id: int | None = None) -> dict:
|
|
||||||
"""Build a move plan and leave it `ready` for the operator to read.
|
|
||||||
|
|
||||||
Reads rows and stats destinations; moves nothing.
|
|
||||||
"""
|
|
||||||
SessionLocal = _sync_session_factory()
|
|
||||||
with SessionLocal() as session:
|
|
||||||
run = library_layout.plan_placement(
|
|
||||||
session, IMAGES_ROOT, artist_id=artist_id,
|
|
||||||
)
|
|
||||||
session.commit()
|
|
||||||
return {
|
|
||||||
"run_id": run.id, "status": run.status,
|
|
||||||
"planned_count": run.planned_count,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
@celery.task(
|
|
||||||
name="backend.app.tasks.library_placement.apply_placement",
|
|
||||||
soft_time_limit=7200, time_limit=7500,
|
|
||||||
)
|
|
||||||
def apply_placement(run_id: int) -> dict:
|
|
||||||
"""Execute a `ready` run's stored plan. Renames files and rewrites rows.
|
|
||||||
|
|
||||||
No autoretry: a retry would re-enter a half-applied plan on a schedule
|
|
||||||
nobody asked for. Re-running IS safe (the applied rows refuse as "row
|
|
||||||
moved since planning"), but that should be the operator's decision after
|
|
||||||
reading what happened, not the queue's.
|
|
||||||
"""
|
|
||||||
SessionLocal = _sync_session_factory()
|
|
||||||
with SessionLocal() as session:
|
|
||||||
run = session.get(LibraryPlacementRun, run_id)
|
|
||||||
if run is None:
|
|
||||||
return {"run_id": run_id, "status": "missing"}
|
|
||||||
if run.status != "ready":
|
|
||||||
return {"run_id": run_id, "status": run.status, "skipped": True}
|
|
||||||
try:
|
|
||||||
library_layout.apply_run(session, run, chunk=_APPLY_CHUNK)
|
|
||||||
session.commit()
|
|
||||||
except Exception:
|
|
||||||
log.exception("placement apply failed for run %s", run_id)
|
|
||||||
session.rollback()
|
|
||||||
_fail(session, run_id, traceback.format_exc())
|
|
||||||
return {"run_id": run_id, "status": "error"}
|
|
||||||
return {
|
|
||||||
"run_id": run_id, "status": run.status,
|
|
||||||
"moved": run.moved_count, "refused": run.refused_count,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
@celery.task(
|
|
||||||
name="backend.app.tasks.library_placement.revert_placement",
|
|
||||||
soft_time_limit=7200, time_limit=7500,
|
|
||||||
)
|
|
||||||
def revert_placement(run_id: int) -> dict:
|
|
||||||
"""Put an applied run's files back where they came from."""
|
|
||||||
SessionLocal = _sync_session_factory()
|
|
||||||
with SessionLocal() as session:
|
|
||||||
run = session.get(LibraryPlacementRun, run_id)
|
|
||||||
if run is None:
|
|
||||||
return {"run_id": run_id, "status": "missing"}
|
|
||||||
if run.status != "applied":
|
|
||||||
return {"run_id": run_id, "status": run.status, "skipped": True}
|
|
||||||
try:
|
|
||||||
library_layout.revert_run(session, run)
|
|
||||||
session.commit()
|
|
||||||
except Exception:
|
|
||||||
log.exception("placement revert failed for run %s", run_id)
|
|
||||||
session.rollback()
|
|
||||||
_fail(session, run_id, traceback.format_exc())
|
|
||||||
return {"run_id": run_id, "status": "error"}
|
|
||||||
return {
|
|
||||||
"run_id": run_id, "status": run.status,
|
|
||||||
"refused": run.refused_count,
|
|
||||||
}
|
|
||||||
+1
-2
@@ -1,7 +1,7 @@
|
|||||||
# FabledCurator Firefox Extension
|
# FabledCurator Firefox Extension
|
||||||
|
|
||||||
Self-hosted Firefox extension that pushes session cookies from supported
|
Self-hosted Firefox extension that pushes session cookies from supported
|
||||||
platforms (Patreon, SubscribeStar, Hentai-Foundry, Discord, Pixiv)
|
platforms (Patreon, SubscribeStar, Hentai-Foundry, Discord)
|
||||||
into FabledCurator, and lets you add a creator as a Source from their
|
into FabledCurator, and lets you add a creator as a Source from their
|
||||||
page in one click.
|
page in one click.
|
||||||
|
|
||||||
@@ -33,7 +33,6 @@ npm run build # unsigned XPI in web-ext-artifacts/
|
|||||||
- [ ] Options page accepts FC URL + key, indicator turns green
|
- [ ] Options page accepts FC URL + key, indicator turns green
|
||||||
- [ ] Cookie export: log into patreon.com, click Patreon card → "X cookies exported"
|
- [ ] Cookie export: log into patreon.com, click Patreon card → "X cookies exported"
|
||||||
- [ ] Discord token: open discord.com, click Discord card → "Token captured"
|
- [ ] Discord token: open discord.com, click Discord card → "Token captured"
|
||||||
- [ ] Pixiv OAuth: click Pixiv card → login redirects, token stored
|
|
||||||
- [ ] Add as source: visit patreon.com/<creator>, click floating button → toast
|
- [ ] Add as source: visit patreon.com/<creator>, click floating button → toast
|
||||||
- [ ] Subscriptions list: popup → "Sources" tab → list renders
|
- [ ] Subscriptions list: popup → "Sources" tab → list renders
|
||||||
- [ ] Check now: click play icon on source row → no error toast
|
- [ ] Check now: click play icon on source row → no error toast
|
||||||
|
|||||||
@@ -0,0 +1,138 @@
|
|||||||
|
{
|
||||||
|
"$comment": [
|
||||||
|
"THE SHARED ARTIFACT for the JS<->Py artist-pattern mirror (issue #3093).",
|
||||||
|
"",
|
||||||
|
"extension/lib/platforms.js PLATFORM_ARTIST_PATTERNS and",
|
||||||
|
"backend/app/services/extension_service.py _PLATFORM_PATTERNS are two hand-kept",
|
||||||
|
"copies of one table, and they gate OPPOSITE HALVES of a single interaction:",
|
||||||
|
"the JS copy decides whether the 'Add to FC' button appears, the Python copy",
|
||||||
|
"decides whether the resulting POST is accepted. Drift is therefore never",
|
||||||
|
"cosmetic -- JS looser than Py shows a button that 400s, Py looser than JS",
|
||||||
|
"silently never offers a button for a URL the backend would take. Issue #1485",
|
||||||
|
"was the second of those, and its fix had to be applied to both files by hand.",
|
||||||
|
"",
|
||||||
|
"Neither runtime imports the other. This file is the shared artifact instead:",
|
||||||
|
"both suites read it and assert it against their OWN copy of the patterns, so",
|
||||||
|
"a change to one copy alone fails the other runtime's suite.",
|
||||||
|
" - extension/test/platforms.spec.js (vitest)",
|
||||||
|
" - tests/test_extension_artist_patterns.py (pytest)",
|
||||||
|
"",
|
||||||
|
"Lives under extension/test/ because that path is excluded from BOTH the XPI",
|
||||||
|
"file set and the extension version derivation (see scripts/packaging.sh:",
|
||||||
|
"NOT_PACKAGED_TRACKED and NOT_VERSION_RELEVANT both carry 'test/**'), so",
|
||||||
|
"adding samples here never ships bytes and never forces a re-sign.",
|
||||||
|
"",
|
||||||
|
"Adding a case: put the URL in the platform's match/no_match list with a",
|
||||||
|
"'why'. Run both suites. If only one goes red, you have found drift, which is",
|
||||||
|
"the entire point of the file.",
|
||||||
|
"",
|
||||||
|
"'slug' is read by the Python side only -- its _derive returns (platform,",
|
||||||
|
"slug) where the JS isArtistPage returns a boolean. It is not optional on a",
|
||||||
|
"match entry; the backend deriving the WRONG slug from a URL both copies",
|
||||||
|
"agree on is its own defect class, and this pins it."
|
||||||
|
],
|
||||||
|
|
||||||
|
"patreon": {
|
||||||
|
"match": [
|
||||||
|
{
|
||||||
|
"url": "https://www.patreon.com/maewix",
|
||||||
|
"slug": "maewix",
|
||||||
|
"why": "bare creator root"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://patreon.com/maewix",
|
||||||
|
"slug": "maewix",
|
||||||
|
"why": "www is optional"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "http://patreon.com/maewix",
|
||||||
|
"slug": "maewix",
|
||||||
|
"why": "http as well as https"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://www.patreon.com/c/Atole",
|
||||||
|
"slug": "Atole",
|
||||||
|
"why": "#1485: the /c/ creator shape"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://www.patreon.com/cw/Atole",
|
||||||
|
"slug": "Atole",
|
||||||
|
"why": "#1485: /cw/ is the 'creator workspace' URL Patreon serves once you are SUBSCRIBED -- exactly when the button matters most, and exactly what the pre-#1485 pattern missed"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://www.patreon.com/cw/Atole/posts",
|
||||||
|
"slug": "Atole",
|
||||||
|
"why": "#1485: a creator inner page still derives the creator"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://www.patreon.com/Atole/membership",
|
||||||
|
"slug": "Atole",
|
||||||
|
"why": "#1485: inner page on the bare shape"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"no_match": [
|
||||||
|
{ "url": "https://www.patreon.com/home", "why": "nav page, not a creator" },
|
||||||
|
{ "url": "https://www.patreon.com/search", "why": "nav page" },
|
||||||
|
{ "url": "https://www.patreon.com/messages", "why": "nav page" },
|
||||||
|
{ "url": "https://www.patreon.com/notifications", "why": "nav page" },
|
||||||
|
{ "url": "https://www.patreon.com/library", "why": "nav page" },
|
||||||
|
{ "url": "https://www.patreon.com/settings", "why": "nav page" },
|
||||||
|
{ "url": "https://www.patreon.com/posts", "why": "nav page; also the post-permalink prefix" },
|
||||||
|
{ "url": "https://www.patreon.com/home/anything", "why": "a nav page's inner path is still not a creator" },
|
||||||
|
{ "url": "https://www.patreon.com/settings/profile", "why": "as above" }
|
||||||
|
]
|
||||||
|
},
|
||||||
|
|
||||||
|
"subscribestar": {
|
||||||
|
"match": [
|
||||||
|
{
|
||||||
|
"url": "https://www.subscribestar.com/foobar",
|
||||||
|
"slug": "foobar",
|
||||||
|
"why": "creator root on the .com TLD"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://subscribestar.adult/foobar",
|
||||||
|
"slug": "foobar",
|
||||||
|
"why": "creator root on the .adult TLD"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://subscribestar.adult/foobar/",
|
||||||
|
"slug": "foobar",
|
||||||
|
"why": "trailing slash is tolerated"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"no_match": [
|
||||||
|
{ "url": "https://subscribestar.adult/feed", "why": "nav page" },
|
||||||
|
{ "url": "https://subscribestar.adult/messages", "why": "nav page" },
|
||||||
|
{ "url": "https://subscribestar.adult/library", "why": "nav page" },
|
||||||
|
{
|
||||||
|
"url": "https://subscribestar.adult/foobar/posts",
|
||||||
|
"why": "SubscribeStar's pattern end-anchors on the creator root, unlike Patreon's -- an inner page does NOT derive. Pinned so the asymmetry between the two platforms stays deliberate rather than becoming a silently-fixed bug in one copy only."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
|
||||||
|
"hentaifoundry": {
|
||||||
|
"match": [
|
||||||
|
{
|
||||||
|
"url": "https://www.hentai-foundry.com/user/Foo",
|
||||||
|
"slug": "Foo",
|
||||||
|
"why": "user page"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://www.hentai-foundry.com/user/Foo/profile",
|
||||||
|
"slug": "Foo",
|
||||||
|
"why": "inner page still derives the user"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://hentai-foundry.com/user/Foo",
|
||||||
|
"slug": "Foo",
|
||||||
|
"why": "www is optional"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"no_match": [
|
||||||
|
{ "url": "https://www.hentai-foundry.com/pictures/popular", "why": "gallery listing, not a user" },
|
||||||
|
{ "url": "https://www.hentai-foundry.com/", "why": "site root" }
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -190,3 +190,60 @@ describe('manifest.json agrees with the platform table', () => {
|
|||||||
}
|
}
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
describe('the JS<->Py artist-pattern mirror (#3093)', () => {
|
||||||
|
// PLATFORM_ARTIST_PATTERNS here and extension_service._PLATFORM_PATTERNS in
|
||||||
|
// the backend are two hand-kept copies of one table, and they gate OPPOSITE
|
||||||
|
// halves of a single interaction: this copy decides whether the "Add to FC"
|
||||||
|
// button appears, the Python copy decides whether the resulting POST is
|
||||||
|
// accepted. So JS-looser-than-Py shows a button that 400s, and
|
||||||
|
// Py-looser-than-JS never offers a button for a URL the backend would take.
|
||||||
|
// #1485 was the second of those, and its fix had to be applied to both
|
||||||
|
// files by hand.
|
||||||
|
//
|
||||||
|
// "Keep in sync by hand; reviewers catch drift" is the same guarantee
|
||||||
|
// manifest.json had before #3069, where deviantart survived seven weeks.
|
||||||
|
//
|
||||||
|
// The two-runtimes objection to a shared SOURCE file is fair, so the shared
|
||||||
|
// artifact is the SAMPLES instead: both suites read this JSON and assert it
|
||||||
|
// against their own copy of the patterns, and neither imports the other.
|
||||||
|
// The sibling half is tests/test_extension_artist_patterns.py; adding a
|
||||||
|
// sample there covers it here for free, and vice versa.
|
||||||
|
const samples = Object.fromEntries(
|
||||||
|
Object.entries(
|
||||||
|
JSON.parse(readFileSync(path.join(EXT_DIR, 'test', 'artist-url-samples.json'), 'utf8'))
|
||||||
|
).filter(([key]) => !key.startsWith('$'))
|
||||||
|
)
|
||||||
|
|
||||||
|
for (const [platform, spec] of Object.entries(samples)) {
|
||||||
|
// `slug` on a match entry is read by the Python half only — isArtistPage
|
||||||
|
// answers a boolean, while the backend's _derive returns (platform, slug).
|
||||||
|
for (const { url, why } of spec.match) {
|
||||||
|
it(`shows the button on ${url} — ${why}`, () => {
|
||||||
|
expect(isArtistPage(url, platform)).toBe(true)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
for (const { url, why } of spec.no_match) {
|
||||||
|
it(`hides the button on ${url} — ${why}`, () => {
|
||||||
|
expect(isArtistPage(url, platform)).toBe(false)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
it('has samples for every platform that has an artist pattern', () => {
|
||||||
|
// The guard's own coverage check: without it, deleting a platform's
|
||||||
|
// samples would make this block pass by testing less. Discord is
|
||||||
|
// deliberately in neither — it is channel-based, with no creator page to
|
||||||
|
// put a button on, so it has no artist pattern on either side.
|
||||||
|
expect(Object.keys(samples).sort()).toEqual(Object.keys(PLATFORM_ARTIST_PATTERNS).sort())
|
||||||
|
})
|
||||||
|
|
||||||
|
it('has samples in both directions for every platform', () => {
|
||||||
|
// A platform with only positive samples pins half the invariant. The
|
||||||
|
// no_match half is the one that catches a pattern quietly widening.
|
||||||
|
for (const [platform, spec] of Object.entries(samples)) {
|
||||||
|
expect(spec.match.length, `${platform} match samples`).toBeGreaterThan(0)
|
||||||
|
expect(spec.no_match.length, `${platform} no_match samples`).toBeGreaterThan(0)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|||||||
@@ -54,7 +54,6 @@
|
|||||||
Self-healing and repair: missing files, thumbnails, database upkeep.
|
Self-healing and repair: missing files, thumbnails, database upkeep.
|
||||||
</p>
|
</p>
|
||||||
<div class="fc-tile-grid">
|
<div class="fc-tile-grid">
|
||||||
<PlacementCard />
|
|
||||||
<MissingFileRepairCard />
|
<MissingFileRepairCard />
|
||||||
<ThumbnailBackfillCard />
|
<ThumbnailBackfillCard />
|
||||||
<DbMaintenanceCard />
|
<DbMaintenanceCard />
|
||||||
@@ -81,7 +80,6 @@ import MLBackfillCard from './MLBackfillCard.vue'
|
|||||||
import ThumbnailBackfillCard from './ThumbnailBackfillCard.vue'
|
import ThumbnailBackfillCard from './ThumbnailBackfillCard.vue'
|
||||||
import ArchiveReextractCard from './ArchiveReextractCard.vue'
|
import ArchiveReextractCard from './ArchiveReextractCard.vue'
|
||||||
import MissingFileRepairCard from './MissingFileRepairCard.vue'
|
import MissingFileRepairCard from './MissingFileRepairCard.vue'
|
||||||
import PlacementCard from './PlacementCard.vue'
|
|
||||||
import GpuTriageCard from './GpuTriageCard.vue'
|
import GpuTriageCard from './GpuTriageCard.vue'
|
||||||
import DbMaintenanceCard from './DbMaintenanceCard.vue'
|
import DbMaintenanceCard from './DbMaintenanceCard.vue'
|
||||||
import VideoEmbeddingCard from './VideoEmbeddingCard.vue'
|
import VideoEmbeddingCard from './VideoEmbeddingCard.vue'
|
||||||
|
|||||||
@@ -1,312 +0,0 @@
|
|||||||
<template>
|
|
||||||
<MaintenanceTile
|
|
||||||
icon="mdi-folder-move"
|
|
||||||
title="File placement"
|
|
||||||
blurb="Put every image in its own artist's folder, one artist at a time."
|
|
||||||
>
|
|
||||||
<p class="fc-muted text-body-2 mb-4">
|
|
||||||
The library keeps one folder per artist, named after them. Files written
|
|
||||||
under older rules can sit in another artist's folder — this moves them
|
|
||||||
home, updating the record and the file together. Every run can be
|
|
||||||
reverted, so the safe way to use it is one artist at a time: run it,
|
|
||||||
look at the gallery, then continue or put it back.
|
|
||||||
</p>
|
|
||||||
|
|
||||||
<v-alert
|
|
||||||
v-if="error" type="warning" variant="tonal" density="compact" class="mb-3"
|
|
||||||
>{{ error }}</v-alert>
|
|
||||||
|
|
||||||
<!-- Survey -->
|
|
||||||
<div class="fc-settings-row mb-3">
|
|
||||||
<v-btn
|
|
||||||
variant="tonal" rounded="pill" prepend-icon="mdi-magnify"
|
|
||||||
:loading="surveying" @click="onSurvey"
|
|
||||||
>Check placement</v-btn>
|
|
||||||
<span v-if="layout" class="fc-muted text-body-2">
|
|
||||||
{{ layout.misplaced_rows.toLocaleString() }} of
|
|
||||||
{{ layout.total_rows.toLocaleString() }} images are in the wrong folder
|
|
||||||
<template v-if="layout.misplaced_rows">
|
|
||||||
— across {{ layout.artists.length }} artists
|
|
||||||
</template>
|
|
||||||
</span>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<v-table v-if="layout && layout.artists.length" density="compact" class="mb-2">
|
|
||||||
<thead>
|
|
||||||
<tr>
|
|
||||||
<th>Artist</th>
|
|
||||||
<th class="text-right">To move</th>
|
|
||||||
<th>Currently in</th>
|
|
||||||
<th class="text-right">Plan</th>
|
|
||||||
</tr>
|
|
||||||
</thead>
|
|
||||||
<tbody>
|
|
||||||
<tr v-for="a in layout.artists" :key="a.artist_id">
|
|
||||||
<td>{{ a.name }}</td>
|
|
||||||
<td class="text-right fc-tabular">{{ a.misplaced_rows.toLocaleString() }}</td>
|
|
||||||
<td class="fc-muted text-body-2">{{ a.stray_dirs.join(', ') }}</td>
|
|
||||||
<td class="text-right">
|
|
||||||
<v-btn
|
|
||||||
size="small" variant="text" :loading="planningId === a.artist_id"
|
|
||||||
@click="onPlan(a.artist_id)"
|
|
||||||
>Plan</v-btn>
|
|
||||||
</td>
|
|
||||||
</tr>
|
|
||||||
</tbody>
|
|
||||||
</v-table>
|
|
||||||
|
|
||||||
<v-divider class="my-4" />
|
|
||||||
|
|
||||||
<!-- Runs -->
|
|
||||||
<h3 class="fc-section-title">Runs</h3>
|
|
||||||
<v-table density="compact">
|
|
||||||
<thead>
|
|
||||||
<tr>
|
|
||||||
<th>When</th>
|
|
||||||
<th>Scope</th>
|
|
||||||
<th>Status</th>
|
|
||||||
<th class="text-right">Planned</th>
|
|
||||||
<th class="text-right">Moved</th>
|
|
||||||
<th class="text-right">Refused</th>
|
|
||||||
<th class="text-right">Actions</th>
|
|
||||||
</tr>
|
|
||||||
</thead>
|
|
||||||
<tbody>
|
|
||||||
<tr v-for="r in store.placementRuns" :key="r.id">
|
|
||||||
<td class="fc-tabular" :title="r.started_at">
|
|
||||||
{{ formatRelative(r.started_at) }}
|
|
||||||
</td>
|
|
||||||
<td>{{ artistName(r.artist_id) }}</td>
|
|
||||||
<td>
|
|
||||||
<v-icon size="small" :color="statusColor(r.status)">
|
|
||||||
{{ statusIcon(r.status) }}
|
|
||||||
</v-icon>
|
|
||||||
{{ r.status }}
|
|
||||||
</td>
|
|
||||||
<td class="text-right fc-tabular">{{ r.planned_count.toLocaleString() }}</td>
|
|
||||||
<td class="text-right fc-tabular">{{ r.moved_count.toLocaleString() }}</td>
|
|
||||||
<td class="text-right fc-tabular">
|
|
||||||
<span :class="r.refused_count ? 'text-warning' : 'fc-muted'">
|
|
||||||
{{ r.refused_count.toLocaleString() }}
|
|
||||||
</span>
|
|
||||||
</td>
|
|
||||||
<td class="text-right">
|
|
||||||
<KebabMenu :label="`Actions for placement run ${r.id}`">
|
|
||||||
<v-list-item title="Review…" @click="onReview(r)" />
|
|
||||||
<v-list-item
|
|
||||||
title="Move files" :disabled="r.status !== 'ready' || !r.planned_count"
|
|
||||||
@click="onApply(r)"
|
|
||||||
/>
|
|
||||||
<v-list-item
|
|
||||||
title="Put back…" :disabled="r.status !== 'applied'"
|
|
||||||
@click="onRevert(r)"
|
|
||||||
/>
|
|
||||||
</KebabMenu>
|
|
||||||
</td>
|
|
||||||
</tr>
|
|
||||||
<tr v-if="!store.placementRuns.length">
|
|
||||||
<td colspan="7" class="text-center fc-muted py-4">
|
|
||||||
No runs yet. Check placement above, then plan one artist.
|
|
||||||
</td>
|
|
||||||
</tr>
|
|
||||||
</tbody>
|
|
||||||
</v-table>
|
|
||||||
|
|
||||||
<!-- Review: the plan, which is what gets agreed to -->
|
|
||||||
<v-dialog v-model="reviewOpen" max-width="900">
|
|
||||||
<v-card>
|
|
||||||
<v-card-title>
|
|
||||||
Run {{ reviewRun?.id }} — {{ reviewRun?.planned_count?.toLocaleString() }} moves
|
|
||||||
</v-card-title>
|
|
||||||
<v-card-text>
|
|
||||||
<p class="fc-muted text-body-2 mb-3">
|
|
||||||
Showing the first {{ REVIEW_LIMIT }}. Each row moves the file and
|
|
||||||
its record together; nothing is overwritten.
|
|
||||||
</p>
|
|
||||||
<v-table density="compact">
|
|
||||||
<tbody>
|
|
||||||
<tr v-for="m in reviewMoves" :key="m.image_id">
|
|
||||||
<td class="fc-muted text-caption">{{ m.from }}</td>
|
|
||||||
<td class="text-caption">→ {{ m.to }}</td>
|
|
||||||
</tr>
|
|
||||||
</tbody>
|
|
||||||
</v-table>
|
|
||||||
<div v-if="reviewRun?.refusals?.length" class="mt-4">
|
|
||||||
<h4 class="fc-section-title">Refused ({{ reviewRun.refusals.length }})</h4>
|
|
||||||
<p class="fc-muted text-body-2">
|
|
||||||
Rows the run declined to touch — the source moved, the
|
|
||||||
destination was taken, or the record changed since planning.
|
|
||||||
</p>
|
|
||||||
<div
|
|
||||||
v-for="f in reviewRun.refusals.slice(0, REVIEW_LIMIT)" :key="f.image_id"
|
|
||||||
class="text-caption fc-muted"
|
|
||||||
>#{{ f.image_id }} — {{ f.reason }}</div>
|
|
||||||
</div>
|
|
||||||
</v-card-text>
|
|
||||||
<v-card-actions>
|
|
||||||
<v-spacer />
|
|
||||||
<v-btn variant="text" @click="reviewOpen = false">Close</v-btn>
|
|
||||||
</v-card-actions>
|
|
||||||
</v-card>
|
|
||||||
</v-dialog>
|
|
||||||
|
|
||||||
<v-dialog v-model="confirmOpen" max-width="520">
|
|
||||||
<v-card>
|
|
||||||
<v-card-title>{{ confirmTitle }}</v-card-title>
|
|
||||||
<v-card-text class="text-body-2">{{ confirmMessage }}</v-card-text>
|
|
||||||
<v-card-actions>
|
|
||||||
<v-spacer />
|
|
||||||
<v-btn variant="text" @click="confirmOpen = false">Cancel</v-btn>
|
|
||||||
<v-btn
|
|
||||||
color="accent" variant="flat" rounded="pill"
|
|
||||||
@click="confirmOpen = false; onConfirmed()"
|
|
||||||
>Go ahead</v-btn>
|
|
||||||
</v-card-actions>
|
|
||||||
</v-card>
|
|
||||||
</v-dialog>
|
|
||||||
</MaintenanceTile>
|
|
||||||
</template>
|
|
||||||
|
|
||||||
<script setup>
|
|
||||||
import { computed, onMounted, onUnmounted, ref } from 'vue'
|
|
||||||
|
|
||||||
import KebabMenu from '../common/KebabMenu.vue'
|
|
||||||
import MaintenanceTile from '../common/MaintenanceTile.vue'
|
|
||||||
import { useCleanupStore } from '../../stores/cleanup.js'
|
|
||||||
import { formatRelative } from '../../utils/date.js'
|
|
||||||
import { toast } from '../../utils/toast.js'
|
|
||||||
|
|
||||||
// A whole-library plan is tens of thousands of moves. The dialog is for
|
|
||||||
// judging whether the moves look right, which a sample answers — the full
|
|
||||||
// list is in the run row if anyone needs it.
|
|
||||||
const REVIEW_LIMIT = 200
|
|
||||||
|
|
||||||
const store = useCleanupStore()
|
|
||||||
const layout = ref(null)
|
|
||||||
const artists = ref({})
|
|
||||||
const surveying = ref(false)
|
|
||||||
const planningId = ref(null)
|
|
||||||
const error = ref('')
|
|
||||||
|
|
||||||
const reviewOpen = ref(false)
|
|
||||||
const reviewRun = ref(null)
|
|
||||||
const reviewMoves = computed(() => (reviewRun.value?.moves || []).slice(0, REVIEW_LIMIT))
|
|
||||||
|
|
||||||
const confirmOpen = ref(false)
|
|
||||||
const confirmTitle = ref('')
|
|
||||||
const confirmMessage = ref('')
|
|
||||||
let confirmAction = null
|
|
||||||
|
|
||||||
let poll = null
|
|
||||||
|
|
||||||
function artistName(id) {
|
|
||||||
if (id === null || id === undefined) return 'Whole library'
|
|
||||||
return artists.value[id] || `#${id}`
|
|
||||||
}
|
|
||||||
|
|
||||||
function statusIcon(s) {
|
|
||||||
return {
|
|
||||||
ready: 'mdi-clipboard-text-outline', applied: 'mdi-check-circle',
|
|
||||||
reverted: 'mdi-undo-variant', error: 'mdi-close-circle',
|
|
||||||
running: 'mdi-timer-sand', cancelled: 'mdi-cancel',
|
|
||||||
}[s] || 'mdi-help-circle'
|
|
||||||
}
|
|
||||||
function statusColor(s) {
|
|
||||||
return {
|
|
||||||
ready: 'accent', applied: 'success', reverted: 'info',
|
|
||||||
error: 'error', running: 'warning',
|
|
||||||
}[s] || 'grey'
|
|
||||||
}
|
|
||||||
|
|
||||||
async function guard(fn) {
|
|
||||||
error.value = ''
|
|
||||||
try {
|
|
||||||
return await fn()
|
|
||||||
} catch (e) {
|
|
||||||
error.value = e?.message || String(e)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async function onSurvey() {
|
|
||||||
surveying.value = true
|
|
||||||
try {
|
|
||||||
await guard(async () => { layout.value = await store.loadLayout() })
|
|
||||||
} finally {
|
|
||||||
surveying.value = false
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async function onPlan(artistId) {
|
|
||||||
planningId.value = artistId
|
|
||||||
try {
|
|
||||||
await guard(async () => {
|
|
||||||
await store.planPlacement(artistId)
|
|
||||||
toast('Planning — the run appears below when it is ready')
|
|
||||||
await store.loadPlacementRuns()
|
|
||||||
})
|
|
||||||
} finally {
|
|
||||||
planningId.value = null
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async function onReview(run) {
|
|
||||||
await guard(async () => {
|
|
||||||
reviewRun.value = await store.getPlacementRun(run.id)
|
|
||||||
reviewOpen.value = true
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
function ask(title, message, action) {
|
|
||||||
confirmTitle.value = title
|
|
||||||
confirmMessage.value = message
|
|
||||||
confirmAction = action
|
|
||||||
confirmOpen.value = true
|
|
||||||
}
|
|
||||||
|
|
||||||
function onApply(run) {
|
|
||||||
ask(
|
|
||||||
'Move these files?',
|
|
||||||
`${run.planned_count.toLocaleString()} files move into their artist's `
|
|
||||||
+ 'folder. The record and the file move together, and this run can be '
|
|
||||||
+ 'put back afterwards.',
|
|
||||||
async () => {
|
|
||||||
await store.applyPlacement(run.id)
|
|
||||||
toast('Moving files — this runs in the background')
|
|
||||||
await store.loadPlacementRuns()
|
|
||||||
},
|
|
||||||
)
|
|
||||||
}
|
|
||||||
|
|
||||||
function onRevert(run) {
|
|
||||||
ask(
|
|
||||||
'Put these files back?',
|
|
||||||
`${run.moved_count.toLocaleString()} files return to where they were `
|
|
||||||
+ 'before this run.',
|
|
||||||
async () => {
|
|
||||||
await store.revertPlacement(run.id)
|
|
||||||
toast('Putting files back — this runs in the background')
|
|
||||||
await store.loadPlacementRuns()
|
|
||||||
},
|
|
||||||
)
|
|
||||||
}
|
|
||||||
|
|
||||||
async function onConfirmed() {
|
|
||||||
const action = confirmAction
|
|
||||||
confirmAction = null
|
|
||||||
if (action) await guard(action)
|
|
||||||
}
|
|
||||||
|
|
||||||
onMounted(async () => {
|
|
||||||
await guard(async () => {
|
|
||||||
artists.value = await store.loadArtistNames()
|
|
||||||
await store.loadPlacementRuns()
|
|
||||||
})
|
|
||||||
// Runs change state in a worker, so the table needs refreshing. The rows
|
|
||||||
// are server-side, which is why no localStorage resurfacing is needed here
|
|
||||||
// (useMaintenanceTask's job) — a reload or another machine sees the same
|
|
||||||
// state because the state IS the row.
|
|
||||||
poll = setInterval(() => { store.loadPlacementRuns().catch(() => {}) }, 5000)
|
|
||||||
})
|
|
||||||
|
|
||||||
onUnmounted(() => { if (poll) clearInterval(poll) })
|
|
||||||
</script>
|
|
||||||
@@ -71,66 +71,10 @@ export const useCleanupStore = defineStore('cleanup', () => {
|
|||||||
return await api.post(`/api/cleanup/audit/${id}/cancel`)
|
return await api.post(`/api/cleanup/audit/${id}/cancel`)
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- placement reconciler (milestone #421) --------------------------------
|
|
||||||
//
|
|
||||||
// Runs are server-side rows, so the DATABASE is the durable state here —
|
|
||||||
// no localStorage resurfacing (useMaintenanceTask) is needed. Reload the
|
|
||||||
// page, open it on another machine, and the run and its status are simply
|
|
||||||
// there. That also means a plan survives being walked away from for a day.
|
|
||||||
|
|
||||||
const placementRuns = ref([])
|
|
||||||
const layout = ref(null)
|
|
||||||
|
|
||||||
// The survey: which rows sit outside their artist's directory. check_disk
|
|
||||||
// additionally stats every destination (collisions, missing sources) and
|
|
||||||
// costs one stat per misplaced row over NFS, so it is opt-in.
|
|
||||||
async function loadLayout(checkDisk = false) {
|
|
||||||
layout.value = await api.get('/api/cleanup/layout', {
|
|
||||||
params: checkDisk ? { check_disk: 1 } : {},
|
|
||||||
})
|
|
||||||
return layout.value
|
|
||||||
}
|
|
||||||
|
|
||||||
// id -> name, so a run row can say "Conto" instead of "#47". The runs
|
|
||||||
// endpoint carries artist_id alone: the name belongs to the artist, and
|
|
||||||
// denormalising it into every run would go stale the moment one is renamed.
|
|
||||||
async function loadArtistNames() {
|
|
||||||
const rows = await api.get('/api/artists/names')
|
|
||||||
return Object.fromEntries((rows || []).map(a => [a.id, a.name]))
|
|
||||||
}
|
|
||||||
|
|
||||||
async function loadPlacementRuns(limit = 25) {
|
|
||||||
const body = await api.get('/api/cleanup/placement/runs', { params: { limit } })
|
|
||||||
placementRuns.value = body.runs || []
|
|
||||||
return placementRuns.value
|
|
||||||
}
|
|
||||||
|
|
||||||
// Detail carries `moves` — the plan the operator reads before agreeing.
|
|
||||||
async function getPlacementRun(id) {
|
|
||||||
return await api.get(`/api/cleanup/placement/runs/${id}`)
|
|
||||||
}
|
|
||||||
|
|
||||||
async function planPlacement(artistId = null) {
|
|
||||||
return await api.post('/api/cleanup/placement/plan', {
|
|
||||||
body: artistId === null ? {} : { artist_id: artistId },
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
async function applyPlacement(id) {
|
|
||||||
return await api.post(`/api/cleanup/placement/runs/${id}/apply`)
|
|
||||||
}
|
|
||||||
|
|
||||||
async function revertPlacement(id) {
|
|
||||||
return await api.post(`/api/cleanup/placement/runs/${id}/revert`)
|
|
||||||
}
|
|
||||||
|
|
||||||
return {
|
return {
|
||||||
defaults, recentRuns,
|
defaults, recentRuns,
|
||||||
loadDefaults,
|
loadDefaults,
|
||||||
previewMinDim, deleteMinDim,
|
previewMinDim, deleteMinDim,
|
||||||
startAudit, getAudit, loadHistory, latestAuditForRule, applyAudit, cancelAudit,
|
startAudit, getAudit, loadHistory, latestAuditForRule, applyAudit, cancelAudit,
|
||||||
placementRuns, layout,
|
|
||||||
loadLayout, loadArtistNames, loadPlacementRuns, getPlacementRun,
|
|
||||||
planPlacement, applyPlacement, revertPlacement,
|
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -1,111 +0,0 @@
|
|||||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'
|
|
||||||
import { setActivePinia, createPinia } from 'pinia'
|
|
||||||
import { useCleanupStore } from '../src/stores/cleanup.js'
|
|
||||||
import { stubFetch } from './stubFetch.js'
|
|
||||||
|
|
||||||
|
|
||||||
describe('placement reconciler store (milestone #421)', () => {
|
|
||||||
beforeEach(() => setActivePinia(createPinia()))
|
|
||||||
afterEach(() => vi.restoreAllMocks())
|
|
||||||
|
|
||||||
it('loadLayout leaves the disk check off by default', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
let seen = ''
|
|
||||||
stubFetch((url) => {
|
|
||||||
seen = url
|
|
||||||
return { status: 200, body: { total_rows: 10, misplaced_rows: 2, artists: [] } }
|
|
||||||
})
|
|
||||||
await s.loadLayout()
|
|
||||||
// One stat per misplaced row over NFS is the cost; it must be opt-in.
|
|
||||||
expect(seen).not.toContain('check_disk')
|
|
||||||
expect(s.layout.misplaced_rows).toBe(2)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('loadLayout asks for the disk check when requested', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
let seen = ''
|
|
||||||
stubFetch((url) => {
|
|
||||||
seen = url
|
|
||||||
return { status: 200, body: { total_rows: 0, misplaced_rows: 0, artists: [] } }
|
|
||||||
})
|
|
||||||
await s.loadLayout(true)
|
|
||||||
expect(seen).toContain('check_disk=1')
|
|
||||||
})
|
|
||||||
|
|
||||||
it('planPlacement scopes to an artist when given one', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
let sent = null
|
|
||||||
stubFetch((url, init) => {
|
|
||||||
sent = JSON.parse(init.body)
|
|
||||||
return { status: 202, body: { status: 'dispatched' } }
|
|
||||||
})
|
|
||||||
await s.planPlacement(47)
|
|
||||||
expect(sent).toEqual({ artist_id: 47 })
|
|
||||||
})
|
|
||||||
|
|
||||||
it('planPlacement sends no scope for the whole library', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
let sent = null
|
|
||||||
stubFetch((url, init) => {
|
|
||||||
sent = JSON.parse(init.body)
|
|
||||||
return { status: 202, body: { status: 'dispatched' } }
|
|
||||||
})
|
|
||||||
await s.planPlacement()
|
|
||||||
// Not `{artist_id: null}` — the endpoint rejects a non-integer, and an
|
|
||||||
// absent key is how "whole library" is spelled.
|
|
||||||
expect(sent).toEqual({})
|
|
||||||
})
|
|
||||||
|
|
||||||
it('loadPlacementRuns keeps the rows for the table', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
stubFetch(() => ({
|
|
||||||
status: 200,
|
|
||||||
body: { runs: [{ id: 3, status: 'ready', planned_count: 12 }] },
|
|
||||||
}))
|
|
||||||
await s.loadPlacementRuns()
|
|
||||||
expect(s.placementRuns).toHaveLength(1)
|
|
||||||
expect(s.placementRuns[0].status).toBe('ready')
|
|
||||||
})
|
|
||||||
|
|
||||||
it('getPlacementRun carries the moves — it is the preview', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
stubFetch(() => ({
|
|
||||||
status: 200,
|
|
||||||
body: {
|
|
||||||
id: 3, status: 'ready', planned_count: 1,
|
|
||||||
moves: [{ image_id: 9, from: '/images/Conto/x.png', to: '/images/conto/x.png' }],
|
|
||||||
},
|
|
||||||
}))
|
|
||||||
const run = await s.getPlacementRun(3)
|
|
||||||
expect(run.moves[0].from).toBe('/images/Conto/x.png')
|
|
||||||
expect(run.moves[0].to).toBe('/images/conto/x.png')
|
|
||||||
})
|
|
||||||
|
|
||||||
it('loadArtistNames maps id to name for the run rows', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
stubFetch(() => ({
|
|
||||||
status: 200,
|
|
||||||
body: [{ id: 47, name: 'Conto', slug: 'conto' }],
|
|
||||||
}))
|
|
||||||
expect(await s.loadArtistNames()).toEqual({ 47: 'Conto' })
|
|
||||||
})
|
|
||||||
|
|
||||||
it('loadArtistNames survives an empty roster', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
stubFetch(() => ({ status: 200, body: [] }))
|
|
||||||
expect(await s.loadArtistNames()).toEqual({})
|
|
||||||
})
|
|
||||||
|
|
||||||
it('applyPlacement and revertPlacement post to their own run', async () => {
|
|
||||||
const s = useCleanupStore()
|
|
||||||
const urls = []
|
|
||||||
stubFetch((url) => {
|
|
||||||
urls.push(url)
|
|
||||||
return { status: 202, body: { status: 'dispatched' } }
|
|
||||||
})
|
|
||||||
await s.applyPlacement(5)
|
|
||||||
await s.revertPlacement(5)
|
|
||||||
expect(urls[0]).toContain('/api/cleanup/placement/runs/5/apply')
|
|
||||||
expect(urls[1]).toContain('/api/cleanup/placement/runs/5/revert')
|
|
||||||
})
|
|
||||||
})
|
|
||||||
@@ -1,24 +0,0 @@
|
|||||||
import { vi } from 'vitest'
|
|
||||||
|
|
||||||
// The canonical fetch stub for store specs.
|
|
||||||
//
|
|
||||||
// `handler(url, init)` returns `{ status, body }`; body is JSON-encoded, and
|
|
||||||
// `ok` is derived from the status so a store's error path can be exercised by
|
|
||||||
// returning 4xx/5xx. Returns the vi.fn so a caller can assert on calls.
|
|
||||||
//
|
|
||||||
// Extracted 2026-09-21 from six specs carrying byte-identical copies
|
|
||||||
// (adminStore, credentials, dbMaintenance, gallery, suggestions,
|
|
||||||
// galleryRelatedStrip). Those still hold their own; migrate each the next
|
|
||||||
// time it is touched rather than in one sweep.
|
|
||||||
export function stubFetch (handler) {
|
|
||||||
globalThis.fetch = vi.fn(async (url, init) => {
|
|
||||||
const { status, body } = handler(url, init)
|
|
||||||
return {
|
|
||||||
ok: status >= 200 && status < 300,
|
|
||||||
status,
|
|
||||||
statusText: String(status),
|
|
||||||
text: async () => (body == null ? '' : JSON.stringify(body)),
|
|
||||||
}
|
|
||||||
})
|
|
||||||
return globalThis.fetch
|
|
||||||
}
|
|
||||||
-129
@@ -1,129 +0,0 @@
|
|||||||
{
|
|
||||||
"illusts": [
|
|
||||||
{
|
|
||||||
"id": 111,
|
|
||||||
"title": "Multi Page Adventure",
|
|
||||||
"type": "illust",
|
|
||||||
"caption": "Two-page set. <a href=\"https://example.com/wip\">WIP thread</a>",
|
|
||||||
"create_date": "2026-06-20T18:00:00+09:00",
|
|
||||||
"user": {"id": 99, "name": "Example Artist", "account": "exartist"},
|
|
||||||
"tags": [
|
|
||||||
{"name": "オリジナル", "translated_name": "original"},
|
|
||||||
{"name": "女の子", "translated_name": "girl"}
|
|
||||||
],
|
|
||||||
"page_count": 2,
|
|
||||||
"width": 1200,
|
|
||||||
"height": 1600,
|
|
||||||
"x_restrict": 0,
|
|
||||||
"series": {"id": 4242, "title": "Adventure Series"},
|
|
||||||
"total_view": 1000,
|
|
||||||
"total_bookmarks": 250,
|
|
||||||
"is_bookmarked": false,
|
|
||||||
"illust_ai_type": 1,
|
|
||||||
"meta_single_page": {},
|
|
||||||
"meta_pages": [
|
|
||||||
{
|
|
||||||
"image_urls": {
|
|
||||||
"square_medium": "https://i.pximg.net/c/360x360_70/img-master/img/2026/06/20/18/00/00/111_p0_square1200.jpg",
|
|
||||||
"original": "https://i.pximg.net/img-original/img/2026/06/20/18/00/00/111_p0.png"
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"image_urls": {
|
|
||||||
"square_medium": "https://i.pximg.net/c/360x360_70/img-master/img/2026/06/20/18/00/00/111_p1_square1200.jpg",
|
|
||||||
"original": "https://i.pximg.net/img-original/img/2026/06/20/18/00/00/111_p1.png"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 222,
|
|
||||||
"title": "Single Piece",
|
|
||||||
"type": "illust",
|
|
||||||
"caption": "",
|
|
||||||
"create_date": "2026-06-18T12:30:00+09:00",
|
|
||||||
"user": {"id": 99, "name": "Example Artist", "account": "exartist"},
|
|
||||||
"tags": [{"name": "落書き", "translated_name": "doodle"}],
|
|
||||||
"page_count": 1,
|
|
||||||
"width": 900,
|
|
||||||
"height": 900,
|
|
||||||
"x_restrict": 1,
|
|
||||||
"series": null,
|
|
||||||
"total_view": 500,
|
|
||||||
"total_bookmarks": 60,
|
|
||||||
"is_bookmarked": true,
|
|
||||||
"illust_ai_type": 0,
|
|
||||||
"meta_single_page": {
|
|
||||||
"original_image_url": "https://i.pximg.net/img-original/img/2026/06/18/12/30/00/222_p0.jpg"
|
|
||||||
},
|
|
||||||
"meta_pages": []
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 333,
|
|
||||||
"title": "Wiggle Loop",
|
|
||||||
"type": "ugoira",
|
|
||||||
"caption": "animated",
|
|
||||||
"create_date": "2026-06-15T09:00:00+09:00",
|
|
||||||
"user": {"id": 99, "name": "Example Artist", "account": "exartist"},
|
|
||||||
"tags": [{"name": "うごイラ", "translated_name": "ugoira"}],
|
|
||||||
"page_count": 1,
|
|
||||||
"width": 600,
|
|
||||||
"height": 600,
|
|
||||||
"x_restrict": 0,
|
|
||||||
"series": null,
|
|
||||||
"total_view": 300,
|
|
||||||
"total_bookmarks": 40,
|
|
||||||
"is_bookmarked": false,
|
|
||||||
"illust_ai_type": 0,
|
|
||||||
"meta_single_page": {
|
|
||||||
"original_image_url": "https://i.pximg.net/img-original/img/2026/06/15/09/00/00/333_ugoira0.jpg"
|
|
||||||
},
|
|
||||||
"meta_pages": []
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 444,
|
|
||||||
"title": "Blocked Work",
|
|
||||||
"type": "illust",
|
|
||||||
"caption": "",
|
|
||||||
"create_date": "2026-06-10T00:00:00+09:00",
|
|
||||||
"user": {"id": 99, "name": "Example Artist", "account": "exartist"},
|
|
||||||
"tags": [],
|
|
||||||
"page_count": 1,
|
|
||||||
"width": 0,
|
|
||||||
"height": 0,
|
|
||||||
"x_restrict": 2,
|
|
||||||
"series": null,
|
|
||||||
"total_view": 0,
|
|
||||||
"total_bookmarks": 0,
|
|
||||||
"is_bookmarked": false,
|
|
||||||
"illust_ai_type": 0,
|
|
||||||
"meta_single_page": {
|
|
||||||
"original_image_url": "https://s.pximg.net/common/images/limit_sanity_level_360.png"
|
|
||||||
},
|
|
||||||
"meta_pages": []
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 555,
|
|
||||||
"title": "Ghost Work",
|
|
||||||
"type": "illust",
|
|
||||||
"caption": "",
|
|
||||||
"create_date": "2026-06-01T00:00:00+09:00",
|
|
||||||
"user": {"id": 0, "name": "", "account": ""},
|
|
||||||
"tags": [],
|
|
||||||
"page_count": 1,
|
|
||||||
"width": 0,
|
|
||||||
"height": 0,
|
|
||||||
"x_restrict": 0,
|
|
||||||
"series": null,
|
|
||||||
"total_view": 0,
|
|
||||||
"total_bookmarks": 0,
|
|
||||||
"is_bookmarked": false,
|
|
||||||
"illust_ai_type": 0,
|
|
||||||
"meta_single_page": {
|
|
||||||
"original_image_url": "https://i.pximg.net/img-original/img/2026/06/01/00/00/00/555_p0.png"
|
|
||||||
},
|
|
||||||
"meta_pages": []
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"next_url": "https://app-api.pixiv.net/v1/user/illusts?user_id=99&offset=30"
|
|
||||||
}
|
|
||||||
@@ -87,29 +87,21 @@ async def test_quick_add_reuses_source_artist_after_rename(client, ext_key):
|
|||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_resolve_artist_name_dispatches_per_platform(db, monkeypatch):
|
async def test_resolve_artist_name_dispatches_per_platform(db, monkeypatch):
|
||||||
# #130: each native platform resolves its real display name at add-time
|
# #130: each native platform resolves its real display name at add-time
|
||||||
# (pixiv=token API, patreon=campaigns API, subscribestar=profile page);
|
# (patreon=campaigns API, subscribestar=profile page); gallery-dl platforms
|
||||||
# gallery-dl platforms and any failure fall back to the URL handle.
|
# and any failure fall back to the URL handle.
|
||||||
from backend.app.services import patreon_resolver
|
from backend.app.services import patreon_resolver
|
||||||
from backend.app.services.credential_service import CredentialService
|
from backend.app.services.credential_service import CredentialService
|
||||||
from backend.app.services.extension_service import ExtensionService
|
from backend.app.services.extension_service import ExtensionService
|
||||||
from backend.app.services.pixiv_client import PixivClient
|
|
||||||
from backend.app.services.subscribestar_client import SubscribeStarClient
|
from backend.app.services.subscribestar_client import SubscribeStarClient
|
||||||
|
|
||||||
async def _tok(self, platform):
|
|
||||||
return "tok"
|
|
||||||
|
|
||||||
async def _cookies(self, platform):
|
async def _cookies(self, platform):
|
||||||
return "/tmp/cookies.txt"
|
return "/tmp/cookies.txt"
|
||||||
|
|
||||||
monkeypatch.setattr(CredentialService, "get_token", _tok)
|
|
||||||
monkeypatch.setattr(CredentialService, "get_cookies_path", _cookies)
|
monkeypatch.setattr(CredentialService, "get_cookies_path", _cookies)
|
||||||
monkeypatch.setattr(PixivClient, "resolve_display_name", lambda self, uid: "Pixiv Name")
|
|
||||||
monkeypatch.setattr(patreon_resolver, "resolve_display_name", lambda v, c: "Patreon Name")
|
monkeypatch.setattr(patreon_resolver, "resolve_display_name", lambda v, c: "Patreon Name")
|
||||||
monkeypatch.setattr(SubscribeStarClient, "resolve_display_name", lambda self, u: "SS Name")
|
monkeypatch.setattr(SubscribeStarClient, "resolve_display_name", lambda self, u: "SS Name")
|
||||||
|
|
||||||
svc = ExtensionService(db, crypto=object()) # crypto seam only (calls stubbed)
|
svc = ExtensionService(db, crypto=object()) # crypto seam only (calls stubbed)
|
||||||
assert await svc._resolve_artist_name(
|
|
||||||
"pixiv", "555", "https://www.pixiv.net/users/555") == "Pixiv Name"
|
|
||||||
assert await svc._resolve_artist_name(
|
assert await svc._resolve_artist_name(
|
||||||
"patreon", "maewix", "https://patreon.com/maewix") == "Patreon Name"
|
"patreon", "maewix", "https://patreon.com/maewix") == "Patreon Name"
|
||||||
assert await svc._resolve_artist_name(
|
assert await svc._resolve_artist_name(
|
||||||
@@ -117,7 +109,7 @@ async def test_resolve_artist_name_dispatches_per_platform(db, monkeypatch):
|
|||||||
# gallery-dl platform → readable handle passthrough (no resolver).
|
# gallery-dl platform → readable handle passthrough (no resolver).
|
||||||
assert await svc._resolve_artist_name("hentaifoundry", "Foo", "u") == "Foo"
|
assert await svc._resolve_artist_name("hentaifoundry", "Foo", "u") == "Foo"
|
||||||
# No crypto → no resolution attempt → the raw handle.
|
# No crypto → no resolution attempt → the raw handle.
|
||||||
assert await ExtensionService(db)._resolve_artist_name("pixiv", "555", "u") == "555"
|
assert await ExtensionService(db)._resolve_artist_name("patreon", "maewix", "u") == "maewix"
|
||||||
# Resolver returns None → fall back to the handle.
|
# Resolver returns None → fall back to the handle.
|
||||||
monkeypatch.setattr(patreon_resolver, "resolve_display_name", lambda v, c: None)
|
monkeypatch.setattr(patreon_resolver, "resolve_display_name", lambda v, c: None)
|
||||||
assert await svc._resolve_artist_name("patreon", "maewix", "u") == "maewix"
|
assert await svc._resolve_artist_name("patreon", "maewix", "u") == "maewix"
|
||||||
|
|||||||
@@ -1,167 +0,0 @@
|
|||||||
"""The placement reconciler's task + API surface (milestone #421, slice 3b).
|
|
||||||
|
|
||||||
The move logic itself is covered in tests/test_library_layout.py; this module
|
|
||||||
covers the wrapper — that the tasks are registered and routed, that the
|
|
||||||
endpoints gate on run state, and that a list response stays small.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
from sqlalchemy import select
|
|
||||||
|
|
||||||
import backend.app.tasks.library_placement # noqa: F401 — register tasks
|
|
||||||
from backend.app.celery_app import celery
|
|
||||||
from backend.app.models import Artist, LibraryPlacementRun
|
|
||||||
|
|
||||||
pytestmark = pytest.mark.integration
|
|
||||||
|
|
||||||
_TASKS = (
|
|
||||||
"backend.app.tasks.library_placement.plan_placement",
|
|
||||||
"backend.app.tasks.library_placement.apply_placement",
|
|
||||||
"backend.app.tasks.library_placement.revert_placement",
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize("name", _TASKS)
|
|
||||||
def test_placement_tasks_are_registered(name):
|
|
||||||
assert name in celery.tasks
|
|
||||||
|
|
||||||
|
|
||||||
def test_placement_runs_on_the_long_maintenance_lane():
|
|
||||||
"""33k renames must not sit in the quick lane, which is where the
|
|
||||||
self-healing sweeps live (the 2026-06-07 starvation)."""
|
|
||||||
routes = celery.conf.task_routes
|
|
||||||
assert routes["backend.app.tasks.library_placement.*"] == {
|
|
||||||
"queue": "maintenance_long"
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def _run(db, status="ready", moves=None, artist_id=None):
|
|
||||||
"""Adds the row; the caller awaits the COMMIT.
|
|
||||||
|
|
||||||
Commit, not flush: the app under test runs on its own session and
|
|
||||||
connection, so a flush that stays inside this test's transaction is
|
|
||||||
invisible to the endpoint — the row simply is not there yet. Same reason
|
|
||||||
`_seed_runs` in test_api_system_backup commits."""
|
|
||||||
run = LibraryPlacementRun(
|
|
||||||
status=status, artist_id=artist_id, moves=moves or [],
|
|
||||||
planned_count=len(moves or []),
|
|
||||||
)
|
|
||||||
db.add(run)
|
|
||||||
return run
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_runs_list_omits_the_moves(client, db):
|
|
||||||
"""An applied whole-library run carries tens of thousands of entries.
|
|
||||||
Fine in Postgres, wrong in every list response."""
|
|
||||||
run = LibraryPlacementRun(
|
|
||||||
status="applied",
|
|
||||||
moves=[{"image_id": 1, "from": "/images/A/x.png", "to": "/images/a/x.png"}],
|
|
||||||
planned_count=1, moved_count=1,
|
|
||||||
)
|
|
||||||
db.add(run)
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
resp = await client.get("/api/cleanup/placement/runs")
|
|
||||||
assert resp.status_code == 200
|
|
||||||
body = await resp.get_json()
|
|
||||||
assert body["runs"][0]["planned_count"] == 1
|
|
||||||
assert "moves" not in body["runs"][0]
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_run_detail_carries_the_moves(client, db):
|
|
||||||
"""The detail IS the preview the operator reads before agreeing."""
|
|
||||||
run = LibraryPlacementRun(
|
|
||||||
status="ready",
|
|
||||||
moves=[{"image_id": 7, "from": "/images/Conto/x.png", "to": "/images/conto/x.png"}],
|
|
||||||
planned_count=1,
|
|
||||||
)
|
|
||||||
db.add(run)
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
resp = await client.get(f"/api/cleanup/placement/runs/{run.id}")
|
|
||||||
assert resp.status_code == 200
|
|
||||||
body = await resp.get_json()
|
|
||||||
assert body["moves"][0]["from"] == "/images/Conto/x.png"
|
|
||||||
assert body["moves"][0]["to"] == "/images/conto/x.png"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_run_detail_404s_for_an_unknown_run(client):
|
|
||||||
resp = await client.get("/api/cleanup/placement/runs/999999")
|
|
||||||
assert resp.status_code == 404
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_plan_rejects_a_non_integer_artist(client):
|
|
||||||
resp = await client.post(
|
|
||||||
"/api/cleanup/placement/plan", json={"artist_id": "conto"},
|
|
||||||
)
|
|
||||||
assert resp.status_code == 400
|
|
||||||
assert (await resp.get_json())["error"] == "invalid_artist_id"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_plan_accepts_an_artist_scope(client, db, monkeypatch):
|
|
||||||
sent = {}
|
|
||||||
from backend.app.tasks import library_placement
|
|
||||||
|
|
||||||
monkeypatch.setattr(
|
|
||||||
library_placement.plan_placement, "delay",
|
|
||||||
lambda artist_id=None: sent.update(artist_id=artist_id),
|
|
||||||
)
|
|
||||||
artist = Artist(name="Conto", slug="conto")
|
|
||||||
db.add(artist)
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
resp = await client.post(
|
|
||||||
"/api/cleanup/placement/plan", json={"artist_id": artist.id},
|
|
||||||
)
|
|
||||||
assert resp.status_code == 202
|
|
||||||
assert sent["artist_id"] == artist.id
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_apply_refuses_a_run_that_is_not_ready(client, db):
|
|
||||||
"""The gate is here as well as in the service — an applied run must not
|
|
||||||
be re-applied by a stray POST."""
|
|
||||||
run = _run(db, status="applied")
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
resp = await client.post(f"/api/cleanup/placement/runs/{run.id}/apply")
|
|
||||||
assert resp.status_code == 400
|
|
||||||
assert (await resp.get_json())["error"] == "not_ready"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_revert_refuses_a_run_that_was_never_applied(client, db):
|
|
||||||
run = _run(db, status="ready")
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
resp = await client.post(f"/api/cleanup/placement/runs/{run.id}/revert")
|
|
||||||
assert resp.status_code == 400
|
|
||||||
assert (await resp.get_json())["error"] == "not_applied"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_apply_dispatches_for_a_ready_run(client, db, monkeypatch):
|
|
||||||
sent = {}
|
|
||||||
from backend.app.tasks import library_placement
|
|
||||||
|
|
||||||
monkeypatch.setattr(
|
|
||||||
library_placement.apply_placement, "delay",
|
|
||||||
lambda run_id: sent.update(run_id=run_id),
|
|
||||||
)
|
|
||||||
run = _run(db, status="ready")
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
resp = await client.post(f"/api/cleanup/placement/runs/{run.id}/apply")
|
|
||||||
assert resp.status_code == 202
|
|
||||||
assert sent["run_id"] == run.id
|
|
||||||
# Dispatch only — the endpoint must not have moved anything itself.
|
|
||||||
still = (await db.execute(
|
|
||||||
select(LibraryPlacementRun.status)
|
|
||||||
.where(LibraryPlacementRun.id == run.id)
|
|
||||||
)).scalar_one()
|
|
||||||
assert still == "ready"
|
|
||||||
@@ -274,6 +274,47 @@ async def test_backfill_endpoint_start_and_stop(client, artist, db):
|
|||||||
assert (await stopped.get_json())["backfill_state"] is None
|
assert (await stopped.get_json())["backfill_state"] is None
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_backfill_endpoint_refuses_a_disabled_source(client, artist, db):
|
||||||
|
"""#4279: a source FC deliberately stopped must not be armable for a deep
|
||||||
|
walk. Arming one is how Ebi77 got a failure nobody could clear — the walk
|
||||||
|
cannot complete without access, the recovery sweep strands it, and a
|
||||||
|
disabled source is never scheduled again to reset the count."""
|
||||||
|
src = Source(
|
||||||
|
artist_id=artist.id, platform="patreon",
|
||||||
|
url="https://patreon.com/alice-stopped", enabled=False,
|
||||||
|
)
|
||||||
|
db.add(src)
|
||||||
|
await db.commit()
|
||||||
|
|
||||||
|
for action in ("start", "recover", "recapture"):
|
||||||
|
resp = await client.post(
|
||||||
|
f"/api/sources/{src.id}/backfill", json={"action": action},
|
||||||
|
)
|
||||||
|
assert resp.status_code == 400, action
|
||||||
|
assert (await resp.get_json())["error"] == "source_disabled"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_backfill_stop_still_works_on_a_disabled_source(client, artist, db):
|
||||||
|
"""Only the ARMING actions are gated. Cancelling a walk on a source that
|
||||||
|
was disabled mid-backfill must stay available, or the arm becomes a
|
||||||
|
one-way door."""
|
||||||
|
src = Source(
|
||||||
|
artist_id=artist.id, platform="patreon",
|
||||||
|
url="https://patreon.com/alice-stopping", enabled=False,
|
||||||
|
config_overrides={"_backfill_state": "running"},
|
||||||
|
)
|
||||||
|
db.add(src)
|
||||||
|
await db.commit()
|
||||||
|
|
||||||
|
resp = await client.post(
|
||||||
|
f"/api/sources/{src.id}/backfill", json={"action": "stop"},
|
||||||
|
)
|
||||||
|
assert resp.status_code == 200
|
||||||
|
assert (await resp.get_json())["backfill_state"] is None
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_backfill_endpoint_defaults_to_start(client, artist, db):
|
async def test_backfill_endpoint_defaults_to_start(client, artist, db):
|
||||||
src = Source(
|
src = Source(
|
||||||
|
|||||||
@@ -7,7 +7,6 @@ import pytest
|
|||||||
|
|
||||||
from backend.app.services.download_backends import (
|
from backend.app.services.download_backends import (
|
||||||
NATIVE_INGESTER_PLATFORMS,
|
NATIVE_INGESTER_PLATFORMS,
|
||||||
_campaign_resolution_error,
|
|
||||||
_native_ingester_cls,
|
_native_ingester_cls,
|
||||||
_unsupported_platform_message,
|
_unsupported_platform_message,
|
||||||
run_download,
|
run_download,
|
||||||
@@ -15,7 +14,6 @@ from backend.app.services.download_backends import (
|
|||||||
verify_source_credential,
|
verify_source_credential,
|
||||||
)
|
)
|
||||||
from backend.app.services.gallery_dl import ErrorType
|
from backend.app.services.gallery_dl import ErrorType
|
||||||
from backend.app.services.pixiv_ingester import PixivIngester
|
|
||||||
|
|
||||||
|
|
||||||
def test_native_platforms():
|
def test_native_platforms():
|
||||||
@@ -25,8 +23,9 @@ def test_native_platforms():
|
|||||||
|
|
||||||
|
|
||||||
def test_pixiv_is_no_longer_native():
|
def test_pixiv_is_no_longer_native():
|
||||||
"""Retired at milestone #406. The refusal below is what stops it falling
|
"""Retired at milestone #406 — unregistered in phase 1, deleted in phase 2.
|
||||||
through to gallery-dl now that it is not native."""
|
The refusal below is what stops it falling through to gallery-dl now that
|
||||||
|
it is neither native nor registered."""
|
||||||
assert uses_native_ingester("pixiv") is False
|
assert uses_native_ingester("pixiv") is False
|
||||||
assert "pixiv" not in NATIVE_INGESTER_PLATFORMS
|
assert "pixiv" not in NATIVE_INGESTER_PLATFORMS
|
||||||
|
|
||||||
@@ -113,11 +112,14 @@ def test_unknown_platform_is_not_native():
|
|||||||
assert uses_native_ingester("nonsense") is False
|
assert uses_native_ingester("nonsense") is False
|
||||||
|
|
||||||
|
|
||||||
def test_pixiv_dispatches_to_its_ingester():
|
def test_every_native_platform_dispatches_to_an_ingester():
|
||||||
assert _native_ingester_cls("pixiv") is PixivIngester
|
"""The dispatch table and NATIVE_INGESTER_PLATFORMS have to agree, or a
|
||||||
|
platform that routes native raises KeyError mid-download instead of being
|
||||||
|
refused up front. Written over the set rather than per-platform so adding
|
||||||
|
one to NATIVE_INGESTER_PLATFORMS and forgetting the class fails here.
|
||||||
|
|
||||||
|
(This replaces the per-platform dispatch tests, one of which was pixiv's;
|
||||||
|
it was deleted with pixiv at milestone #406 phase 2.)"""
|
||||||
|
for platform in NATIVE_INGESTER_PLATFORMS:
|
||||||
|
assert _native_ingester_cls(platform) is not None
|
||||||
|
|
||||||
def test_pixiv_resolution_error_names_the_expected_url_shape():
|
|
||||||
msg = _campaign_resolution_error("pixiv", "https://www.pixiv.net/artworks/1")
|
|
||||||
assert "pixiv user id" in msg
|
|
||||||
assert "users/<id>" in msg
|
|
||||||
|
|||||||
@@ -0,0 +1,121 @@
|
|||||||
|
"""The Python half of the JS<->Py artist-pattern mirror guard (#3093).
|
||||||
|
|
||||||
|
`extension/lib/platforms.js` PLATFORM_ARTIST_PATTERNS and
|
||||||
|
`extension_service._PLATFORM_PATTERNS` are two hand-kept copies of one table.
|
||||||
|
Both files say "keep in sync by hand; reviewers catch drift" — the same
|
||||||
|
guarantee `manifest.json` had before #3069, where deviantart sat in the
|
||||||
|
manifest for seven weeks after the product dropped it.
|
||||||
|
|
||||||
|
Drift here is worse than the manifest case, because the two copies gate
|
||||||
|
opposite halves of ONE interaction:
|
||||||
|
|
||||||
|
- the **JS** copy decides whether the "Add to FC" button appears;
|
||||||
|
- the **Python** copy decides whether the resulting POST is accepted.
|
||||||
|
|
||||||
|
So JS-looser-than-Py shows the operator a button that 400s, and
|
||||||
|
Py-looser-than-JS silently never offers a button for a URL the backend would
|
||||||
|
happily take. #1485 (Patreon's `/c/` and `/cw/` shapes) was exactly the
|
||||||
|
second, and its fix had to be applied to both files by hand.
|
||||||
|
|
||||||
|
The two-runtimes objection to a shared source file is fair, so this tests the
|
||||||
|
INVARIANT rather than the source: one table of URL samples, read by both
|
||||||
|
suites and asserted against each one's own copy of the patterns. Neither
|
||||||
|
runtime imports the other. A change to one copy alone turns the other
|
||||||
|
runtime's suite red.
|
||||||
|
|
||||||
|
The sibling half is `extension/test/platforms.spec.js`, which reads the same
|
||||||
|
file. Adding a sample there covers it here for free, and vice versa — which is
|
||||||
|
the property that makes the guard cheap enough to keep using.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from backend.app.services.extension_service import (
|
||||||
|
ExtensionService,
|
||||||
|
UnknownPlatformError,
|
||||||
|
)
|
||||||
|
|
||||||
|
_SAMPLES_PATH = (
|
||||||
|
Path(__file__).resolve().parents[1]
|
||||||
|
/ "extension" / "test" / "artist-url-samples.json"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _load_samples() -> dict:
|
||||||
|
"""The shared table, minus its `$comment` preamble."""
|
||||||
|
raw = json.loads(_SAMPLES_PATH.read_text(encoding="utf-8"))
|
||||||
|
return {k: v for k, v in raw.items() if not k.startswith("$")}
|
||||||
|
|
||||||
|
|
||||||
|
_SAMPLES = _load_samples()
|
||||||
|
|
||||||
|
_MATCH_CASES = [
|
||||||
|
pytest.param(platform, entry["url"], entry["slug"], entry["why"],
|
||||||
|
id=f"{platform}-match-{i}")
|
||||||
|
for platform, spec in _SAMPLES.items()
|
||||||
|
for i, entry in enumerate(spec["match"])
|
||||||
|
]
|
||||||
|
|
||||||
|
_NO_MATCH_CASES = [
|
||||||
|
pytest.param(platform, entry["url"], entry["why"],
|
||||||
|
id=f"{platform}-nomatch-{i}")
|
||||||
|
for platform, spec in _SAMPLES.items()
|
||||||
|
for i, entry in enumerate(spec["no_match"])
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _derive(url: str) -> tuple[str, str]:
|
||||||
|
"""`_derive` needs no session — it is pure regex over the URL."""
|
||||||
|
return ExtensionService(session=None)._derive(url)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("platform,url,slug,why", _MATCH_CASES)
|
||||||
|
def test_creator_url_derives_the_expected_platform_and_slug(platform, url, slug, why):
|
||||||
|
"""A URL the extension would show the button on must be one the backend
|
||||||
|
accepts, and it must derive the SAME creator. The slug is asserted, not
|
||||||
|
just the platform: deriving the wrong creator from a URL both copies agree
|
||||||
|
on is its own defect, and nothing else pins it."""
|
||||||
|
assert _derive(url) == (platform, slug), why
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("platform,url,why", _NO_MATCH_CASES)
|
||||||
|
def test_non_creator_url_derives_nothing(platform, url, why):
|
||||||
|
"""The other direction, and the one that fails silently. A URL the
|
||||||
|
extension refuses to show the button on must also be one the backend
|
||||||
|
refuses — otherwise the backend is quietly looser than the button, and
|
||||||
|
nobody finds out because nothing visibly breaks.
|
||||||
|
|
||||||
|
`_derive` is asserted to raise rather than merely to miss `platform`: it
|
||||||
|
tries every pattern in turn, so a nav page that some OTHER platform's
|
||||||
|
pattern happened to swallow would still be 'accepted by the backend',
|
||||||
|
which is the same defect wearing a different platform name.
|
||||||
|
"""
|
||||||
|
with pytest.raises(UnknownPlatformError):
|
||||||
|
_derive(url)
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_sample_table_covers_every_platform_that_has_a_pattern():
|
||||||
|
"""The guard's own coverage check. Without it, deleting a platform's
|
||||||
|
samples would make this file pass by testing less — the failure mode that
|
||||||
|
makes absence-based tests untrustworthy (snippet #3352).
|
||||||
|
|
||||||
|
Discord is deliberately absent from both: it has no artist pattern on
|
||||||
|
either side, because it is channel-based and has no creator page to put a
|
||||||
|
button on.
|
||||||
|
"""
|
||||||
|
from backend.app.services.extension_service import _PLATFORM_PATTERNS
|
||||||
|
|
||||||
|
assert set(_SAMPLES) == {platform for platform, _ in _PLATFORM_PATTERNS}
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_platform_has_samples_in_both_directions():
|
||||||
|
"""A platform with only positive samples pins half the invariant. The
|
||||||
|
no_match half is the one that catches a pattern quietly widening."""
|
||||||
|
for platform, spec in _SAMPLES.items():
|
||||||
|
assert spec["match"], f"{platform} has no match samples"
|
||||||
|
assert spec["no_match"], f"{platform} has no no_match samples"
|
||||||
@@ -1,403 +0,0 @@
|
|||||||
"""Milestone #421 — the shared predicate behind the consolidation.
|
|
||||||
|
|
||||||
`destination_for` is pure and tested without a database; `survey_layout` gets
|
|
||||||
the integration treatment because the counts are the number the apply is
|
|
||||||
checked against.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
from sqlalchemy import select
|
|
||||||
|
|
||||||
from backend.app.models import Artist, ImageRecord
|
|
||||||
from backend.app.services.library_layout import (
|
|
||||||
RESERVED_TOP_LEVEL,
|
|
||||||
_misplaced_conditions,
|
|
||||||
canonical_dir,
|
|
||||||
destination_for,
|
|
||||||
survey_layout,
|
|
||||||
)
|
|
||||||
|
|
||||||
ROOT = Path("/images")
|
|
||||||
|
|
||||||
|
|
||||||
# --- destination_for (pure) -------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def test_destination_rewrites_only_the_artist_segment():
|
|
||||||
assert destination_for(
|
|
||||||
"/images/Conto/patreon/2026-01_a_Post/x.png", ROOT, "conto"
|
|
||||||
) == Path("/images/conto/patreon/2026-01_a_Post/x.png")
|
|
||||||
|
|
||||||
|
|
||||||
def test_destination_is_identity_for_a_row_already_in_place():
|
|
||||||
p = "/images/conto/patreon/x.png"
|
|
||||||
assert destination_for(p, ROOT, "conto") == Path(p)
|
|
||||||
|
|
||||||
|
|
||||||
def test_destination_pulls_a_root_level_row_under_its_artist():
|
|
||||||
"""Diverges from canonical_subdir deliberately: the row CARRIES an
|
|
||||||
artist_id, so a file at the root is an anomaly with a known home."""
|
|
||||||
assert destination_for("/images/loose.png", ROOT, "conto") == Path(
|
|
||||||
"/images/conto/loose.png"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def test_destination_refuses_paths_outside_the_images_root():
|
|
||||||
assert destination_for("/srv/elsewhere/x.png", ROOT, "conto") is None
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize("reserved", sorted(RESERVED_TOP_LEVEL))
|
|
||||||
def test_destination_refuses_the_reserved_stores(reserved):
|
|
||||||
"""Relocating these would move the thumbnail cache, the attachment blobs
|
|
||||||
or the credential key into an artist folder."""
|
|
||||||
assert destination_for(f"/images/{reserved}/aa/x.png", ROOT, "conto") is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_destination_is_idempotent():
|
|
||||||
once = destination_for("/images/Conto/patreon/x.png", ROOT, "conto")
|
|
||||||
assert destination_for(str(once), ROOT, "conto") == once
|
|
||||||
|
|
||||||
|
|
||||||
# --- the predicate ----------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def test_canonical_prefix_carries_a_separator():
|
|
||||||
"""Without the trailing slash, artist `ara` matches every path under
|
|
||||||
`arbuzbudesh/` — one artist reads as fully placed while another's rows
|
|
||||||
are silently skipped."""
|
|
||||||
conds = _misplaced_conditions(ROOT, 1, "ara")
|
|
||||||
rendered = str(conds[-1].compile(compile_kwargs={"literal_binds": True}))
|
|
||||||
assert "/images/ara/" in rendered
|
|
||||||
|
|
||||||
|
|
||||||
# --- survey_layout (integration) --------------------------------------------
|
|
||||||
#
|
|
||||||
# Marked per-test rather than with a module-level `pytestmark`: the
|
|
||||||
# destination_for cases above are pure and belong in the fast unit lane.
|
|
||||||
|
|
||||||
|
|
||||||
def _artist(db, name, slug):
|
|
||||||
a = Artist(name=name, slug=slug)
|
|
||||||
db.add(a)
|
|
||||||
db.flush()
|
|
||||||
return a
|
|
||||||
|
|
||||||
|
|
||||||
def _image(db, path, artist=None, n=0):
|
|
||||||
rec = ImageRecord(
|
|
||||||
path=path, sha256=f"{n:064d}", size_bytes=1, mime="image/png",
|
|
||||||
width=10, height=10, origin="imported_filesystem",
|
|
||||||
integrity_status="unknown",
|
|
||||||
artist_id=artist.id if artist else None,
|
|
||||||
)
|
|
||||||
db.add(rec)
|
|
||||||
db.flush()
|
|
||||||
return rec
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_survey_splits_canonical_from_misplaced(db_sync):
|
|
||||||
conto = _artist(db_sync, "Conto", "conto")
|
|
||||||
_image(db_sync, "/images/conto/patreon/a.png", conto, 1)
|
|
||||||
_image(db_sync, "/images/Conto/patreon/b.png", conto, 2)
|
|
||||||
_image(db_sync, "/images/Conto/patreon/c.png", conto, 3)
|
|
||||||
|
|
||||||
report = survey_layout(db_sync, ROOT, check_disk=False)
|
|
||||||
|
|
||||||
assert report.misplaced_rows == 2
|
|
||||||
assert report.canonical_rows == 1
|
|
||||||
row = next(a for a in report.artists if a.slug == "conto")
|
|
||||||
assert row.stray_dirs == ["Conto"]
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_survey_does_not_confuse_a_prefix_sharing_artist(db_sync):
|
|
||||||
"""`ara` vs `arbuzbudesh` — the reason the predicate anchors on a
|
|
||||||
separator. Both are real artists in the operator's library."""
|
|
||||||
ara = _artist(db_sync, "Ara", "ara")
|
|
||||||
arbuz = _artist(db_sync, "ArbuzBudesh", "arbuzbudesh")
|
|
||||||
_image(db_sync, "/images/ara/x.png", ara, 4)
|
|
||||||
_image(db_sync, "/images/arbuzbudesh/y.png", arbuz, 5)
|
|
||||||
|
|
||||||
report = survey_layout(db_sync, ROOT, check_disk=False)
|
|
||||||
|
|
||||||
assert report.misplaced_rows == 0
|
|
||||||
assert report.canonical_rows == 2
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_survey_counts_two_rows_landing_on_one_destination(db_sync):
|
|
||||||
"""A collision is the case the apply must refuse, so the report has to
|
|
||||||
surface it rather than promise a move that cannot happen."""
|
|
||||||
sticky = _artist(db_sync, "StickySpoodge", "stickyspoodge")
|
|
||||||
_image(db_sync, "/images/StickySpoodge/p/dup.png", sticky, 6)
|
|
||||||
_image(db_sync, "/images/Stickyspoodge/p/dup.png", sticky, 7)
|
|
||||||
|
|
||||||
report = survey_layout(db_sync, ROOT, check_disk=False)
|
|
||||||
|
|
||||||
assert report.collision_count == 1
|
|
||||||
row = next(a for a in report.artists if a.slug == "stickyspoodge")
|
|
||||||
assert row.collisions == ["/images/stickyspoodge/p/dup.png"]
|
|
||||||
assert row.stray_dirs == ["StickySpoodge", "Stickyspoodge"]
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_survey_reports_unattributed_rows_without_moving_them(db_sync):
|
|
||||||
"""The 660 loose root files have no artist_id, so no predicate reaches
|
|
||||||
them. They are counted, and left for task #4247."""
|
|
||||||
_image(db_sync, "/images/orphan.png", None, 8)
|
|
||||||
|
|
||||||
report = survey_layout(db_sync, ROOT, check_disk=False)
|
|
||||||
|
|
||||||
assert report.unattributed_rows == 1
|
|
||||||
assert report.misplaced_rows == 0
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_survey_refuses_a_row_under_a_reserved_store(db_sync):
|
|
||||||
thumbs = _artist(db_sync, "Thumbsy", "thumbsy")
|
|
||||||
_image(db_sync, "/images/thumbs/aa/weird.png", thumbs, 9)
|
|
||||||
|
|
||||||
report = survey_layout(db_sync, ROOT, check_disk=False)
|
|
||||||
|
|
||||||
assert report.unmovable == 1
|
|
||||||
row = next(a for a in report.artists if a.slug == "thumbsy")
|
|
||||||
assert row.misplaced_rows == 1
|
|
||||||
assert row.collisions == []
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_survey_counts_a_missing_source_file(db_sync, tmp_path):
|
|
||||||
"""check_disk is what separates "would move" from "can move"."""
|
|
||||||
gone = _artist(db_sync, "Gone", "gone")
|
|
||||||
_image(db_sync, str(tmp_path / "Gone" / "missing.png"), gone, 10)
|
|
||||||
|
|
||||||
report = survey_layout(db_sync, tmp_path, check_disk=True)
|
|
||||||
|
|
||||||
assert report.missing_files == 1
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_survey_flags_a_destination_that_already_exists(db_sync, tmp_path):
|
|
||||||
occupied = _artist(db_sync, "Occupied", "occupied")
|
|
||||||
src = tmp_path / "Occupied" / "x.png"
|
|
||||||
src.parent.mkdir(parents=True)
|
|
||||||
src.write_bytes(b"src")
|
|
||||||
dest = canonical_dir(tmp_path, "occupied") / "x.png"
|
|
||||||
dest.parent.mkdir(parents=True)
|
|
||||||
dest.write_bytes(b"already here")
|
|
||||||
_image(db_sync, str(src), occupied, 11)
|
|
||||||
|
|
||||||
report = survey_layout(db_sync, tmp_path, check_disk=True)
|
|
||||||
|
|
||||||
assert report.collision_count == 1
|
|
||||||
assert dest.read_bytes() == b"already here" # read-only: nothing moved
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_survey_is_read_only(db_sync, tmp_path):
|
|
||||||
a = _artist(db_sync, "Reader", "reader")
|
|
||||||
src = tmp_path / "Reader" / "x.png"
|
|
||||||
src.parent.mkdir(parents=True)
|
|
||||||
src.write_bytes(b"x")
|
|
||||||
rec = _image(db_sync, str(src), a, 12)
|
|
||||||
before = rec.path
|
|
||||||
|
|
||||||
survey_layout(db_sync, tmp_path, check_disk=True)
|
|
||||||
|
|
||||||
db_sync.expire_all()
|
|
||||||
assert db_sync.get(ImageRecord, rec.id).path == before
|
|
||||||
assert src.exists()
|
|
||||||
assert db_sync.execute(
|
|
||||||
select(ImageRecord.path).where(ImageRecord.id == rec.id)
|
|
||||||
).scalar_one() == before
|
|
||||||
|
|
||||||
|
|
||||||
# --- plan / apply / revert (#4246) ------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def _staged(db, tmp_path, slug, stray, name="x.png", n=100):
|
|
||||||
"""An artist with one file sitting in `stray`'s directory."""
|
|
||||||
artist = _artist(db, slug.title(), slug)
|
|
||||||
src = tmp_path / stray / name
|
|
||||||
src.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
src.write_bytes(b"pixels")
|
|
||||||
rec = _image(db, str(src), artist, n)
|
|
||||||
return artist, rec, src
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_plan_records_where_each_file_came_from(db_sync, tmp_path):
|
|
||||||
from backend.app.services.library_layout import plan_placement
|
|
||||||
|
|
||||||
_, rec, src = _staged(db_sync, tmp_path, "conto", "Conto", n=20)
|
|
||||||
run = plan_placement(db_sync, tmp_path)
|
|
||||||
|
|
||||||
assert run.status == "ready"
|
|
||||||
assert run.planned_count == 1
|
|
||||||
assert run.moves == [{
|
|
||||||
"image_id": rec.id,
|
|
||||||
"from": str(src),
|
|
||||||
"to": str(tmp_path / "conto" / "x.png"),
|
|
||||||
}]
|
|
||||||
# Planning touches nothing.
|
|
||||||
assert src.exists()
|
|
||||||
assert db_sync.get(ImageRecord, rec.id).path == str(src)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_apply_moves_file_and_row_together(db_sync, tmp_path):
|
|
||||||
from backend.app.services.library_layout import apply_run, plan_placement
|
|
||||||
|
|
||||||
_, rec, src = _staged(db_sync, tmp_path, "conto", "Conto", n=21)
|
|
||||||
run = apply_run(db_sync, plan_placement(db_sync, tmp_path))
|
|
||||||
|
|
||||||
dest = tmp_path / "conto" / "x.png"
|
|
||||||
assert run.status == "applied"
|
|
||||||
assert run.moved_count == 1 and run.refused_count == 0
|
|
||||||
assert dest.exists() and not src.exists()
|
|
||||||
db_sync.expire_all()
|
|
||||||
assert db_sync.get(ImageRecord, rec.id).path == str(dest)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_revert_puts_it_back(db_sync, tmp_path):
|
|
||||||
"""The whole reason `from` is retained: do one artist, look, undo."""
|
|
||||||
from backend.app.services.library_layout import (
|
|
||||||
apply_run,
|
|
||||||
plan_placement,
|
|
||||||
revert_run,
|
|
||||||
)
|
|
||||||
|
|
||||||
_, rec, src = _staged(db_sync, tmp_path, "conto", "Conto", n=22)
|
|
||||||
run = revert_run(db_sync, apply_run(db_sync, plan_placement(db_sync, tmp_path)))
|
|
||||||
|
|
||||||
assert run.status == "reverted"
|
|
||||||
assert src.exists()
|
|
||||||
assert not (tmp_path / "conto" / "x.png").exists()
|
|
||||||
db_sync.expire_all()
|
|
||||||
assert db_sync.get(ImageRecord, rec.id).path == str(src)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_apply_refuses_a_row_that_moved_since_planning(db_sync, tmp_path):
|
|
||||||
"""A supersede or an earlier run can rewrite a path between plan and
|
|
||||||
apply. The stale entry is declined, not forced."""
|
|
||||||
from backend.app.services.library_layout import apply_run, plan_placement
|
|
||||||
|
|
||||||
_, rec, src = _staged(db_sync, tmp_path, "conto", "Conto", n=23)
|
|
||||||
run = plan_placement(db_sync, tmp_path)
|
|
||||||
|
|
||||||
elsewhere = tmp_path / "conto" / "already-here.png"
|
|
||||||
elsewhere.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
elsewhere.write_bytes(b"pixels")
|
|
||||||
rec.path = str(elsewhere)
|
|
||||||
db_sync.flush()
|
|
||||||
|
|
||||||
run = apply_run(db_sync, run)
|
|
||||||
|
|
||||||
assert run.moved_count == 0 and run.refused_count == 1
|
|
||||||
assert run.refusals[0]["reason"] == "row moved since planning"
|
|
||||||
assert src.exists() # untouched
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_apply_never_overwrites_an_occupied_destination(db_sync, tmp_path):
|
|
||||||
from backend.app.services.library_layout import apply_run, plan_placement
|
|
||||||
|
|
||||||
_, rec, src = _staged(db_sync, tmp_path, "conto", "Conto", n=24)
|
|
||||||
run = plan_placement(db_sync, tmp_path)
|
|
||||||
|
|
||||||
squatter = tmp_path / "conto" / "x.png"
|
|
||||||
squatter.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
squatter.write_bytes(b"someone else")
|
|
||||||
|
|
||||||
run = apply_run(db_sync, run)
|
|
||||||
|
|
||||||
assert run.refused_count == 1
|
|
||||||
assert run.refusals[0]["reason"] == "destination occupied"
|
|
||||||
assert squatter.read_bytes() == b"someone else"
|
|
||||||
db_sync.expire_all()
|
|
||||||
assert db_sync.get(ImageRecord, rec.id).path == str(src)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_apply_leaves_the_row_alone_when_the_source_is_gone(db_sync, tmp_path):
|
|
||||||
from backend.app.services.library_layout import apply_run, plan_placement
|
|
||||||
|
|
||||||
_, rec, src = _staged(db_sync, tmp_path, "conto", "Conto", n=25)
|
|
||||||
run = plan_placement(db_sync, tmp_path)
|
|
||||||
src.unlink()
|
|
||||||
|
|
||||||
run = apply_run(db_sync, run)
|
|
||||||
|
|
||||||
assert run.refusals[0]["reason"] == "source missing"
|
|
||||||
db_sync.expire_all()
|
|
||||||
# The row still points at the missing file rather than at a file that
|
|
||||||
# was never created — a broken row is recoverable, a lying one is not.
|
|
||||||
assert db_sync.get(ImageRecord, rec.id).path == str(src)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_plan_scopes_to_one_artist(db_sync, tmp_path):
|
|
||||||
"""Per-artist scope is what makes this incremental instead of one
|
|
||||||
irreversible sweep."""
|
|
||||||
from backend.app.services.library_layout import plan_placement
|
|
||||||
|
|
||||||
conto, _, _ = _staged(db_sync, tmp_path, "conto", "Conto", n=26)
|
|
||||||
_staged(db_sync, tmp_path, "maewix", "Maewix", name="y.png", n=27)
|
|
||||||
|
|
||||||
run = plan_placement(db_sync, tmp_path, artist_id=conto.id)
|
|
||||||
|
|
||||||
assert run.planned_count == 1
|
|
||||||
assert run.artist_id == conto.id
|
|
||||||
assert "Conto" in run.moves[0]["from"]
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_plan_skips_both_rows_when_two_want_one_destination(db_sync, tmp_path):
|
|
||||||
"""Which of two colliding rows 'wins' is not this sweep's call."""
|
|
||||||
from backend.app.services.library_layout import plan_placement
|
|
||||||
|
|
||||||
artist = _artist(db_sync, "Sticky", "sticky")
|
|
||||||
for stray, n in (("StickySpoodge", 28), ("Stickyspoodge", 29)):
|
|
||||||
p = tmp_path / stray / "dup.png"
|
|
||||||
p.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
p.write_bytes(b"pixels")
|
|
||||||
_image(db_sync, str(p), artist, n)
|
|
||||||
|
|
||||||
run = plan_placement(db_sync, tmp_path)
|
|
||||||
|
|
||||||
assert run.planned_count == 0
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_thumbnails_do_not_move(db_sync, tmp_path):
|
|
||||||
"""Thumbs are sha-addressed (`thumbs/<xx>/<sha>.jpg`), not path-keyed, so
|
|
||||||
a placement move must not touch them. Pinned so nobody 'fixes' it."""
|
|
||||||
from backend.app.services.library_layout import apply_run, plan_placement
|
|
||||||
|
|
||||||
artist, rec, _ = _staged(db_sync, tmp_path, "conto", "Conto", n=30)
|
|
||||||
thumb = tmp_path / "thumbs" / "ab" / "abc.jpg"
|
|
||||||
thumb.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
thumb.write_bytes(b"thumb")
|
|
||||||
rec.thumbnail_path = str(thumb)
|
|
||||||
db_sync.flush()
|
|
||||||
|
|
||||||
apply_run(db_sync, plan_placement(db_sync, tmp_path))
|
|
||||||
|
|
||||||
db_sync.expire_all()
|
|
||||||
assert thumb.exists()
|
|
||||||
assert db_sync.get(ImageRecord, rec.id).thumbnail_path == str(thumb)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
|
||||||
def test_apply_refuses_a_run_that_is_not_ready(db_sync, tmp_path):
|
|
||||||
from backend.app.services.library_layout import apply_run, plan_placement
|
|
||||||
|
|
||||||
_staged(db_sync, tmp_path, "conto", "Conto", n=31)
|
|
||||||
run = apply_run(db_sync, plan_placement(db_sync, tmp_path))
|
|
||||||
with pytest.raises(ValueError):
|
|
||||||
apply_run(db_sync, run)
|
|
||||||
@@ -1,419 +0,0 @@
|
|||||||
"""PixivClient tests — parsing + iteration against canned pages, no network.
|
|
||||||
|
|
||||||
The fixture mirrors a real `/v1/user/illusts` page (multi-page work, single
|
|
||||||
page, ugoira, sanity-limited placeholder, deleted-author ghost). HTTP is
|
|
||||||
stubbed at the requests-session seam (oauth POST + API GET), so the exact
|
|
||||||
gallery-dl-parity request profile — headers, oauth form, pagination params —
|
|
||||||
is asserted rather than assumed.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import json
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
|
|
||||||
from backend.app.services.pixiv_client import (
|
|
||||||
PIXIV_APP_HEADERS,
|
|
||||||
MediaItem,
|
|
||||||
PixivAPIError,
|
|
||||||
PixivAuthError,
|
|
||||||
PixivClient,
|
|
||||||
PixivDriftError,
|
|
||||||
rating_label,
|
|
||||||
user_id_from_url,
|
|
||||||
)
|
|
||||||
|
|
||||||
_FIXTURE = Path(__file__).parent / "fixtures" / "pixiv_user_illusts_page1.json"
|
|
||||||
|
|
||||||
|
|
||||||
class FakeResponse:
|
|
||||||
def __init__(self, status_code=200, json_data=None, headers=None):
|
|
||||||
self.status_code = status_code
|
|
||||||
self._json = json_data
|
|
||||||
self.headers = headers or {}
|
|
||||||
|
|
||||||
def json(self):
|
|
||||||
if self._json is None:
|
|
||||||
raise ValueError("no JSON")
|
|
||||||
return self._json
|
|
||||||
|
|
||||||
|
|
||||||
class FakeSession:
|
|
||||||
"""Minimal requests.Session stand-in: canned responses per (method, url
|
|
||||||
fragment), recording every call for profile assertions."""
|
|
||||||
|
|
||||||
def __init__(self):
|
|
||||||
self.headers = dict(PIXIV_APP_HEADERS)
|
|
||||||
self.responses = []
|
|
||||||
self.calls = []
|
|
||||||
|
|
||||||
def queue(self, response):
|
|
||||||
self.responses.append(response)
|
|
||||||
return self
|
|
||||||
|
|
||||||
def _next(self):
|
|
||||||
if not self.responses:
|
|
||||||
raise AssertionError("FakeSession ran out of queued responses")
|
|
||||||
return self.responses.pop(0)
|
|
||||||
|
|
||||||
def post(self, url, data=None, headers=None, timeout=None):
|
|
||||||
self.calls.append(("POST", url, data, headers))
|
|
||||||
return self._next()
|
|
||||||
|
|
||||||
def get(self, url, params=None, timeout=None, headers=None):
|
|
||||||
self.calls.append(("GET", url, params, headers))
|
|
||||||
return self._next()
|
|
||||||
|
|
||||||
|
|
||||||
def _oauth_ok():
|
|
||||||
return FakeResponse(200, {
|
|
||||||
"response": {
|
|
||||||
"access_token": "acc-token",
|
|
||||||
"expires_in": 3600,
|
|
||||||
"user": {"id": "77", "account": "operator", "name": "Op"},
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
|
||||||
def page1():
|
|
||||||
return json.loads(_FIXTURE.read_text())
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
|
||||||
def client():
|
|
||||||
# Never issues a request in pure-parsing tests.
|
|
||||||
return PixivClient("refresh-tok", session=FakeSession())
|
|
||||||
|
|
||||||
|
|
||||||
def _post_for(client, page1, work_id):
|
|
||||||
for work in page1["illusts"]:
|
|
||||||
if work["id"] == work_id:
|
|
||||||
return client._normalize(work)
|
|
||||||
raise AssertionError(f"no work {work_id} in fixture")
|
|
||||||
|
|
||||||
|
|
||||||
# -- URL → user id ----------------------------------------------------------
|
|
||||||
|
|
||||||
def test_user_id_from_url_variants():
|
|
||||||
assert user_id_from_url("https://www.pixiv.net/users/12345678") == "12345678"
|
|
||||||
assert user_id_from_url("https://www.pixiv.net/en/users/42") == "42"
|
|
||||||
assert user_id_from_url("https://pixiv.net/users/7/artworks") == "7"
|
|
||||||
assert user_id_from_url("https://www.pixiv.net/member.php?id=99") == "99"
|
|
||||||
|
|
||||||
|
|
||||||
def test_user_id_from_url_rejects_non_matches():
|
|
||||||
assert user_id_from_url("https://www.pixiv.net/artworks/111") is None
|
|
||||||
assert user_id_from_url("https://www.pixiv.net/users/notdigits") is None
|
|
||||||
assert user_id_from_url("https://example.com/users/5") is None
|
|
||||||
assert user_id_from_url("") is None
|
|
||||||
|
|
||||||
|
|
||||||
# -- normalization ------------------------------------------------------------
|
|
||||||
|
|
||||||
def test_normalize_maps_attributes(client, page1):
|
|
||||||
post = _post_for(client, page1, 111)
|
|
||||||
attrs = post["attributes"]
|
|
||||||
assert post["id"] == 111
|
|
||||||
assert attrs["title"] == "Multi Page Adventure"
|
|
||||||
assert "Two-page set." in attrs["content"]
|
|
||||||
assert attrs["published_at"] == "2026-06-20T18:00:00+09:00"
|
|
||||||
assert attrs["post_type"] == "illust"
|
|
||||||
assert post["_work"]["total_bookmarks"] == 250
|
|
||||||
|
|
||||||
|
|
||||||
def test_post_record_key_and_meta(client, page1):
|
|
||||||
post = _post_for(client, page1, 222)
|
|
||||||
assert client.post_record_key(post) == ("post:222", "222")
|
|
||||||
meta = client.post_meta(post)
|
|
||||||
assert meta["title"] == "Single Piece"
|
|
||||||
assert meta["date"] == "2026-06-18T12:30:00+09:00"
|
|
||||||
assert client.post_record_key({"id": None}) is None
|
|
||||||
|
|
||||||
|
|
||||||
# -- gating -------------------------------------------------------------------
|
|
||||||
|
|
||||||
def test_post_is_gated_limit_placeholder(client, page1):
|
|
||||||
assert client.post_is_gated(_post_for(client, page1, 444)) is True
|
|
||||||
|
|
||||||
|
|
||||||
def test_post_is_gated_deleted_author(client, page1):
|
|
||||||
assert client.post_is_gated(_post_for(client, page1, 555)) is True
|
|
||||||
|
|
||||||
|
|
||||||
def test_post_is_gated_normal_works(client, page1):
|
|
||||||
assert client.post_is_gated(_post_for(client, page1, 111)) is False
|
|
||||||
assert client.post_is_gated(_post_for(client, page1, 222)) is False
|
|
||||||
|
|
||||||
|
|
||||||
# -- extract_media --------------------------------------------------------------
|
|
||||||
|
|
||||||
def test_extract_media_multi_page(client, page1):
|
|
||||||
items = client.extract_media(_post_for(client, page1, 111), {})
|
|
||||||
assert len(items) == 2
|
|
||||||
assert all(isinstance(m, MediaItem) for m in items)
|
|
||||||
assert [m.media_id for m in items] == ["p0", "p1"]
|
|
||||||
assert items[0].url.endswith("/111_p0.png")
|
|
||||||
assert items[1].url.endswith("/111_p1.png")
|
|
||||||
# gallery-dl filename parity: {id}_{title[:50]}_{num:>02}.{extension}
|
|
||||||
assert items[0].filename == "111_Multi Page Adventure_00.png"
|
|
||||||
assert items[1].filename == "111_Multi Page Adventure_01.png"
|
|
||||||
assert all(m.post_id == "111" for m in items)
|
|
||||||
|
|
||||||
|
|
||||||
def test_extract_media_single_page(client, page1):
|
|
||||||
items = client.extract_media(_post_for(client, page1, 222), {})
|
|
||||||
assert len(items) == 1
|
|
||||||
assert items[0].media_id == "p0"
|
|
||||||
assert items[0].url.endswith("/222_p0.jpg")
|
|
||||||
assert items[0].filename == "222_Single Piece_00.jpg"
|
|
||||||
|
|
||||||
|
|
||||||
def test_extract_media_gated_yields_nothing(client, page1):
|
|
||||||
assert client.extract_media(_post_for(client, page1, 444), {}) == []
|
|
||||||
assert client.extract_media(_post_for(client, page1, 555), {}) == []
|
|
||||||
|
|
||||||
|
|
||||||
def test_extract_media_ugoira_zip_swap(client, page1, monkeypatch):
|
|
||||||
frames = [{"file": "000000.jpg", "delay": 90}, {"file": "000001.jpg", "delay": 90}]
|
|
||||||
|
|
||||||
def fake_call(endpoint, params):
|
|
||||||
assert endpoint == "/v1/ugoira/metadata"
|
|
||||||
assert params == {"illust_id": "333"}
|
|
||||||
return {"ugoira_metadata": {
|
|
||||||
"zip_urls": {"medium": (
|
|
||||||
"https://i.pximg.net/img-zip-ugoira/img/2026/06/15/09/00/00/"
|
|
||||||
"333_ugoira600x600.zip"
|
|
||||||
)},
|
|
||||||
"frames": frames,
|
|
||||||
}}
|
|
||||||
|
|
||||||
monkeypatch.setattr(client, "_call", fake_call)
|
|
||||||
post = _post_for(client, page1, 333)
|
|
||||||
items = client.extract_media(post, {})
|
|
||||||
assert len(items) == 1
|
|
||||||
assert items[0].media_id == "ugoira"
|
|
||||||
assert items[0].kind == "ugoira"
|
|
||||||
assert items[0].url.endswith("333_ugoira1920x1080.zip")
|
|
||||||
assert items[0].filename == "333_Wiggle Loop_00.zip"
|
|
||||||
# Frame delays memoized for the post record (future video conversion).
|
|
||||||
assert post["_work"]["_ugoira_frames"] == frames
|
|
||||||
|
|
||||||
|
|
||||||
def test_fetch_ugoira_frames_memoizes_and_shares_one_call(client, page1, monkeypatch):
|
|
||||||
# fetch_ugoira_frames (called by write_post_record, which the core runs
|
|
||||||
# BEFORE extract_media) populates frames; extract_media then REUSES the
|
|
||||||
# memoized metadata — exactly one /v1/ugoira/metadata call total.
|
|
||||||
frames = [{"file": "000000.jpg", "delay": 90}]
|
|
||||||
calls = []
|
|
||||||
|
|
||||||
def fake_call(endpoint, params):
|
|
||||||
calls.append(endpoint)
|
|
||||||
return {"ugoira_metadata": {
|
|
||||||
"zip_urls": {"medium": (
|
|
||||||
"https://i.pximg.net/img-zip-ugoira/x/333_ugoira600x600.zip"
|
|
||||||
)},
|
|
||||||
"frames": frames,
|
|
||||||
}}
|
|
||||||
|
|
||||||
monkeypatch.setattr(client, "_call", fake_call)
|
|
||||||
post = _post_for(client, page1, 333)
|
|
||||||
client.fetch_ugoira_frames(post)
|
|
||||||
assert post["_work"]["_ugoira_frames"] == frames
|
|
||||||
items = client.extract_media(post, {}) # reuses memoized meta
|
|
||||||
assert items[0].media_id == "ugoira"
|
|
||||||
assert calls == ["/v1/ugoira/metadata"] # ONE fetch, not two
|
|
||||||
|
|
||||||
|
|
||||||
def test_fetch_ugoira_frames_noop_for_non_ugoira(client, page1, monkeypatch):
|
|
||||||
monkeypatch.setattr(client, "_call", lambda e, p: pytest.fail("should not fetch"))
|
|
||||||
post = _post_for(client, page1, 222) # a single-page illust
|
|
||||||
client.fetch_ugoira_frames(post)
|
|
||||||
assert "_ugoira_frames" not in post["_work"]
|
|
||||||
|
|
||||||
|
|
||||||
def test_extract_media_ugoira_metadata_failure_downgrades(
|
|
||||||
client, page1, monkeypatch
|
|
||||||
):
|
|
||||||
def fake_call(endpoint, params):
|
|
||||||
raise PixivAPIError("boom", status_code=500)
|
|
||||||
|
|
||||||
monkeypatch.setattr(client, "_call", fake_call)
|
|
||||||
assert client.extract_media(_post_for(client, page1, 333), {}) == []
|
|
||||||
|
|
||||||
|
|
||||||
def test_extract_media_ugoira_auth_failure_stays_loud(
|
|
||||||
client, page1, monkeypatch
|
|
||||||
):
|
|
||||||
def fake_call(endpoint, params):
|
|
||||||
raise PixivAuthError("token dead", status_code=400)
|
|
||||||
|
|
||||||
monkeypatch.setattr(client, "_call", fake_call)
|
|
||||||
with pytest.raises(PixivAuthError):
|
|
||||||
client.extract_media(_post_for(client, page1, 333), {})
|
|
||||||
|
|
||||||
|
|
||||||
# -- iteration ------------------------------------------------------------------
|
|
||||||
|
|
||||||
def test_iter_posts_paginates_and_carries_cursor(page1, monkeypatch):
|
|
||||||
client = PixivClient("refresh-tok", session=FakeSession())
|
|
||||||
page2 = {
|
|
||||||
"illusts": [dict(page1["illusts"][1], id=666, title="Older")],
|
|
||||||
"next_url": None,
|
|
||||||
}
|
|
||||||
seen_params = []
|
|
||||||
|
|
||||||
def fake_call(endpoint, params):
|
|
||||||
assert endpoint == "/v1/user/illusts"
|
|
||||||
seen_params.append(dict(params))
|
|
||||||
return page1 if len(seen_params) == 1 else page2
|
|
||||||
|
|
||||||
monkeypatch.setattr(client, "_call", fake_call)
|
|
||||||
rows = list(client.iter_posts("99"))
|
|
||||||
|
|
||||||
assert seen_params[0] == {"user_id": "99"}
|
|
||||||
# Page 2 params come verbatim from next_url's query string.
|
|
||||||
assert seen_params[1] == {"user_id": "99", "offset": "30"}
|
|
||||||
# Page-1 posts carry cursor None; page-2 posts carry the fetching cursor.
|
|
||||||
assert [c for _, _, c in rows[:5]] == [None] * 5
|
|
||||||
assert rows[5][2] == "user_id=99&offset=30"
|
|
||||||
assert rows[5][0]["id"] == 666
|
|
||||||
|
|
||||||
|
|
||||||
def test_iter_posts_resumes_from_cursor(page1, monkeypatch):
|
|
||||||
client = PixivClient("refresh-tok", session=FakeSession())
|
|
||||||
captured = {}
|
|
||||||
|
|
||||||
def fake_call(endpoint, params):
|
|
||||||
captured["params"] = dict(params)
|
|
||||||
return {"illusts": [], "next_url": None}
|
|
||||||
|
|
||||||
monkeypatch.setattr(client, "_call", fake_call)
|
|
||||||
list(client.iter_posts("99", cursor="user_id=99&offset=60"))
|
|
||||||
assert captured["params"] == {"user_id": "99", "offset": "60"}
|
|
||||||
|
|
||||||
|
|
||||||
def test_iter_posts_rejects_non_numeric_campaign():
|
|
||||||
client = PixivClient("refresh-tok", session=FakeSession())
|
|
||||||
with pytest.raises(PixivDriftError):
|
|
||||||
next(client.iter_posts("https://www.pixiv.net/users/99"))
|
|
||||||
|
|
||||||
|
|
||||||
def test_iter_posts_drift_on_missing_illusts(monkeypatch):
|
|
||||||
client = PixivClient("refresh-tok", session=FakeSession())
|
|
||||||
monkeypatch.setattr(client, "_call", lambda e, p: {"error": None, "body": 1})
|
|
||||||
with pytest.raises(PixivDriftError):
|
|
||||||
next(client.iter_posts("99"))
|
|
||||||
|
|
||||||
|
|
||||||
# -- auth ------------------------------------------------------------------------
|
|
||||||
|
|
||||||
def test_login_sends_gallery_dl_profile_and_sets_bearer():
|
|
||||||
session = FakeSession().queue(_oauth_ok())
|
|
||||||
client = PixivClient("refresh-tok", session=session)
|
|
||||||
client._login()
|
|
||||||
|
|
||||||
method, url, data, headers = session.calls[0]
|
|
||||||
assert method == "POST"
|
|
||||||
assert url == "https://oauth.secure.pixiv.net/auth/token"
|
|
||||||
assert data["grant_type"] == "refresh_token"
|
|
||||||
assert data["refresh_token"] == "refresh-tok"
|
|
||||||
assert data["client_id"] == "MOBrBDS8blbauoSck0ZfDbtuzpyT"
|
|
||||||
assert data["get_secure_url"] == "1"
|
|
||||||
# X-Client-Time/-Hash pair: ISO + literal +00:00, md5 hex digest.
|
|
||||||
assert headers["X-Client-Time"].endswith("+00:00")
|
|
||||||
assert len(headers["X-Client-Hash"]) == 32
|
|
||||||
assert session.headers["Authorization"] == "Bearer acc-token"
|
|
||||||
# Token is fresh → a second login is a no-op (no extra POST).
|
|
||||||
client._login()
|
|
||||||
assert len(session.calls) == 1
|
|
||||||
|
|
||||||
|
|
||||||
def test_login_maps_rejection_to_auth_error():
|
|
||||||
session = FakeSession().queue(FakeResponse(400, {"has_error": True}))
|
|
||||||
client = PixivClient("refresh-tok", session=session)
|
|
||||||
with pytest.raises(PixivAuthError):
|
|
||||||
client._login()
|
|
||||||
|
|
||||||
|
|
||||||
def test_login_without_token_is_auth_error():
|
|
||||||
client = PixivClient(None, session=FakeSession())
|
|
||||||
with pytest.raises(PixivAuthError):
|
|
||||||
client._login()
|
|
||||||
|
|
||||||
|
|
||||||
def test_call_maps_rate_limit_message():
|
|
||||||
session = FakeSession().queue(_oauth_ok()).queue(FakeResponse(403, {
|
|
||||||
"error": {"message": "Rate Limit", "user_message": ""},
|
|
||||||
}))
|
|
||||||
client = PixivClient("refresh-tok", session=session)
|
|
||||||
with pytest.raises(PixivAPIError) as exc_info:
|
|
||||||
client._call("/v1/user/illusts", {"user_id": "1"})
|
|
||||||
err = exc_info.value
|
|
||||||
assert not isinstance(err, PixivAuthError)
|
|
||||||
assert err.status_code == 429
|
|
||||||
assert err.retry_after == 300.0
|
|
||||||
|
|
||||||
|
|
||||||
def test_call_maps_403_to_auth_error():
|
|
||||||
session = FakeSession().queue(_oauth_ok()).queue(FakeResponse(403, {
|
|
||||||
"error": {"message": "invalid access token", "user_message": ""},
|
|
||||||
}))
|
|
||||||
client = PixivClient("refresh-tok", session=session)
|
|
||||||
with pytest.raises(PixivAuthError):
|
|
||||||
client._call("/v1/user/illusts", {"user_id": "1"})
|
|
||||||
|
|
||||||
|
|
||||||
def test_call_maps_404_status():
|
|
||||||
session = FakeSession().queue(_oauth_ok()).queue(FakeResponse(404, {
|
|
||||||
"error": {"message": "Not Found", "user_message": ""},
|
|
||||||
}))
|
|
||||||
client = PixivClient("refresh-tok", session=session)
|
|
||||||
with pytest.raises(PixivAPIError) as exc_info:
|
|
||||||
client._call("/v1/user/illusts", {"user_id": "1"})
|
|
||||||
assert exc_info.value.status_code == 404
|
|
||||||
|
|
||||||
|
|
||||||
def test_verify_auth_reports_account():
|
|
||||||
session = FakeSession().queue(_oauth_ok())
|
|
||||||
client = PixivClient("refresh-tok", session=session)
|
|
||||||
ok, message = client.verify_auth()
|
|
||||||
assert ok is True
|
|
||||||
assert "operator" in message
|
|
||||||
|
|
||||||
|
|
||||||
def test_verify_auth_bad_token():
|
|
||||||
session = FakeSession().queue(FakeResponse(400, {"has_error": True}))
|
|
||||||
client = PixivClient("bad", session=session)
|
|
||||||
ok, message = client.verify_auth()
|
|
||||||
assert ok is False
|
|
||||||
assert "rotate" in message.lower() or "rejected" in message.lower()
|
|
||||||
|
|
||||||
|
|
||||||
def test_resolve_display_name(client, monkeypatch):
|
|
||||||
monkeypatch.setattr(
|
|
||||||
client, "_call",
|
|
||||||
lambda e, p: {"user": {"name": "Kurotsuchi Machi", "id": p["user_id"]}},
|
|
||||||
)
|
|
||||||
assert client.resolve_display_name("99") == "Kurotsuchi Machi"
|
|
||||||
|
|
||||||
|
|
||||||
def test_resolve_display_name_none_on_failure(client, monkeypatch):
|
|
||||||
def boom(endpoint, params):
|
|
||||||
raise PixivAPIError("nope", status_code=404)
|
|
||||||
monkeypatch.setattr(client, "_call", boom)
|
|
||||||
assert client.resolve_display_name("99") is None
|
|
||||||
# Empty/whitespace name → None (caller falls back to the id).
|
|
||||||
monkeypatch.setattr(client, "_call", lambda e, p: {"user": {"name": " "}})
|
|
||||||
assert client.resolve_display_name("99") is None
|
|
||||||
|
|
||||||
|
|
||||||
# -- rating ------------------------------------------------------------------------
|
|
||||||
|
|
||||||
def test_rating_label():
|
|
||||||
assert rating_label(0) == "General"
|
|
||||||
assert rating_label(1) == "R-18"
|
|
||||||
assert rating_label(2) == "R-18G"
|
|
||||||
assert rating_label(None) is None
|
|
||||||
assert rating_label(True) is None
|
|
||||||
assert rating_label(9) is None
|
|
||||||
@@ -1,309 +0,0 @@
|
|||||||
"""Unit tests for PixivDownloader — no network, no DB.
|
|
||||||
|
|
||||||
The HTTP layer is stubbed via the `session` seam (a fake session whose `.get`
|
|
||||||
returns canned streaming bytes). Media items come from the real
|
|
||||||
PixivClient.extract_media over the shared fixture page, so the downloader is
|
|
||||||
exercised against the exact shapes the client produces.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import requests
|
|
||||||
|
|
||||||
from backend.app.services.pixiv_client import MediaItem, PixivClient
|
|
||||||
from backend.app.services.pixiv_downloader import PixivDownloader
|
|
||||||
|
|
||||||
_FIXTURE = Path(__file__).parent / "fixtures" / "pixiv_user_illusts_page1.json"
|
|
||||||
|
|
||||||
# Minimal valid PNG so the file_validator passes on the happy path.
|
|
||||||
_PNG_HEAD = b"\x89PNG\r\n\x1a\n"
|
|
||||||
_PNG_TAIL = b"\x00\x00\x00\x00IEND\xaeB`\x82"
|
|
||||||
_PNG_BYTES = _PNG_HEAD + (b"\x00" * 16) + _PNG_TAIL
|
|
||||||
|
|
||||||
|
|
||||||
class _FakeResponse:
|
|
||||||
def __init__(self, payload: bytes, status_code: int = 200, headers=None):
|
|
||||||
self._payload = payload
|
|
||||||
self.status_code = status_code
|
|
||||||
self.headers = headers or {}
|
|
||||||
|
|
||||||
def raise_for_status(self):
|
|
||||||
if self.status_code >= 400:
|
|
||||||
raise requests.HTTPError(f"HTTP {self.status_code}")
|
|
||||||
return None
|
|
||||||
|
|
||||||
def iter_content(self, chunk_size=65536):
|
|
||||||
for i in range(0, len(self._payload), chunk_size):
|
|
||||||
yield self._payload[i : i + chunk_size]
|
|
||||||
|
|
||||||
|
|
||||||
class _FakeSession:
|
|
||||||
"""Records GETs and returns canned bytes (per-URL, default _PNG_BYTES)."""
|
|
||||||
|
|
||||||
def __init__(self, payloads: dict[str, bytes] | None = None):
|
|
||||||
self.payloads = payloads or {}
|
|
||||||
self.calls: list[str] = []
|
|
||||||
|
|
||||||
def get(self, url, stream=False, timeout=None, headers=None):
|
|
||||||
self.calls.append(url)
|
|
||||||
return _FakeResponse(self.payloads.get(url, _PNG_BYTES))
|
|
||||||
|
|
||||||
|
|
||||||
def _client():
|
|
||||||
# Parsing only — extract_media for non-ugoira works never issues a request.
|
|
||||||
return PixivClient("tok", session=_FakeSession())
|
|
||||||
|
|
||||||
|
|
||||||
def _post_and_media(work_id):
|
|
||||||
page = json.loads(_FIXTURE.read_text())
|
|
||||||
client = _client()
|
|
||||||
for work in page["illusts"]:
|
|
||||||
if work["id"] == work_id:
|
|
||||||
post = client._normalize(work)
|
|
||||||
return post, client.extract_media(post, {})
|
|
||||||
raise AssertionError(f"no work {work_id} in fixture")
|
|
||||||
|
|
||||||
|
|
||||||
def _post_only(work_id):
|
|
||||||
"""Normalized post WITHOUT media extraction — the write_post_record tests
|
|
||||||
don't need media, and extracting an ugoira would hit the API seam."""
|
|
||||||
page = json.loads(_FIXTURE.read_text())
|
|
||||||
for work in page["illusts"]:
|
|
||||||
if work["id"] == work_id:
|
|
||||||
return _client()._normalize(work)
|
|
||||||
raise AssertionError(f"no work {work_id} in fixture")
|
|
||||||
|
|
||||||
|
|
||||||
def _downloader(tmp_path, session=None):
|
|
||||||
# validate=False: the stub payload is PNG bytes regardless of the target
|
|
||||||
# extension, so real validation would quarantine the .jpg cases. The
|
|
||||||
# validate/quarantine plumbing is BaseNativeDownloader's, covered by the
|
|
||||||
# Patreon downloader tests.
|
|
||||||
return PixivDownloader(
|
|
||||||
tmp_path, validate=False,
|
|
||||||
session=session if session is not None else _FakeSession(),
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def test_download_post_writes_gallery_dl_layout(tmp_path):
|
|
||||||
post, media = _post_and_media(111)
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcomes = dl.download_post(post, media, "artist-a")
|
|
||||||
|
|
||||||
assert [o.status for o in outcomes] == ["downloaded", "downloaded"]
|
|
||||||
flat = tmp_path / "artist-a" / "pixiv" / "pixiv"
|
|
||||||
p0 = flat / "111_Multi Page Adventure_00.png"
|
|
||||||
p1 = flat / "111_Multi Page Adventure_01.png"
|
|
||||||
assert p0.is_file()
|
|
||||||
assert p1.is_file()
|
|
||||||
# No per-post directory — pixiv's layout is flat (gallery-dl parity).
|
|
||||||
assert {p.parent for p in (p0, p1)} == {flat}
|
|
||||||
|
|
||||||
|
|
||||||
def test_download_post_writes_minimal_sidecar(tmp_path):
|
|
||||||
post, media = _post_and_media(222)
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcomes = dl.download_post(post, media, "artist-a")
|
|
||||||
|
|
||||||
assert outcomes[0].status == "downloaded"
|
|
||||||
sidecar = outcomes[0].path.with_suffix(".json")
|
|
||||||
data = json.loads(sidecar.read_text())
|
|
||||||
# Post-first: image identity ONLY — the body lives in the post record.
|
|
||||||
assert data == {
|
|
||||||
"category": "pixiv",
|
|
||||||
"id": "222",
|
|
||||||
"source_url": media[0].url,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def test_download_post_skips_seen(tmp_path):
|
|
||||||
post, media = _post_and_media(222)
|
|
||||||
session = _FakeSession()
|
|
||||||
dl = _downloader(tmp_path, session=session)
|
|
||||||
outcomes = dl.download_post(post, media, "artist-a", is_seen=lambda m: True)
|
|
||||||
assert [o.status for o in outcomes] == ["skipped_seen"]
|
|
||||||
assert session.calls == []
|
|
||||||
|
|
||||||
|
|
||||||
def test_download_post_skips_on_disk(tmp_path):
|
|
||||||
post, media = _post_and_media(222)
|
|
||||||
flat = tmp_path / "artist-a" / "pixiv" / "pixiv"
|
|
||||||
flat.mkdir(parents=True)
|
|
||||||
existing = flat / "222_Single Piece_00.jpg"
|
|
||||||
existing.write_bytes(b"already here")
|
|
||||||
|
|
||||||
session = _FakeSession()
|
|
||||||
dl = _downloader(tmp_path, session=session)
|
|
||||||
outcomes = dl.download_post(post, media, "artist-a")
|
|
||||||
assert [o.status for o in outcomes] == ["skipped_disk"]
|
|
||||||
assert outcomes[0].path == existing
|
|
||||||
assert session.calls == []
|
|
||||||
assert existing.read_bytes() == b"already here"
|
|
||||||
|
|
||||||
|
|
||||||
def test_recapture_surfaces_on_disk_path_for_seen_media(tmp_path):
|
|
||||||
post, media = _post_and_media(222)
|
|
||||||
flat = tmp_path / "artist-a" / "pixiv" / "pixiv"
|
|
||||||
flat.mkdir(parents=True)
|
|
||||||
(flat / "222_Single Piece_00.jpg").write_bytes(b"kept")
|
|
||||||
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcomes = dl.download_post(
|
|
||||||
post, media, "artist-a", is_seen=lambda m: True, recapture=True,
|
|
||||||
)
|
|
||||||
# On disk → surfaced with its path (relink channel); seen-but-missing
|
|
||||||
# stays skipped_seen (recovery's job, not recapture's).
|
|
||||||
assert outcomes[0].status == "skipped_disk"
|
|
||||||
assert outcomes[0].path is not None
|
|
||||||
|
|
||||||
|
|
||||||
def test_recapture_does_not_redownload_missing_seen_media(tmp_path):
|
|
||||||
post, media = _post_and_media(222)
|
|
||||||
session = _FakeSession()
|
|
||||||
dl = _downloader(tmp_path, session=session)
|
|
||||||
outcomes = dl.download_post(
|
|
||||||
post, media, "artist-a", is_seen=lambda m: True, recapture=True,
|
|
||||||
)
|
|
||||||
assert [o.status for o in outcomes] == ["skipped_seen"]
|
|
||||||
assert session.calls == []
|
|
||||||
|
|
||||||
|
|
||||||
def test_download_post_honours_should_stop(tmp_path):
|
|
||||||
post, media = _post_and_media(111)
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcomes = dl.download_post(post, media, "artist-a", should_stop=lambda: True)
|
|
||||||
assert outcomes == []
|
|
||||||
|
|
||||||
|
|
||||||
def test_filename_matches_gallery_dl_linux_restrict(tmp_path):
|
|
||||||
# gallery-dl on Linux replaces ONLY "/" and deletes control chars — the
|
|
||||||
# Windows-forbidden set (<>:"|?*) stays RAW in the on-disk name. Matching
|
|
||||||
# that byte-for-byte is what lets the tier-2 disk-skip recognize
|
|
||||||
# pre-cutover files; a stricter sanitizer would re-download them.
|
|
||||||
post, _ = _post_and_media(222)
|
|
||||||
weird = MediaItem(
|
|
||||||
url="https://i.pximg.net/img-original/img/2026/06/18/12/30/00/222_p0.jpg",
|
|
||||||
filename='222_What? A "Title"/Slash\x0900.jpg', # ? " stay, / → _, \t dropped
|
|
||||||
kind="image",
|
|
||||||
filehash=None,
|
|
||||||
post_id="222",
|
|
||||||
media_id="p0",
|
|
||||||
)
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcomes = dl.download_post(post, [weird], "artist-a")
|
|
||||||
assert outcomes[0].status == "downloaded"
|
|
||||||
assert outcomes[0].path.name == '222_What? A "Title"_Slash00.jpg'
|
|
||||||
|
|
||||||
|
|
||||||
def test_gdl_clean_filename_parity():
|
|
||||||
from backend.app.services.pixiv_downloader import gdl_clean_filename
|
|
||||||
# Only "/" is replaced; the Windows-forbidden set is untouched.
|
|
||||||
assert gdl_clean_filename('a<>:"|?*b.png') == 'a<>:"|?*b.png'
|
|
||||||
assert gdl_clean_filename("a/b/c.png") == "a_b_c.png"
|
|
||||||
# Control chars (newline, tab, DEL) are deleted, not replaced.
|
|
||||||
assert gdl_clean_filename("a\nb\tc\x7fd.png") == "abcd.png"
|
|
||||||
# Trailing dots/spaces are NOT stripped on Linux.
|
|
||||||
assert gdl_clean_filename("title . .png") == "title . .png"
|
|
||||||
|
|
||||||
|
|
||||||
def test_own_session_carries_app_referer(tmp_path):
|
|
||||||
# Built session (no injection) must carry the app profile — i.pximg.net
|
|
||||||
# 403s GETs without the app-api Referer.
|
|
||||||
dl = PixivDownloader(tmp_path)
|
|
||||||
assert dl.session.headers["Referer"] == "https://app-api.pixiv.net/"
|
|
||||||
assert dl.session.headers["App-OS"] == "ios"
|
|
||||||
|
|
||||||
|
|
||||||
def test_write_post_record_enriched(tmp_path):
|
|
||||||
post = _post_only(111)
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcome = dl.write_post_record(post, "artist-a")
|
|
||||||
|
|
||||||
assert outcome.path == (
|
|
||||||
tmp_path / "artist-a" / "pixiv" / "pixiv" / "_post_111.json"
|
|
||||||
)
|
|
||||||
assert outcome.title == "Multi Page Adventure"
|
|
||||||
assert outcome.post_type == "illust"
|
|
||||||
assert outcome.body_chars == len(post["attributes"]["content"])
|
|
||||||
|
|
||||||
data = json.loads(outcome.path.read_text())
|
|
||||||
assert data["category"] == "pixiv"
|
|
||||||
assert data["id"] == "111"
|
|
||||||
assert data["published_at"] == "2026-06-20T18:00:00+09:00"
|
|
||||||
assert data["rating"] == "General"
|
|
||||||
assert data["page_count"] == 2
|
|
||||||
assert data["total_bookmarks"] == 250
|
|
||||||
assert data["total_view"] == 1000
|
|
||||||
assert data["illust_ai_type"] == 1
|
|
||||||
assert data["series"] == {"id": 4242, "title": "Adventure Series"}
|
|
||||||
assert data["tags"] == [
|
|
||||||
{"name": "オリジナル", "translated_name": "original"},
|
|
||||||
{"name": "女の子", "translated_name": "girl"},
|
|
||||||
]
|
|
||||||
assert data["user"] == {"id": 99, "account": "exartist", "name": "Example Artist"}
|
|
||||||
# JP tags stay human-readable on disk (ensure_ascii=False).
|
|
||||||
assert "オリジナル" in outcome.path.read_text()
|
|
||||||
|
|
||||||
|
|
||||||
def test_write_post_record_rating_r18(tmp_path):
|
|
||||||
post = _post_only(222)
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcome = dl.write_post_record(post, "artist-a")
|
|
||||||
data = json.loads(outcome.path.read_text())
|
|
||||||
assert data["rating"] == "R-18"
|
|
||||||
assert data["is_bookmarked"] is True
|
|
||||||
|
|
||||||
|
|
||||||
def test_write_post_record_includes_ugoira_frames(tmp_path):
|
|
||||||
post = _post_only(333)
|
|
||||||
frames = [{"file": "000000.jpg", "delay": 90}]
|
|
||||||
post["_work"]["_ugoira_frames"] = frames
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcome = dl.write_post_record(post, "artist-a")
|
|
||||||
data = json.loads(outcome.path.read_text())
|
|
||||||
assert data["type"] == "ugoira"
|
|
||||||
assert data["ugoira_frames"] == frames
|
|
||||||
|
|
||||||
|
|
||||||
def test_write_post_record_fetches_ugoira_frames_when_absent(tmp_path):
|
|
||||||
# The core writes the post record BEFORE extract_media, so frames aren't
|
|
||||||
# memoized yet — the injected fetcher must populate them so the record keeps
|
|
||||||
# the timings (regression: they were silently always empty).
|
|
||||||
post = _post_only(333)
|
|
||||||
assert "_ugoira_frames" not in post["_work"]
|
|
||||||
frames = [{"file": "000000.jpg", "delay": 120}]
|
|
||||||
calls = []
|
|
||||||
|
|
||||||
def fetcher(p):
|
|
||||||
calls.append(p)
|
|
||||||
p["_work"]["_ugoira_frames"] = frames
|
|
||||||
|
|
||||||
dl = PixivDownloader(
|
|
||||||
tmp_path, validate=False, session=_FakeSession(),
|
|
||||||
ugoira_frames_fetcher=fetcher,
|
|
||||||
)
|
|
||||||
outcome = dl.write_post_record(post, "artist-a")
|
|
||||||
data = json.loads(outcome.path.read_text())
|
|
||||||
assert data["ugoira_frames"] == frames
|
|
||||||
assert len(calls) == 1
|
|
||||||
|
|
||||||
|
|
||||||
def test_write_post_record_non_ugoira_does_not_call_fetcher(tmp_path):
|
|
||||||
post = _post_only(111) # a plain multi-page illust
|
|
||||||
calls = []
|
|
||||||
dl = PixivDownloader(
|
|
||||||
tmp_path, validate=False, session=_FakeSession(),
|
|
||||||
ugoira_frames_fetcher=lambda p: calls.append(p),
|
|
||||||
)
|
|
||||||
dl.write_post_record(post, "artist-a")
|
|
||||||
assert calls == []
|
|
||||||
|
|
||||||
|
|
||||||
def test_write_post_record_without_id(tmp_path):
|
|
||||||
dl = _downloader(tmp_path)
|
|
||||||
outcome = dl.write_post_record({"id": None, "attributes": {}}, "artist-a")
|
|
||||||
assert outcome.path is None
|
|
||||||
assert outcome.body_chars == 0
|
|
||||||
@@ -1,305 +0,0 @@
|
|||||||
"""PixivIngester tests — the adapter's wiring over the shared core.
|
|
||||||
|
|
||||||
The core's walk/skip/cursor/budget behavior is exercised exhaustively by
|
|
||||||
test_patreon_ingester / test_subscribestar_native; these cover what is
|
|
||||||
pixiv-SPECIFIC: the synthesized ledger key, the real Postgres pixiv ledgers
|
|
||||||
(the upserts' constraint names must match migration 0076), the failure
|
|
||||||
mapping including the rate-limit retry_after carry, and the #862 body-canary
|
|
||||||
opt-out (caption-less pixiv feeds must not fail API_DRIFT).
|
|
||||||
"""
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
from sqlalchemy import func, select
|
|
||||||
from sqlalchemy.orm import sessionmaker
|
|
||||||
|
|
||||||
from backend.app.models import Artist, PixivFailedMedia, PixivSeenMedia, Source
|
|
||||||
from backend.app.services.gallery_dl import ErrorType
|
|
||||||
from backend.app.services.ingest_core import _CANARY_MIN_SAMPLE
|
|
||||||
from backend.app.services.native_ingest_common import MediaOutcome, PostRecordOutcome
|
|
||||||
from backend.app.services.pixiv_client import (
|
|
||||||
MediaItem,
|
|
||||||
PixivAPIError,
|
|
||||||
PixivAuthError,
|
|
||||||
PixivDriftError,
|
|
||||||
)
|
|
||||||
from backend.app.services.pixiv_ingester import PixivIngester, _ledger_key
|
|
||||||
|
|
||||||
pytestmark = pytest.mark.integration
|
|
||||||
|
|
||||||
|
|
||||||
# --- fakes ----------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def _media(post_id, num, *, filehash=None, media_id=None):
|
|
||||||
mid = media_id if media_id is not None else f"p{num}"
|
|
||||||
return MediaItem(
|
|
||||||
url=f"https://i.pximg.net/img-original/img/2026/07/01/00/00/00/{post_id}_{mid}.png",
|
|
||||||
filename=f"{post_id}_Work_{num:02d}.png",
|
|
||||||
kind="image",
|
|
||||||
filehash=filehash,
|
|
||||||
post_id=str(post_id),
|
|
||||||
media_id=mid,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
class _FakeClient:
|
|
||||||
"""Stub PixivClient. `pages` is a list of (page_cursor, [posts]); each post
|
|
||||||
is (post_id, [MediaItem]). `raise_exc` trips a client-level failure;
|
|
||||||
`captions` toggles per-post body text (the canary input)."""
|
|
||||||
|
|
||||||
def __init__(self, pages, raise_exc=None, captions=True):
|
|
||||||
self._pages = pages
|
|
||||||
self._raise_exc = raise_exc
|
|
||||||
self._captions = captions
|
|
||||||
|
|
||||||
def iter_posts(self, campaign_id, cursor=None):
|
|
||||||
if self._raise_exc is not None:
|
|
||||||
raise self._raise_exc
|
|
||||||
for page_cursor, posts in self._pages:
|
|
||||||
for post_id, media in posts:
|
|
||||||
caption = f"<p>caption {post_id}</p>" if self._captions else ""
|
|
||||||
yield (
|
|
||||||
{
|
|
||||||
"id": post_id,
|
|
||||||
"_media": media,
|
|
||||||
"attributes": {
|
|
||||||
"title": f"Work {post_id}",
|
|
||||||
"post_type": "illust",
|
|
||||||
"content": caption,
|
|
||||||
"published_at": "2026-07-01T00:00:00+09:00",
|
|
||||||
},
|
|
||||||
"_work": {},
|
|
||||||
},
|
|
||||||
{},
|
|
||||||
page_cursor,
|
|
||||||
)
|
|
||||||
|
|
||||||
def extract_media(self, post, included_index):
|
|
||||||
return post["_media"]
|
|
||||||
|
|
||||||
def post_meta(self, post):
|
|
||||||
return {"title": post.get("id"), "date": None}
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def post_is_gated(post):
|
|
||||||
return False
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def post_record_key(post):
|
|
||||||
pid = str(post.get("id") or "")
|
|
||||||
return (f"post:{pid}", pid) if pid else None
|
|
||||||
|
|
||||||
|
|
||||||
class _FakeDownloader:
|
|
||||||
"""Stub PixivDownloader. Honors the injected is_seen; media whose ledger
|
|
||||||
key is in `error` report a download failure."""
|
|
||||||
|
|
||||||
def __init__(self, tmp_path, error=None):
|
|
||||||
self.tmp_path = tmp_path
|
|
||||||
self.error = set(error or ())
|
|
||||||
self.download_calls = 0
|
|
||||||
|
|
||||||
def download_post(self, post, media_items, artist_slug, *, is_seen,
|
|
||||||
should_stop=lambda: False, recapture=False):
|
|
||||||
outcomes = []
|
|
||||||
for m in media_items:
|
|
||||||
if should_stop():
|
|
||||||
break
|
|
||||||
if is_seen(m) and not recapture:
|
|
||||||
outcomes.append(
|
|
||||||
MediaOutcome(media=m, status="skipped_seen", path=None, error=None)
|
|
||||||
)
|
|
||||||
elif _ledger_key(m) in self.error:
|
|
||||||
self.download_calls += 1
|
|
||||||
outcomes.append(
|
|
||||||
MediaOutcome(media=m, status="error", path=None, error="403 pximg")
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
self.download_calls += 1
|
|
||||||
p = self.tmp_path / m.filename
|
|
||||||
p.write_bytes(b"x")
|
|
||||||
outcomes.append(
|
|
||||||
MediaOutcome(media=m, status="downloaded", path=p, error=None)
|
|
||||||
)
|
|
||||||
return outcomes
|
|
||||||
|
|
||||||
def write_post_record(self, post, artist_slug):
|
|
||||||
p = self.tmp_path / f"_post_{post.get('id')}.json"
|
|
||||||
p.write_text("{}")
|
|
||||||
attrs = post.get("attributes") or {}
|
|
||||||
body = attrs.get("content")
|
|
||||||
return PostRecordOutcome(
|
|
||||||
path=p,
|
|
||||||
post_type=attrs.get("post_type"),
|
|
||||||
title=attrs.get("title"),
|
|
||||||
body_chars=len(body) if isinstance(body, str) else 0,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
|
||||||
async def source_id(db):
|
|
||||||
artist = Artist(name="Pixland", slug="pixland")
|
|
||||||
db.add(artist)
|
|
||||||
await db.flush()
|
|
||||||
source = Source(
|
|
||||||
artist_id=artist.id, platform="pixiv",
|
|
||||||
url="https://www.pixiv.net/users/99", enabled=True, config_overrides={},
|
|
||||||
)
|
|
||||||
db.add(source)
|
|
||||||
await db.commit()
|
|
||||||
return source.id
|
|
||||||
|
|
||||||
|
|
||||||
def _ingester(sync_engine, tmp_path, client, downloader):
|
|
||||||
factory = sessionmaker(sync_engine, expire_on_commit=False)
|
|
||||||
return PixivIngester(
|
|
||||||
images_root=tmp_path, cookies_path=None, session_factory=factory,
|
|
||||||
client=client, downloader=downloader,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def _run(ing, source_id, mode="tick"):
|
|
||||||
return ing.run(
|
|
||||||
source_id=source_id, campaign_id="99", artist_slug="pixland",
|
|
||||||
url="https://www.pixiv.net/users/99", mode=mode,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def _seen_keys(sync_engine, source_id):
|
|
||||||
factory = sessionmaker(sync_engine, expire_on_commit=False)
|
|
||||||
with factory() as s:
|
|
||||||
return set(s.execute(
|
|
||||||
select(PixivSeenMedia.filehash).where(
|
|
||||||
PixivSeenMedia.source_id == source_id
|
|
||||||
)
|
|
||||||
).scalars().all())
|
|
||||||
|
|
||||||
|
|
||||||
# --- ledger key -------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def test_ledger_key_shapes():
|
|
||||||
assert _ledger_key(_media(111, 0)) == "111:p0"
|
|
||||||
assert _ledger_key(_media(111, 4)) == "111:p4"
|
|
||||||
assert _ledger_key(_media(333, 0, media_id="ugoira")) == "333:ugoira"
|
|
||||||
assert _ledger_key(_media(1, 0, filehash="f" * 32)) == "f" * 32
|
|
||||||
long_key = _ledger_key(_media("9" * 200, 0))
|
|
||||||
assert len(long_key) <= 128
|
|
||||||
|
|
||||||
|
|
||||||
# --- walk + ledgers ---------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_tick_downloads_and_marks_seen(source_id, sync_engine, tmp_path):
|
|
||||||
m0, m1 = _media(111, 0), _media(111, 1)
|
|
||||||
client = _FakeClient([(None, [(111, [m0, m1])])])
|
|
||||||
downloader = _FakeDownloader(tmp_path)
|
|
||||||
ing = _ingester(sync_engine, tmp_path, client, downloader)
|
|
||||||
|
|
||||||
result = _run(ing, source_id)
|
|
||||||
|
|
||||||
assert result.success is True
|
|
||||||
assert result.files_downloaded == 2
|
|
||||||
assert len(result.post_record_paths) == 1
|
|
||||||
# 2 media keys + the synthetic post key, in the pixiv-shaped key format.
|
|
||||||
assert _seen_keys(sync_engine, source_id) == {"111:p0", "111:p1", "post:111"}
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_tick_skips_seen_via_ledger(source_id, sync_engine, tmp_path):
|
|
||||||
m0 = _media(111, 0)
|
|
||||||
factory = sessionmaker(sync_engine, expire_on_commit=False)
|
|
||||||
with factory() as s:
|
|
||||||
s.add(PixivSeenMedia(
|
|
||||||
source_id=source_id, filehash=_ledger_key(m0), post_id="111",
|
|
||||||
))
|
|
||||||
s.commit()
|
|
||||||
|
|
||||||
client = _FakeClient([(None, [(111, [m0])])])
|
|
||||||
downloader = _FakeDownloader(tmp_path)
|
|
||||||
ing = _ingester(sync_engine, tmp_path, client, downloader)
|
|
||||||
|
|
||||||
result = _run(ing, source_id)
|
|
||||||
assert result.files_downloaded == 0
|
|
||||||
assert downloader.download_calls == 0
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_failed_media_lands_in_dead_letter_ledger(
|
|
||||||
source_id, sync_engine, tmp_path
|
|
||||||
):
|
|
||||||
m0 = _media(111, 0)
|
|
||||||
client = _FakeClient([(None, [(111, [m0])])])
|
|
||||||
downloader = _FakeDownloader(tmp_path, error={_ledger_key(m0)})
|
|
||||||
ing = _ingester(sync_engine, tmp_path, client, downloader)
|
|
||||||
|
|
||||||
result = _run(ing, source_id)
|
|
||||||
assert result.run_stats["per_item_failures"] == 1
|
|
||||||
|
|
||||||
factory = sessionmaker(sync_engine, expire_on_commit=False)
|
|
||||||
with factory() as s:
|
|
||||||
row = s.execute(
|
|
||||||
select(PixivFailedMedia).where(
|
|
||||||
PixivFailedMedia.source_id == source_id
|
|
||||||
)
|
|
||||||
).scalar_one()
|
|
||||||
assert row.filehash == "111:p0"
|
|
||||||
assert row.attempts == 1
|
|
||||||
assert "403" in row.last_error
|
|
||||||
|
|
||||||
|
|
||||||
# --- failure mapping ----------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_auth_error_maps_to_auth_error(source_id, sync_engine, tmp_path):
|
|
||||||
client = _FakeClient([], raise_exc=PixivAuthError(
|
|
||||||
"rotate the token", status_code=400,
|
|
||||||
))
|
|
||||||
ing = _ingester(sync_engine, tmp_path, client, _FakeDownloader(tmp_path))
|
|
||||||
result = _run(ing, source_id)
|
|
||||||
assert result.success is False
|
|
||||||
assert result.error_type == ErrorType.AUTH_ERROR
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_drift_error_maps_to_api_drift(source_id, sync_engine, tmp_path):
|
|
||||||
client = _FakeClient([], raise_exc=PixivDriftError("no illusts list"))
|
|
||||||
ing = _ingester(sync_engine, tmp_path, client, _FakeDownloader(tmp_path))
|
|
||||||
result = _run(ing, source_id)
|
|
||||||
assert result.error_type == ErrorType.API_DRIFT
|
|
||||||
assert "Pixiv app API changed" in result.error_message
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_rate_limit_carries_retry_after(source_id, sync_engine, tmp_path):
|
|
||||||
client = _FakeClient([], raise_exc=PixivAPIError(
|
|
||||||
"rate limited", status_code=429, retry_after=300.0,
|
|
||||||
))
|
|
||||||
ing = _ingester(sync_engine, tmp_path, client, _FakeDownloader(tmp_path))
|
|
||||||
result = _run(ing, source_id)
|
|
||||||
assert result.error_type == ErrorType.RATE_LIMITED
|
|
||||||
assert result.retry_after_seconds == 300.0
|
|
||||||
|
|
||||||
|
|
||||||
# --- body canary opt-out --------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_captionless_feed_does_not_trip_body_canary(
|
|
||||||
source_id, sync_engine, tmp_path
|
|
||||||
):
|
|
||||||
"""#862 canary opt-out: a pixiv backfill recording ≥ the canary sample of
|
|
||||||
caption-less works is NORMAL (many artists never write captions) — it must
|
|
||||||
complete, not fail API_DRIFT the way a zero-body Patreon walk does."""
|
|
||||||
n = _CANARY_MIN_SAMPLE + 5
|
|
||||||
posts = [(1000 + i, [_media(1000 + i, 0)]) for i in range(n)]
|
|
||||||
client = _FakeClient([(None, posts)], captions=False)
|
|
||||||
downloader = _FakeDownloader(tmp_path)
|
|
||||||
ing = _ingester(sync_engine, tmp_path, client, downloader)
|
|
||||||
|
|
||||||
result = _run(ing, source_id, mode="backfill")
|
|
||||||
assert result.success is True
|
|
||||||
assert result.error_type is None
|
|
||||||
assert result.files_downloaded == n
|
|
||||||
@@ -26,6 +26,40 @@ def test_pixiv_is_retired():
|
|||||||
assert "pixiv" not in known_platform_keys()
|
assert "pixiv" not in known_platform_keys()
|
||||||
|
|
||||||
|
|
||||||
|
def test_pixiv_code_and_tables_are_gone():
|
||||||
|
"""Milestone #406 phase 2 (step 7). Phase 1 made pixiv unreachable; this is
|
||||||
|
the guard that it stays deleted rather than drifting back in.
|
||||||
|
|
||||||
|
Both halves assert absence from a DATA STRUCTURE — the module table and the
|
||||||
|
declarative metadata — not from prose. Snippet #3352's trap is an absence
|
||||||
|
check against a comment, which passes the moment someone rewords the
|
||||||
|
comment; a re-added module or a re-declared table cannot hide from these.
|
||||||
|
|
||||||
|
DeviantArt is the reason this exists: #3069 retired it in code on
|
||||||
|
2026-08-27 and its credential row was still sitting in the database seven
|
||||||
|
weeks later (#3980). A retirement nothing asserts is a retirement that
|
||||||
|
half-happens.
|
||||||
|
"""
|
||||||
|
import importlib
|
||||||
|
|
||||||
|
from backend.app.models.base import Base
|
||||||
|
|
||||||
|
for module in (
|
||||||
|
"backend.app.services.pixiv_client",
|
||||||
|
"backend.app.services.pixiv_downloader",
|
||||||
|
"backend.app.services.pixiv_ingester",
|
||||||
|
"backend.app.services.platforms.pixiv",
|
||||||
|
"backend.app.models.pixiv_seen_media",
|
||||||
|
"backend.app.models.pixiv_failed_media",
|
||||||
|
):
|
||||||
|
with pytest.raises(ModuleNotFoundError):
|
||||||
|
importlib.import_module(module)
|
||||||
|
|
||||||
|
# Alembic 0102 drops these; nothing may re-declare them.
|
||||||
|
assert "pixiv_seen_media" not in Base.metadata.tables
|
||||||
|
assert "pixiv_failed_media" not in Base.metadata.tables
|
||||||
|
|
||||||
|
|
||||||
def test_fanbox_not_in_registry():
|
def test_fanbox_not_in_registry():
|
||||||
# Sanity check — FC-3a added 'fanbox' by mistake; it's not a GS platform.
|
# Sanity check — FC-3a added 'fanbox' by mistake; it's not a GS platform.
|
||||||
assert "fanbox" not in PLATFORMS
|
assert "fanbox" not in PLATFORMS
|
||||||
|
|||||||
@@ -682,3 +682,83 @@ async def test_new_disabled_source_skips_backfill(db):
|
|||||||
enabled=False,
|
enabled=False,
|
||||||
)
|
)
|
||||||
assert rec.backfill_runs_remaining == 0
|
assert rec.backfill_runs_remaining == 0
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_a_disabled_source_is_not_failing(db):
|
||||||
|
"""#4279: "stopped because you no longer subscribe" is not "failing".
|
||||||
|
|
||||||
|
Ebi77 was stopped by the membership sweep as `former_patron`, then a deep
|
||||||
|
scan armed on it got stranded by the recovery sweep. The banner showed it
|
||||||
|
for six days with no action available: a disabled source is never
|
||||||
|
scheduled (so no run clears the count), `update` only clears on an
|
||||||
|
explicit disable (it was already disabled), and Retry routes to /check,
|
||||||
|
which refuses a disabled source.
|
||||||
|
"""
|
||||||
|
artist = await _artist(db)
|
||||||
|
svc = SourceService(db)
|
||||||
|
rec = await svc.create(
|
||||||
|
artist_id=artist.id, platform="patreon",
|
||||||
|
url="https://patreon.com/stopped",
|
||||||
|
)
|
||||||
|
source = (await db.execute(
|
||||||
|
select(Source).where(Source.id == rec.id)
|
||||||
|
)).scalar_one()
|
||||||
|
source.enabled = False
|
||||||
|
source.consecutive_failures = 1
|
||||||
|
source.last_error = "stranded by recovery sweep (no terminal status after time_limit)"
|
||||||
|
await db.commit()
|
||||||
|
|
||||||
|
assert [r.id for r in await svc.list(failing=True)] == []
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_an_enabled_source_that_errors_is_still_failing(db):
|
||||||
|
"""The other half — folding `enabled` in must not hide a real failure on
|
||||||
|
a live source."""
|
||||||
|
artist = await _artist(db)
|
||||||
|
svc = SourceService(db)
|
||||||
|
rec = await svc.create(
|
||||||
|
artist_id=artist.id, platform="patreon", url="https://patreon.com/live",
|
||||||
|
)
|
||||||
|
source = (await db.execute(
|
||||||
|
select(Source).where(Source.id == rec.id)
|
||||||
|
)).scalar_one()
|
||||||
|
source.enabled = True
|
||||||
|
source.consecutive_failures = 2
|
||||||
|
source.last_error = "auth failed"
|
||||||
|
await db.commit()
|
||||||
|
|
||||||
|
assert [r.id for r in await svc.list(failing=True)] == [rec.id]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_the_failing_list_and_the_ribbon_count_agree(db):
|
||||||
|
"""The two callers of `failing_sources_clause` disagreed before #4279:
|
||||||
|
the scheduler's count paired it with `enabled.is_(True)`, the list did
|
||||||
|
not, so one counted Ebi77 and the other did not. The predicate owns the
|
||||||
|
whole definition now — assert the two surfaces match rather than trusting
|
||||||
|
that they were both updated."""
|
||||||
|
from backend.app.services.scheduler_service import scheduler_status
|
||||||
|
|
||||||
|
artist = await _artist(db)
|
||||||
|
svc = SourceService(db)
|
||||||
|
for url, enabled, fails in (
|
||||||
|
("https://patreon.com/one", True, 3),
|
||||||
|
("https://patreon.com/two", False, 1),
|
||||||
|
("https://patreon.com/three", True, 0),
|
||||||
|
):
|
||||||
|
rec = await svc.create(
|
||||||
|
artist_id=artist.id, platform="patreon", url=url,
|
||||||
|
)
|
||||||
|
s = (await db.execute(
|
||||||
|
select(Source).where(Source.id == rec.id)
|
||||||
|
)).scalar_one()
|
||||||
|
s.enabled = enabled
|
||||||
|
s.consecutive_failures = fails
|
||||||
|
await db.commit()
|
||||||
|
|
||||||
|
listed = len(await svc.list(failing=True))
|
||||||
|
counts = await scheduler_status(db)
|
||||||
|
assert listed == 1
|
||||||
|
assert counts["failing_sources"] == listed
|
||||||
|
|||||||
Reference in New Issue
Block a user