revert: remove the placement reconciler — it manufactured the problem it solved
CI / lint (push) Successful in 2s
CI / extension-version (push) Successful in 3s
Build images / sign-extension (push) Successful in 3s
Build images / build-agent (push) Successful in 6s
CI / frontend-build (push) Successful in 26s
CI / backend-lint-and-test (push) Successful in 33s
Build images / build-web (push) Successful in 59s
Build images / smoke-web (push) Skipped
Build images / build-ml (push) Successful in 1m55s
Build images / promote (push) Skipped
CI / integration (push) Successful in 2m22s

Milestone #421 built a sweep that compared each image's `artist_id` to the
name of the directory holding its file, and called every mismatch a misplaced
image. It reported 33,789 of 63,605 as wrongly filed. That number described
the comparison, not the library.

What it actually was:

  32,475  (97.1%)  one artist's own folder, spelled differently
                   — Telepurte/ vs telepurte/. Same artist, same art.
     657  ( 2.0%)  loose at the images root
     328  ( 1.0%)  in a folder named after a different artist

And the 1% did not mean what the tool assumed either. `ImageProvenance`
records the post and source every file was downloaded from — the
authoritative answer, which the tool never consulted. Querying it for all 328:

    144  provenance agrees with the record  (move would be right)
     87  provenance agrees with the FOLDER  (the record is wrong; move wrong)
     53  provenance names SEVERAL artists   (no single correct folder)
     41  no provenance at all
      3  agrees with neither

So the sweep would have misfiled or arbitrarily picked for ~41% of the only
set it was really needed for. The system already knew where each file came
from; the tool inferred it from a column and a directory name instead.

Operator, 2026-09-21: *"the current system consistently records where items
are and where they came from this is just complicating something works and
doesn't need fixing."* Correct on both counts.

Removed: the service, the tasks, the model and migration 0099's table, the
/api/cleanup/layout and /placement/* endpoints, the Maintenance card and its
store actions, and the tests. 0100 drops the table (rule #22 — no legacy).

KEPT deliberately, per the operator:
- `utils.paths.canonical_subdir` — new filesystem imports derive their
  directory from the artist's slug, matching what the downloader always did.
  Not part of this tool; removing it would be churn that fixes nothing.
- The 327 files run 1 moved (InsoUwu/ -> insouwu/). Same artist either way,
  and the gallery renders them correctly.
- Everything from #4223 (three-gate dedup, 256-bit pHash) and #4234 (backup
  credential exclusion). Those fixed problems that were actually reported.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LVjrnpQjRgHdvq95rASoiR
This commit is contained in:
2026-09-21 18:17:13 -04:00
co-authored by Claude Opus 5
parent 2dd9b956d5
commit 11a01a9686
14 changed files with 111 additions and 1850 deletions
+2 -112
View File
@@ -29,8 +29,8 @@ from quart import Blueprint, jsonify, request
from sqlalchemy import select
from ..extensions import get_session
from ..models import LibraryAuditRun, LibraryPlacementRun
from ..services import cleanup_service, library_layout
from ..models import LibraryAuditRun
from ..services import cleanup_service
from ._responses import error_response as _bad
cleanup_bp = Blueprint("cleanup", __name__, url_prefix="/api/cleanup")
@@ -196,113 +196,3 @@ async def audit_cancel(audit_id: int):
)
await session.commit()
return jsonify({"cancelled": True})
@cleanup_bp.route("/layout", methods=["GET"])
async def layout_survey():
"""Milestone #421 blast radius: which ImageRecord rows sit outside their
artist's canonical slug directory, per artist.
Read-only. `?check_disk=1` additionally stats every destination to find
collisions with a file already there and sources that have gone missing —
the numbers the apply refuses on, at the cost of one stat per misplaced
row over NFS. It is OFF by default because a count-only pass answers "how
big is this" in seconds where the disk pass can run for minutes and time
the request out.
"""
check_disk = request.args.get("check_disk", "").lower() in ("1", "true", "yes")
async with get_session() as session:
report = await session.run_sync(
lambda s: library_layout.survey_layout(
s, IMAGES_ROOT, check_disk=check_disk,
)
)
return jsonify({**report.as_dict(), "checked_disk": check_disk})
def _serialize_placement_run(run: LibraryPlacementRun, *, moves: bool = False) -> dict:
"""`moves` is opt-in: an applied whole-library run carries tens of
thousands of entries, which is a fine thing to hold in Postgres and a
poor thing to put in every list response."""
out = {
"id": run.id,
"status": run.status,
"artist_id": run.artist_id,
"started_at": run.started_at.isoformat() if run.started_at else None,
"finished_at": run.finished_at.isoformat() if run.finished_at else None,
"planned_count": run.planned_count,
"moved_count": run.moved_count,
"refused_count": run.refused_count,
"refusals": run.refusals or [],
"error": run.error,
}
if moves:
out["moves"] = run.moves or []
return out
@cleanup_bp.route("/placement/runs", methods=["GET"])
async def placement_runs():
"""Newest first. Without `moves`, so the list stays small."""
try:
limit = min(int(request.args.get("limit", "25")), 100)
except ValueError:
return _bad("invalid_limit")
async with get_session() as session:
rows = (await session.execute(
select(LibraryPlacementRun)
.order_by(LibraryPlacementRun.id.desc()).limit(limit)
)).scalars().all()
return jsonify({"runs": [_serialize_placement_run(r) for r in rows]})
@cleanup_bp.route("/placement/runs/<int:run_id>", methods=["GET"])
async def placement_run(run_id: int):
"""One run WITH its moves — this is the preview the operator reads before
agreeing, and the record of what happened afterwards."""
async with get_session() as session:
run = await session.get(LibraryPlacementRun, run_id)
if run is None:
return _bad("not_found", status=404)
return jsonify(_serialize_placement_run(run, moves=True))
@cleanup_bp.route("/placement/plan", methods=["POST"])
async def placement_plan():
"""Queue a planning run. `artist_id` scopes it to one artist, which is the
intended use: do one, look at the gallery, then continue or revert."""
body = await request.get_json(silent=True) or {}
artist_id = body.get("artist_id")
if artist_id is not None and not isinstance(artist_id, int):
return _bad("invalid_artist_id")
from ..tasks.library_placement import plan_placement
plan_placement.delay(artist_id)
return jsonify({"status": "dispatched"}), 202
@cleanup_bp.route("/placement/runs/<int:run_id>/apply", methods=["POST"])
async def placement_apply(run_id: int):
"""Execute a ready run. This renames files and rewrites rows."""
async with get_session() as session:
run = await session.get(LibraryPlacementRun, run_id)
if run is None:
return _bad("not_found", status=404)
if run.status != "ready":
return _bad("not_ready", detail=f"run is {run.status}")
from ..tasks.library_placement import apply_placement
apply_placement.delay(run_id)
return jsonify({"status": "dispatched"}), 202
@cleanup_bp.route("/placement/runs/<int:run_id>/revert", methods=["POST"])
async def placement_revert(run_id: int):
"""Put an applied run's files back. The reason the ledger is kept."""
async with get_session() as session:
run = await session.get(LibraryPlacementRun, run_id)
if run is None:
return _bad("not_found", status=404)
if run.status != "applied":
return _bad("not_applied", detail=f"run is {run.status}")
from ..tasks.library_placement import revert_placement
revert_placement.delay(run_id)
return jsonify({"status": "dispatched"}), 202
-3
View File
@@ -35,7 +35,6 @@ def make_celery() -> Celery:
"backend.app.tasks.backup",
"backend.app.tasks.admin",
"backend.app.tasks.library_audit",
"backend.app.tasks.library_placement",
"backend.app.tasks.translation",
],
)
@@ -63,8 +62,6 @@ def make_celery() -> Celery:
# 2026-06-07: a 2h audit blocked vacuum/backup/normalize for hours).
"backend.app.tasks.maintenance.*": {"queue": "maintenance"},
"backend.app.tasks.backup.*": {"queue": "maintenance_long"},
# 33k renames on NFS: long lane, same as backups.
"backend.app.tasks.library_placement.*": {"queue": "maintenance_long"},
"backend.app.tasks.admin.*": {"queue": "maintenance_long"},
"backend.app.tasks.library_audit.*": {"queue": "maintenance_long"},
# Translation backfill hits the LLM (~16s/item) → the long lane so it
-2
View File
@@ -22,7 +22,6 @@ from .import_batch import ImportBatch
from .import_settings import ImportSettings
from .import_task import ImportTask
from .library_audit_run import LibraryAuditRun
from .library_placement_run import LibraryPlacementRun
from .membership_sync import MembershipSync
from .ml_settings import MLSettings
from .patreon_failed_media import PatreonFailedMedia
@@ -86,7 +85,6 @@ __all__ = [
"ImportTask",
"ImportSettings",
"LibraryAuditRun",
"LibraryPlacementRun",
"MembershipSync",
"MLSettings",
"HeadAutoApplyRun",
@@ -1,99 +0,0 @@
"""LibraryPlacementRun — one run of the placement reconciler (milestone #421).
The library is keyed on the Artist row's `slug`, one directory per artist.
Every writer agrees on that now (`utils.paths.canonical_subdir`, task #4244),
but ~33,789 rows were written under older rules and sit in some other
artist's directory. This row is a run of the sweep that trues them up.
State machine, mirroring LibraryAuditRun:
running -> ready -> applied -> reverted
\\-> cancelled
(any) -> error
## The `moves` column does three jobs
`moves` is the plan: `[{"image_id": 1, "from": "...", "to": "..."}, ...]`.
1. **Preview.** It is what the operator reads before agreeing.
2. **Apply.** The apply executes THIS list rather than re-deriving the set,
so the preview cannot describe a different set from the apply. That is
rule 93's guarantee reached the way LibraryAuditRun reaches it — the
plan is materialised, not recomputed.
3. **Revert.** `from` is retained, so a batch that looks wrong in the
gallery goes back where it came from.
## An applied run IS the undo ledger — it must never be pruned
This is the trap lesson #4226 names: a record that answers both "what is the
current plan" and "what happened" gets deleted by whatever forgets the first.
A `ready` run is disposable state. An `applied` run is HISTORY, and it is the
only record of where 33,789 files used to be — delete it and the moves become
irreversible.
No pruning exists for this table today, and that is deliberate. If retention
is ever added here, it may prune `ready`, `cancelled` and `error` runs; an
`applied` run is only safe to drop once someone decides the moves are settled
and undo is no longer wanted, which is an operator decision and not a
timer's.
"""
from datetime import datetime
from typing import Any
from sqlalchemy import DateTime, ForeignKey, Integer, String, Text, func, text
from sqlalchemy.dialects.postgresql import JSONB
from sqlalchemy.orm import Mapped, mapped_column
from .base import Base
class LibraryPlacementRun(Base):
__tablename__ = "library_placement_run"
id: Mapped[int] = mapped_column(Integer, primary_key=True)
status: Mapped[str] = mapped_column(
String(16), nullable=False, default="running", index=True,
server_default="running",
)
# running | ready | applied | reverted | cancelled | error
# Scope. NULL = the whole library; set = one artist, which is how this is
# meant to be used — do one artist, look at it in the gallery, continue or
# revert. ondelete SET NULL rather than CASCADE: deleting an artist must
# not destroy the record of where their files were moved.
artist_id: Mapped[int | None] = mapped_column(
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True,
)
started_at: Mapped[datetime] = mapped_column(
DateTime(timezone=True), nullable=False, server_default=func.now(),
)
finished_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True), nullable=True,
)
planned_count: Mapped[int] = mapped_column(
Integer, nullable=False, default=0, server_default="0",
)
moved_count: Mapped[int] = mapped_column(
Integer, nullable=False, default=0, server_default="0",
)
refused_count: Mapped[int] = mapped_column(
Integer, nullable=False, default=0, server_default="0",
)
# [{"image_id": int, "from": str, "to": str}, ...] — see the module
# docstring. This is the plan, the audit trail and the undo, in that order
# of appearance and in one place.
moves: Mapped[list[dict[str, Any]]] = mapped_column(
JSONB, nullable=False, default=list, server_default=text("'[]'::jsonb"),
)
# [{"image_id": int, "reason": str}, ...] — rows the apply declined to
# touch, with why. A refusal is an expected outcome, not an error: the
# world moves between plan and apply, and every gate fails closed.
refusals: Mapped[list[dict[str, Any]]] = mapped_column(
JSONB, nullable=False, default=list, server_default=text("'[]'::jsonb"),
)
error: Mapped[str | None] = mapped_column(Text, nullable=True)
-434
View File
@@ -1,434 +0,0 @@
"""Milestone #421: where an image file BELONGS, and which rows are not there.
The library is keyed on the Artist row's `slug` — one directory per artist.
It grew a second (and third, and fourth) home for many of them because
`Importer._copy_to_library` used to name the destination after the IMPORT
folder while the downloader wrote under the slug. That writer is fixed
(`utils.paths.canonical_subdir`, task #4244); this module is the other half —
finding the rows whose files are still in the old places, and saying where
each one goes.
## Preview and apply share these predicates, they do not re-derive them
`_misplaced_conditions` and `destination_for` are the whole decision. The
report (task #4245) and the move (task #4246) both spread them rather than
writing their own — the house shape for rule 93, snippet #3087. A preview
that computes its set differently from the apply is a preview that can lie,
and here the apply RENAMES the operator's art.
## What writes and what does not
`survey_layout` and `plan_placement` are read-only — they report and they
record a plan. `apply_run` and `revert_run` are the only functions here that
rename a file or rewrite a row, and each does both for one row at a time,
updating the row only after its rename lands.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from datetime import UTC, datetime
from pathlib import Path
from sqlalchemy import func, select
from sqlalchemy.orm import Session
from ..models import Artist, ImageRecord, LibraryPlacementRun
from ..utils.paths import canonical_subdir
# Top-level directories under the images root that are STORES, not artists.
# A sweep that treats these as misplaced artwork would relocate the
# thumbnail cache, the attachment blobs, or the credential key.
# thumbs/ sha-addressed thumbnail cache (NOT path-keyed — see below)
# attachments/ sha-addressed non-media blobs
# cookies/, secrets/ credential material
# _backups/, _quarantine/ backup artifacts; files pulled out of the library
RESERVED_TOP_LEVEL = frozenset({
"thumbs", "attachments", "cookies", "secrets", "_backups", "_quarantine",
})
def canonical_dir(images_root: Path, slug: str) -> Path:
"""The one directory an artist's files belong under."""
return images_root / slug
def _misplaced_conditions(images_root: Path, artist_id: int, slug: str) -> list:
"""Rows of `artist_id` whose file is NOT under that artist's canonical
directory. Spread into both halves — never restated.
The prefix carries a trailing separator on purpose: without it, artist
`ara` would match every path under `arbuzbudesh/`, and the sweep would
report one artist's whole library as correctly placed while quietly
skipping another's.
`startswith` compiles to LIKE, where `_` and `%` are wildcards, and this
does not escape them. That is safe ONLY because `utils.slug.slugify`
reduces a slug to `[a-z0-9-]` — neither character can reach the pattern.
Widen that charset and this needs `autoescape=True`, or `poch4n_art`
starts matching `poch4nXart` too.
"""
prefix = f"{canonical_dir(images_root, slug)}/"
return [
ImageRecord.artist_id == artist_id,
ImageRecord.path.is_not(None),
~ImageRecord.path.startswith(prefix),
]
def destination_for(path: str, images_root: Path, slug: str) -> Path | None:
"""Where `path`'s file belongs, or None when this row must not be moved.
None means: the path is outside the images root, or its top-level segment
is a reserved store. Both are refusals rather than errors — a row pointing
somewhere unexpected is exactly what should NOT be relocated automatically.
## The one place this diverges from `canonical_subdir`
A file sitting at the images ROOT with a known artist moves under that
artist's directory here, where `canonical_subdir` would leave it alone.
The two answer different questions. At import time an empty subdir means
no artist was resolved, so there is nothing to canonicalise against. Here
the row already CARRIES an artist_id, so a file at the root is an anomaly
with a known correct home — which is the whole point of the sweep.
(The 660 unattributed files at the root have no artist_id at all and are
not reachable from these predicates; task #4247 decides those.)
"""
p = Path(path)
try:
rel_dir = p.parent.relative_to(images_root)
except ValueError:
return None
parts = rel_dir.parts
if parts and parts[0] in RESERVED_TOP_LEVEL:
return None
sub = canonical_subdir(str(rel_dir) if str(rel_dir) != "." else "", slug)
if not sub:
# At the root, with an artist — see the docstring above.
return canonical_dir(images_root, slug) / p.name
return images_root / sub / p.name
@dataclass
class ArtistLayout:
"""One artist's verdict."""
artist_id: int
name: str
slug: str
canonical_rows: int = 0
misplaced_rows: int = 0
# The non-canonical top-level directories this artist's files sit in —
# "Conto", "StickySpoodge", … This is what makes the report readable as
# the family list the disk survey found.
stray_dirs: list[str] = field(default_factory=list)
collisions: list[str] = field(default_factory=list)
missing_files: int = 0
unmovable: int = 0
@dataclass
class LayoutReport:
artists: list[ArtistLayout] = field(default_factory=list)
total_rows: int = 0
canonical_rows: int = 0
misplaced_rows: int = 0
collision_count: int = 0
missing_files: int = 0
unmovable: int = 0
unattributed_rows: int = 0
def as_dict(self) -> dict:
return {
"total_rows": self.total_rows,
"canonical_rows": self.canonical_rows,
"misplaced_rows": self.misplaced_rows,
"collision_count": self.collision_count,
"missing_files": self.missing_files,
"unmovable": self.unmovable,
"unattributed_rows": self.unattributed_rows,
"artists": [
{
"artist_id": a.artist_id,
"name": a.name,
"slug": a.slug,
"canonical_rows": a.canonical_rows,
"misplaced_rows": a.misplaced_rows,
"stray_dirs": a.stray_dirs,
"collisions": a.collisions,
"missing_files": a.missing_files,
"unmovable": a.unmovable,
}
for a in self.artists
if a.misplaced_rows or a.collisions
],
}
def survey_layout(
session: Session, images_root: Path, *, check_disk: bool = True,
) -> LayoutReport:
"""Read-only blast radius for the consolidation.
`check_disk` stats every destination to find rows that would collide with
a file already there, and sources that have already gone missing. It is
the honest number and it is what the apply will refuse on, but it costs
one stat per misplaced row over NFS — turn it off when you only want
counts.
"""
report = LayoutReport()
report.total_rows = session.execute(
select(func.count(ImageRecord.id))
).scalar_one()
report.unattributed_rows = session.execute(
select(func.count(ImageRecord.id)).where(ImageRecord.artist_id.is_(None))
).scalar_one()
artists = session.execute(
select(Artist).order_by(Artist.slug)
).scalars().all()
for artist in artists:
layout = ArtistLayout(
artist_id=artist.id, name=artist.name, slug=artist.slug,
)
conds = _misplaced_conditions(images_root, artist.id, artist.slug)
owned = session.execute(
select(func.count(ImageRecord.id))
.where(ImageRecord.artist_id == artist.id)
).scalar_one()
rows = session.execute(
select(ImageRecord.id, ImageRecord.path).where(*conds)
).all()
layout.misplaced_rows = len(rows)
layout.canonical_rows = owned - len(rows)
strays: set[str] = set()
destinations: dict[str, int] = {}
for row_id, path in rows:
dest = destination_for(path, images_root, artist.slug)
if dest is None:
layout.unmovable += 1
continue
try:
top = Path(path).parent.relative_to(images_root).parts
strays.add(top[0] if top else "<root>")
except ValueError:
strays.add("<outside>")
key = str(dest)
if key in destinations:
layout.collisions.append(key)
else:
destinations[key] = row_id
if check_disk:
if not Path(path).exists():
layout.missing_files += 1
elif dest.exists():
layout.collisions.append(key)
layout.stray_dirs = sorted(strays)
report.artists.append(layout)
report.canonical_rows += layout.canonical_rows
report.misplaced_rows += layout.misplaced_rows
report.collision_count += len(layout.collisions)
report.missing_files += layout.missing_files
report.unmovable += layout.unmovable
return report
# --- the reconciler: plan -> apply -> revert (#4246) -------------------------
#
# The three verbs share one materialised plan rather than each deriving its
# own set. `plan_placement` writes `LibraryPlacementRun.moves`; `apply_run`
# executes THAT list; `revert_run` walks it backwards. A preview that can
# disagree with its apply is the failure this shape exists to prevent, and
# here the apply renames the operator's art.
#
# Every step fails CLOSED. The world moves between planning and applying —
# a download lands, a supersede rewrites a path, a file is deleted — so the
# apply re-checks each row against what the plan recorded and declines the
# ones that moved on, instead of trusting a plan that may be minutes old.
def plan_placement(
session: Session, images_root: Path, *, artist_id: int | None = None,
) -> LibraryPlacementRun:
"""Build (and persist) the move plan. Touches no files.
`artist_id` scopes the run to one artist, which is how this is meant to be
used: do one, look at the gallery, then continue or revert. None plans the
whole library.
"""
stmt = select(Artist).order_by(Artist.slug)
if artist_id is not None:
stmt = stmt.where(Artist.id == artist_id)
artists = session.execute(stmt).scalars().all()
candidates: list[dict] = []
wanted: dict[str, int] = {}
for artist in artists:
rows = session.execute(
select(ImageRecord.id, ImageRecord.path)
.where(*_misplaced_conditions(images_root, artist.id, artist.slug))
).all()
for row_id, path in rows:
dest = destination_for(path, images_root, artist.slug)
if dest is None or dest.exists():
continue
key = str(dest)
candidates.append({"image_id": row_id, "from": path, "to": key})
wanted[key] = wanted.get(key, 0) + 1
# Two rows wanting one destination: plan NEITHER. Which of them "wins" is
# not this sweep's call, and planning one of them would silently pick a
# winner by iteration order. Counting first and filtering after is what
# makes that true — claiming as we go would quietly keep whichever came
# first.
moves = [m for m in candidates if wanted[m["to"]] == 1]
run = LibraryPlacementRun(
status="ready", artist_id=artist_id, moves=moves,
planned_count=len(moves),
)
session.add(run)
session.flush()
return run
def _move_one(src: Path, dest: Path) -> str | None:
"""Rename `src` to `dest`. Returns a refusal reason, or None on success.
A rename within one filesystem, so no copy and no free space needed. The
destination check is not a race-free guarantee — nothing here is — but it
turns the common case of "something already landed there" into a refusal
instead of a silent overwrite.
"""
if not src.exists():
return "source missing"
if dest.exists():
return "destination occupied"
try:
dest.parent.mkdir(parents=True, exist_ok=True)
src.rename(dest)
except OSError as exc:
return f"rename failed: {exc}"
return None
def apply_run(
session: Session, run: LibraryPlacementRun, *, chunk: int = 0,
) -> LibraryPlacementRun:
"""Execute a `ready` run's stored plan: file and row together, per row.
The row is updated ONLY after its rename lands, so a refused or failed
move can never leave `ImageRecord.path` pointing at a file that is not
there. Refusals are recorded and the run continues — one row that moved
on since planning is not a reason to abandon the other 33,788.
`chunk` commits progress every N moves. Set it for any real run: the
ledger is the ONLY record of where a file came from, so a worker that
dies at row 30,000 of 33,789 must not take the undo information for the
first 29,999 with it. Left at 0 (tests, small runs) everything persists
in one go at the end.
Re-running a partially-applied plan is safe rather than clever: the rows
already moved no longer match their `from`, so they refuse as "row moved
since planning" instead of being moved twice.
"""
if run.status != "ready":
raise ValueError(f"run {run.id} is {run.status}, not ready")
refusals: list[dict] = []
moved: list[dict] = []
def _persist() -> None:
# Reassign rather than mutate: SQLAlchemy does not track in-place
# changes to a JSONB list, so an .append() alone would never reach
# the database and the ledger would silently stay empty.
run.moves = list(moved)
run.refusals = list(refusals)
run.moved_count = len(moved)
run.refused_count = len(refusals)
session.commit()
# Snapshot the plan before iterating: `_persist` reassigns `run.moves`,
# and iterating the attribute while rewriting it would walk a list that
# changes underneath the loop.
plan = list(run.moves)
for done, move in enumerate(plan, start=1):
record = session.get(ImageRecord, move["image_id"])
if record is None:
refusals.append({"image_id": move["image_id"], "reason": "row gone"})
continue
if record.path != move["from"]:
# Something rewrote this row since the plan was built — a
# supersede, or an earlier run. The plan is stale for it.
refusals.append({
"image_id": move["image_id"], "reason": "row moved since planning",
})
continue
reason = _move_one(Path(move["from"]), Path(move["to"]))
if reason is not None:
refusals.append({"image_id": move["image_id"], "reason": reason})
continue
record.path = move["to"]
moved.append(move)
if chunk and done % chunk == 0:
_persist()
run.moves = moved
run.refusals = refusals
run.moved_count = len(moved)
run.refused_count = len(refusals)
run.status = "applied"
run.finished_at = datetime.now(UTC)
session.flush()
return run
def revert_run(
session: Session, run: LibraryPlacementRun,
) -> LibraryPlacementRun:
"""Put an applied run's files back where they came from.
This is why `from` is retained. It is the answer to "do one artist, look
at it, and undo if it reads wrong" — which is a cheaper way to settle
whether artist_id or the folder held the truth (#4257) than arguing it
from a sample.
Refuses the same way the apply does: a file someone has since moved or
replaced stays where it is, and its row is left alone.
A revert interrupted half way is resumable by re-running it: the rows
already put back no longer sit at `to`, so they refuse rather than move
twice. Unlike the apply this needs no chunked persistence — it consumes
the ledger rather than producing it, so a crash costs progress, not
information.
"""
if run.status != "applied":
raise ValueError(f"run {run.id} is {run.status}, not applied")
refusals: list[dict] = []
reverted = 0
for move in run.moves:
record = session.get(ImageRecord, move["image_id"])
if record is None or record.path != move["to"]:
refusals.append({
"image_id": move["image_id"], "reason": "row changed since apply",
})
continue
reason = _move_one(Path(move["to"]), Path(move["from"]))
if reason is not None:
refusals.append({"image_id": move["image_id"], "reason": reason})
continue
record.path = move["from"]
reverted += 1
run.refusals = refusals
run.refused_count = len(refusals)
run.moved_count = run.moved_count - reverted
run.status = "reverted"
run.finished_at = datetime.now(UTC)
session.flush()
return run
-125
View File
@@ -1,125 +0,0 @@
"""Placement reconciler tasks — plan, apply, revert (milestone #421).
The service (`services.library_layout`) holds the decisions; this module is
only the async wrapper, matching `tasks.library_audit`: run on the
maintenance queue, mark the run `error` with a traceback if anything escapes,
and return a small summary dict so eager-mode tests can assert on it.
Applying is a long run — 33,789 renames on the operator's library at the time
of writing — so `apply_placement` persists its ledger in chunks rather than
at the end. That ledger is the only record of where each file came from, and
a worker that dies two thirds of the way through must not take the undo
information for the first two thirds with it.
"""
import logging
import traceback
from datetime import UTC, datetime
from pathlib import Path
from sqlalchemy.exc import DBAPIError, OperationalError
from ..celery_app import celery
from ..models import LibraryPlacementRun
from ..services import library_layout
from ._sync_engine import sync_session_factory as _sync_session_factory
log = logging.getLogger(__name__)
IMAGES_ROOT = Path("/images")
# Commit the ledger every this many moves. Small enough that a crash loses
# seconds of work, large enough not to make a COMMIT per rename.
_APPLY_CHUNK = 200
def _fail(session, run_id: int, message: str) -> None:
run = session.get(LibraryPlacementRun, run_id)
if run is not None:
run.status = "error"
run.error = message
run.finished_at = datetime.now(UTC)
session.commit()
@celery.task(
name="backend.app.tasks.library_placement.plan_placement",
autoretry_for=(OperationalError, DBAPIError),
retry_backoff=5, retry_backoff_max=60, retry_jitter=True, max_retries=3,
soft_time_limit=900, time_limit=1000,
)
def plan_placement(artist_id: int | None = None) -> dict:
"""Build a move plan and leave it `ready` for the operator to read.
Reads rows and stats destinations; moves nothing.
"""
SessionLocal = _sync_session_factory()
with SessionLocal() as session:
run = library_layout.plan_placement(
session, IMAGES_ROOT, artist_id=artist_id,
)
session.commit()
return {
"run_id": run.id, "status": run.status,
"planned_count": run.planned_count,
}
@celery.task(
name="backend.app.tasks.library_placement.apply_placement",
soft_time_limit=7200, time_limit=7500,
)
def apply_placement(run_id: int) -> dict:
"""Execute a `ready` run's stored plan. Renames files and rewrites rows.
No autoretry: a retry would re-enter a half-applied plan on a schedule
nobody asked for. Re-running IS safe (the applied rows refuse as "row
moved since planning"), but that should be the operator's decision after
reading what happened, not the queue's.
"""
SessionLocal = _sync_session_factory()
with SessionLocal() as session:
run = session.get(LibraryPlacementRun, run_id)
if run is None:
return {"run_id": run_id, "status": "missing"}
if run.status != "ready":
return {"run_id": run_id, "status": run.status, "skipped": True}
try:
library_layout.apply_run(session, run, chunk=_APPLY_CHUNK)
session.commit()
except Exception:
log.exception("placement apply failed for run %s", run_id)
session.rollback()
_fail(session, run_id, traceback.format_exc())
return {"run_id": run_id, "status": "error"}
return {
"run_id": run_id, "status": run.status,
"moved": run.moved_count, "refused": run.refused_count,
}
@celery.task(
name="backend.app.tasks.library_placement.revert_placement",
soft_time_limit=7200, time_limit=7500,
)
def revert_placement(run_id: int) -> dict:
"""Put an applied run's files back where they came from."""
SessionLocal = _sync_session_factory()
with SessionLocal() as session:
run = session.get(LibraryPlacementRun, run_id)
if run is None:
return {"run_id": run_id, "status": "missing"}
if run.status != "applied":
return {"run_id": run_id, "status": run.status, "skipped": True}
try:
library_layout.revert_run(session, run)
session.commit()
except Exception:
log.exception("placement revert failed for run %s", run_id)
session.rollback()
_fail(session, run_id, traceback.format_exc())
return {"run_id": run_id, "status": "error"}
return {
"run_id": run_id, "status": run.status,
"refused": run.refused_count,
}