fix: a post record is marked seen only after it is upserted, and an hourly sweep dates the posts killed walks left undated (#4436)
CI and images / lint (push) Successful in 3s
CI and images / extension-version (push) Successful in 3s
CI and images / extension-test (push) Successful in 17s
CI and images / frontend-build (push) Successful in 27s
CI and images / backend-lint-and-test (push) Successful in 33s
CI and images / integration (push) Failing after 2m25s
CI and images / sign-extension (push) Skipped
CI and images / build-web (push) Skipped
CI and images / smoke-web (push) Skipped
CI and images / promote (push) Skipped
CI and images / build-agent (push) Skipped

ingest_core marked a post's record key seen when it wrote _post.json, but
the record reaches the database (and dates the post and its images) only
in phase 3. A walk killed before phase 3 left posts the ledger called
recorded and the database never dated; ticks early-out long before
reaching them again. Keys now join the media in mark_seen_after_import.

date_posts_from_records finds each undated native post's record under
its artist's folder and upserts it with the post's own source. Hourly on
maintenance_long; an empty query once everything is dated.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LVjrnpQjRgHdvq95rASoiR
This commit is contained in:
2026-09-25 11:36:17 -04:00
co-authored by Claude Opus 5.5
parent ca802bd885
commit 621c2b8315
7 changed files with 182 additions and 4 deletions
+7
View File
@@ -69,6 +69,8 @@ def make_celery() -> Celery:
# up behind it (2026-09-24: 7 waiting, "all workers busy for 18
# minutes"). An exact name wins over the glob above.
"backend.app.tasks.maintenance.backfill_phash": {"queue": "maintenance_long"},
# Walks a folder tree per artist with undated posts (#4436).
"backend.app.tasks.maintenance.date_posts_from_records": {"queue": "maintenance_long"},
"backend.app.tasks.backup.*": {"queue": "maintenance_long"},
"backend.app.tasks.admin.*": {"queue": "maintenance_long"},
"backend.app.tasks.library_audit.*": {"queue": "maintenance_long"},
@@ -156,6 +158,11 @@ def make_celery() -> Celery:
"schedule": 86400.0, # daily — sweep .part/.partial left by a
# download/import killed mid-write (graceful-shutdown fallout)
},
"date-posts-from-records-hourly": {
"task": "backend.app.tasks.maintenance.date_posts_from_records",
"schedule": 3600.0, # an empty query once every native post is
# dated; otherwise upserts the _post.json a killed walk left (#4436)
},
"backfill-phash-daily": {
"task": "backend.app.tasks.maintenance.backfill_phash",
"schedule": 86400.0, # daily — NULL-only, so a no-op once the
+9 -2
View File
@@ -289,6 +289,11 @@ class Ingester:
# Media handed to phase 3 for import. Marked seen by phase 3 once the
# import has run (`mark_seen_after_import`), not here — see there.
fetched: list[tuple[str, str]] = []
# Post-record keys written this walk. Marked with the media, after
# phase 3 has upserted the records — marking them at write time left a
# walk killed before phase 3 with posts the ledger calls recorded that
# the database never dated, and no later tick walks back to them (#4436).
recorded: list[tuple[str, str]] = []
downloaded = 0
errors = 0
quarantined = 0
@@ -342,7 +347,9 @@ class Ingester:
written_paths=written,
post_record_paths=list(post_records),
relink_source_paths=list(relink),
mark_seen_after_import=lambda: self._mark_seen(source_id, fetched),
mark_seen_after_import=lambda: self._mark_seen(
source_id, fetched + recorded,
),
stdout="\n".join(log_lines),
stderr="",
return_code=return_code,
@@ -511,7 +518,7 @@ class Ingester:
posts_with_body += 1
if rec.path is not None:
post_records.append(str(rec.path))
self._mark_seen(source_id, [(pkey, ppid)])
recorded.append((pkey, ppid))
# Per-post handling line in the run stdout (the existing
# "Raw stdout" panel) — the downloader already read the
# post; we only format its outcome here. post_type beside
@@ -0,0 +1,86 @@
"""Date the posts whose record reached the disk but never the database (#4436).
A native walk writes each post's record (`_post.json`, or Discord's
`<day>_<message id>_post.json`) as it goes, and phase 3 upserts those records
after the walk — which is when the post, and through it its images, get their
date. Until #4436 the walk marked a record's post key seen at write time, so a
walk killed before phase 3 (a restart, a stall) left the post undated with the
ledger saying it was done, and no later tick walked back that far to fix it.
The record files are still on disk. This finds each undated native post's
record under its artist's folder and upserts it with the post's OWN source —
never the (artist, platform) lookup, which picks the artist's first source and
would misfile a Discord channel's post (#4435). Idempotent: a post that is
already dated is never looked at, and upserting a record twice changes nothing.
"""
from __future__ import annotations
import json
import logging
from collections import defaultdict
from pathlib import Path
from sqlalchemy import select
from sqlalchemy.orm import Session
from ..models import Artist, Post, Source
from ..utils.sidecar import parse_sidecar
from .download_backends import NATIVE_INGESTER_PLATFORMS
log = logging.getLogger(__name__)
def _is_record(path: Path) -> bool:
return path.name == "_post.json" or path.name.endswith("_post.json")
def _record_id(path: Path) -> str | None:
try:
data = json.loads(path.read_text("utf-8"))
except (OSError, ValueError):
return None
if not isinstance(data, dict):
return None
return parse_sidecar(data).external_post_id
def date_posts_from_records(session: Session, importer, images_root: Path) -> dict:
"""Upsert the on-disk record of every undated native post. Returns counts."""
rows = session.execute(
select(Post.external_post_id, Post.source_id, Post.artist_id, Source.platform)
.join(Source, Source.id == Post.source_id)
.where(
Post.post_date.is_(None),
Post.synthesized_by.is_(None),
Source.platform.in_(NATIVE_INGESTER_PLATFORMS),
)
).all()
wanted: dict[tuple[int, str], dict[str, int]] = defaultdict(dict)
for epid, source_id, artist_id, platform in rows:
wanted[(artist_id, platform)][epid] = source_id
summary = {"undated": len(rows), "dated": 0, "no_record": 0}
for (artist_id, platform), by_epid in wanted.items():
artist = session.get(Artist, artist_id)
root = Path(images_root) / artist.slug / platform if artist else None
if root is None or not root.is_dir():
summary["no_record"] += len(by_epid)
continue
found = 0
for record in root.rglob("*post.json"):
if not _is_record(record):
continue
epid = _record_id(record)
source_id = by_epid.pop(epid, None) if epid else None
if source_id is None:
continue
source = session.get(Source, source_id)
if importer.upsert_post_record(record, artist=artist, source=source):
found += 1
if not by_epid:
break
summary["dated"] += found
summary["no_record"] += len(by_epid)
if summary["undated"]:
log.info("date_posts_from_records: %s", summary)
return summary
+26
View File
@@ -731,6 +731,32 @@ def recover_stalled_download_events() -> int:
return events_recovered
@celery.task(
name="backend.app.tasks.maintenance.date_posts_from_records",
soft_time_limit=1500,
time_limit=1800,
)
def date_posts_from_records() -> dict:
"""Date undated native posts from the records their walk left on disk
(#4436). Hourly and self-limiting: once every post is dated it is one
empty query."""
from ..services.importer import Importer
from ..services.post_record_repair import date_posts_from_records as _repair
from ..services.thumbnailer import Thumbnailer
images_root = IMAGES_ROOT
SessionLocal = _sync_session_factory()
with SessionLocal() as session:
importer = Importer(
session=session,
images_root=images_root,
import_root=images_root,
thumbnailer=Thumbnailer(images_root=images_root),
settings=ImportSettings.load_sync(session),
)
return _repair(session, importer, images_root)
@celery.task(name="backend.app.tasks.maintenance.recover_stalled_backup_runs")
def recover_stalled_backup_runs() -> int:
"""Flip BackupRun rows stuck in running/restoring past the hard limit