Files
FabledCurator/backend/app/services/ml/training_data.py
T
bvandeusenandClaude Opus 5.5 7b1570f2a5
CI and images / lint (push) Successful in 2s
CI and images / extension-version (push) Successful in 3s
CI and images / extension-test (push) Successful in 20s
CI and images / frontend-build (push) Successful in 23s
CI and images / backend-lint-and-test (push) Successful in 32s
CI and images / integration (push) Successful in 2m24s
CI and images / sign-extension (push) Successful in 3s
CI and images / build-agent (push) Successful in 5s
CI and images / build-web (push) Successful in 1m35s
CI and images / smoke-web (push) Successful in 56s
CI and images / promote (push) Successful in 1s
feat: retire the sketch/doodle WIP title tier, and the review strip asks "Is this a WIP?" (milestone 430)
The soft tier (#1474) tagged `wip` on any post titled sketch, doodle or
scribble: 6,096 of the library's 8,876 wip tags. Its conflict audit flagged
most of them, because finished art scores >= 0.5 on some content head,
which filled the Gallery strip with 2,086 cards. A "sketch" is usually
finished work, so the operator retired the tier. The artist's own
"WIP" / "work in progress" title rule and human wip tags stay.

- Removed: the soft matcher, source and prefilter; the importer and backfill
  soft branches; soft_wip_conflict_audit with its task and beat; the
  wip_soft_title_tagging_enabled setting, API field and toggle; and
  wip_title_soft from _AUTO_SOURCES.
- Migration 0112: a soft tag the operator confirmed, or kept from the strip,
  becomes `manual`. Every other soft tag is deleted, along with the open
  review cards whose tag is gone. The column is dropped.
- Review strip (#4424): each card asks "Is this a WIP?" (or banner, and so on),
  and the buttons read "Is a WIP" / "Is not a WIP". The content tag it also
  scored on is shown as the reason it was flagged.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LVjrnpQjRgHdvq95rASoiR
2026-09-25 08:20:28 -04:00

193 lines
7.4 KiB
Python

"""Shared data-selection + validated-metric helpers for the heads trainer.
Born in the head-vs-centroid eval harness (#1130, tag_eval.py) that proved the
"frozen embedding + small trained head (with negatives)" spine; the harness was
retired 2026-07-02 (operator: the tagging system is proven, the eval isn't
needed) and these survivors moved here — they ARE the heads' production data
pipeline (heads.py trains and scores with them nightly).
numpy/scikit-learn are imported lazily inside the functions that need them so
the API worker (base image, no ML stack) can import this module.
"""
from __future__ import annotations
from typing import Any
from sqlalchemy import func, select
from sqlalchemy.orm import Session
from ...models import (
ImageRecord,
Tag,
TagPositiveConfirmation,
TagSuggestionRejection,
)
from ...models.tag import image_tag
# Auto-apply sources whose tags are PROVISIONAL: they never train a head (or seed
# a CCIP reference) unless the operator confirms them (milestone 139). Keeping
# auto-applied predictions out of training is what makes them "soft" — a misfire
# can't reinforce itself, so the retraction sweep can actually drop it.
# `process_auto` (#1464): wip/editor screenshot applied by the process sweep are
# ALSO provisional — the head must learn only from title (`wip_title`) + manual
# labels, never its own auto-applied output, or it would runaway (operator 2026-07-12).
_AUTO_SOURCES = (
"head_auto", "ccip_auto", "ml_auto", "presentation_auto", "process_auto",
)
def _hygiene_excluded_ids(session: Session) -> set[int]:
"""Ids of images carrying ANY system tag (wip / banner / editor
screenshot — milestone #128). These images are excluded from OTHER
concepts' head training entirely: not positives (a rough wip tagged as a
character drags that head toward 'generic sketch') and not sampled or
rejection negatives (a wip OF character X is not evidence against X) —
simply absent. A system tag's OWN head trains on them unchanged; that is
what makes auto-flagging banners/editor screenshots work.
Item-level by design: a wip-tagged process video contributes (or
withholds) ALL its sampled frames, though some may show the finished
piece. Operator call 2026-07-03: with enough clean data this washes out —
no per-frame handling.
"""
return set(
session.execute(
select(image_tag.c.image_record_id)
.join(Tag, Tag.id == image_tag.c.tag_id)
.where(Tag.is_system.is_(True))
).scalars().all()
)
def _ids_with_tag(session: Session, tag_id: int) -> list[int]:
"""Image ids that count as POSITIVES for this tag's head: human-applied
(manual / accepted) tags PLUS any auto-applied tag the operator explicitly
confirmed (TagPositiveConfirmation). Unconfirmed auto-applied tags are
EXCLUDED — they are provisional and must not train the head that judges
them (milestone 139)."""
confirmed = (
select(TagPositiveConfirmation.image_record_id)
.where(TagPositiveConfirmation.tag_id == tag_id)
)
return [
r[0] for r in session.execute(
select(image_tag.c.image_record_id)
.where(image_tag.c.tag_id == tag_id)
.where(
image_tag.c.source.not_in(_AUTO_SOURCES)
| image_tag.c.image_record_id.in_(confirmed)
)
).all()
]
def _rejected_ids(session: Session, tag_id: int) -> list[int]:
return [
r[0] for r in session.execute(
select(TagSuggestionRejection.image_record_id)
.where(TagSuggestionRejection.tag_id == tag_id)
).all()
]
def _applied_or_rejected(session: Session, tag_ids) -> dict[int, set[int]]:
"""Per-tag skip set for the auto-apply sweeps: every image that ALREADY carries
the tag (ANY source — not just training positives) OR has rejected it. A sweep
never re-applies to these. Shared by auto_apply_sweep + system_tag_auto_apply_sweep
(heads.py) and scheduled_ccip_auto_apply (tasks/ml.py). Callers mutate the returned
sets in-place to also dedupe within a single run."""
skip: dict[int, set[int]] = {}
for tid in tag_ids:
ids = {
r[0] for r in session.execute(
select(image_tag.c.image_record_id).where(image_tag.c.tag_id == tid)
).all()
}
ids.update(_rejected_ids(session, tid))
skip[tid] = ids
return skip
def _sample_unlabeled(session: Session, exclude: set[int], limit: int) -> list[int]:
"""Random image ids (with an embedding) NOT carrying the tag. Concepts are
sparse, so an untagged image is almost always a true negative."""
stmt = (
select(ImageRecord.id)
.where(ImageRecord.siglip_embedding.is_not(None))
.order_by(func.random())
.limit(limit)
)
if exclude:
stmt = stmt.where(ImageRecord.id.not_in(exclude))
return [r[0] for r in session.execute(stmt).all()]
def _load_embeddings(session: Session, ids: list[int]) -> dict[int, Any]:
import numpy as np
out: dict[int, Any] = {}
if not ids:
return out
# Chunk the IN list to stay well under psycopg's parameter ceiling.
for i in range(0, len(ids), 2000):
chunk = ids[i:i + 2000]
for rid, emb in session.execute(
select(ImageRecord.id, ImageRecord.siglip_embedding)
.where(ImageRecord.id.in_(chunk))
.where(ImageRecord.siglip_embedding.is_not(None))
).all():
out[rid] = np.asarray(emb, dtype=np.float32)
return out
def _l2norm(X, np):
n = np.linalg.norm(X, axis=1, keepdims=True)
n[n == 0] = 1.0
return X / n
def _metrics_from_scores(y, scores, np) -> dict[str, float]:
from sklearn.metrics import average_precision_score, precision_recall_curve
ap = float(average_precision_score(y, scores))
prec, rec, thr = precision_recall_curve(y, scores)
f1 = (2 * prec * rec) / np.clip(prec + rec, 1e-9, None)
best = int(np.argmax(f1))
# thr has len = len(prec)-1; map best index safely.
t = float(thr[min(best, len(thr) - 1)]) if len(thr) else 0.5
return {
"ap": round(ap, 4),
"precision": round(float(prec[best]), 4),
"recall": round(float(rec[best]), 4),
"f1": round(float(f1[best]), 4),
"threshold": round(t, 4),
}
def _safe_folds(y, folds, np) -> int:
minority = int(min(np.bincount(y)))
return max(2, min(folds, minority))
def _auto_apply_point(y, scores, target, np) -> dict | None:
"""The auto-apply operating point: the threshold that yields the MOST recall
while holding precision >= target. This answers 'could this concept fire
without a human, and how much would it catch?' Returns None if no threshold
reaches the precision target (concept not auto-apply-ready)."""
from sklearn.metrics import precision_recall_curve
prec, rec, thr = precision_recall_curve(y, scores)
best = None # (threshold, precision, recall) maximizing recall s.t. prec>=target
for i in range(len(thr)): # thr[i] corresponds to prec[i], rec[i]
if prec[i] >= target and (best is None or rec[i] > best[2]):
best = (float(thr[i]), float(prec[i]), float(rec[i]))
if best is None:
return None
return {
"target": round(float(target), 4),
"threshold": round(best[0], 4),
"precision": round(best[1], 4),
"recall": round(best[2], 4),
}