Files
FabledCurator/tests/test_process_auto_apply.py
T
bvandeusenandClaude Opus 5.5 ff70f837d0
CI and images / lint (push) Successful in 2s
CI and images / extension-version (push) Successful in 2s
CI and images / frontend-build (push) Successful in 20s
CI and images / backend-lint-and-test (push) Successful in 30s
CI and images / integration (push) Successful in 2m18s
CI and images / sign-extension (push) Successful in 3s
CI and images / build-agent (push) Successful in 6s
CI and images / build-web (push) Successful in 1m44s
CI and images / smoke-web (push) Successful in 56s
CI and images / promote (push) Skipped
refactor: test row factories and the fetch stub have one copy each (3109)
tests/factories.py holds image_row/make_image/make_image_async/make_tag. The
17 byte-identical _img/_tag helpers (15 modules) now import them under their
old names, so no call site changed. frontend/test/support/stubFetch.js
replaces 15 copies that differed only in formatting. Copies whose bodies
differ (other defaults, other columns, a url-only stub) are left as they
are; folding those needs a look at each caller.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LVjrnpQjRgHdvq95rASoiR
2026-09-24 16:39:38 -04:00

179 lines
7.0 KiB
Python

"""Process-group auto-apply sweep (#1464): wip / editor screenshot auto-tag at a
flat threshold with a PROVISIONAL source (`process_auto`) so the head never trains
on its own output, and stay VISIBLE (unlike chrome). Mirrors the chrome guards.
numpy-only (no sklearn), tested directly via the sync session."""
import pytest
from sqlalchemy import select
from backend.app.models import (
ImageRecord,
MLSettings,
PresentationReview,
Tag,
TagHead,
TagKind,
)
from backend.app.models.tag import image_tag
from backend.app.services.ml.heads import system_tag_auto_apply_sweep
from backend.app.services.ml.training_data import _ids_with_tag
from tests.factories import make_image as _img
pytestmark = pytest.mark.integration
def _emb(slot: int) -> list[float]:
v = [0.0] * 1152
v[slot] = 3.0
return v
def _head(db, tag_id: int, slot: int, *, weight=1.0):
s = db.execute(select(MLSettings).where(MLSettings.id == 1)).scalar_one()
w = [0.0] * 1152
w[slot] = weight
db.add(TagHead(
tag_id=tag_id, embedding_version=s.embedder_model_version,
weights=w, bias=0.0, suggest_threshold=0.5, auto_apply_threshold=0.5,
n_pos=60, n_neg=90, ap=0.9, precision_cv=0.98, recall=0.7,
))
def _system_tag(db, name):
return db.execute(
select(Tag).where(Tag.is_system.is_(True), Tag.name == name)
).scalar_one()
def _enable_process(db):
# process auto-apply is opt-in (default False) — turn it on for these tests.
db.execute(select(MLSettings).where(MLSettings.id == 1)).scalar_one().process_auto_apply_enabled = True
def _source(db, image_id, tag_id):
return db.execute(
select(image_tag.c.source)
.where(image_tag.c.image_record_id == image_id)
.where(image_tag.c.tag_id == tag_id)
).scalar_one_or_none()
def test_process_sweep_applies_wip_and_editor(db_sync):
_enable_process(db_sync)
wip = _system_tag(db_sync, "wip")
editor = _system_tag(db_sync, "editor screenshot")
_head(db_sync, wip.id, 0, weight=3.0)
_head(db_sync, editor.id, 1, weight=3.0)
w_img = _img(db_sync, "a" * 64, _emb(0))
e_img = _img(db_sync, "b" * 64, _emb(1))
db_sync.commit()
res = system_tag_auto_apply_sweep(db_sync, mode="process")
assert res["n_applied"] == 2
assert _source(db_sync, w_img.id, wip.id) == "process_auto"
assert _source(db_sync, e_img.id, editor.id) == "process_auto"
def test_process_sweep_disabled_by_default_is_noop(db_sync):
# process_auto_apply_enabled defaults False (opt-in) — no enable = no-op.
wip = _system_tag(db_sync, "wip")
_head(db_sync, wip.id, 0, weight=3.0)
img = _img(db_sync, "c" * 64, _emb(0))
db_sync.commit()
res = system_tag_auto_apply_sweep(db_sync, mode="process")
assert res["n_applied"] == 0
assert _source(db_sync, img.id, wip.id) is None
def test_process_sweep_skips_valued_image(db_sync):
# Guard 1: never auto-apply to an image the operator already content-tagged.
_enable_process(db_sync)
wip = _system_tag(db_sync, "wip")
_head(db_sync, wip.id, 0, weight=3.0)
content = Tag(name="mychar", kind=TagKind.character)
db_sync.add(content)
db_sync.flush()
img = _img(db_sync, "d" * 64, _emb(0))
db_sync.execute(image_tag.insert().values(
image_record_id=img.id, tag_id=content.id, source="manual"))
db_sync.commit()
res = system_tag_auto_apply_sweep(db_sync, mode="process")
assert res["n_applied"] == 0
assert _source(db_sync, img.id, wip.id) is None
def test_process_sweep_flags_conflict_with_process_mode(db_sync):
# Guard 2: also scores high on a content head → still applied, but flagged
# for review with mode='process' (the ring-loud guard).
_enable_process(db_sync)
wip = _system_tag(db_sync, "wip")
_head(db_sync, wip.id, 0, weight=3.0)
content = Tag(name="looksreal", kind=TagKind.general)
db_sync.add(content)
db_sync.flush()
_head(db_sync, content.id, 0, weight=1.0)
img = _img(db_sync, "e" * 64, _emb(0))
db_sync.commit()
res = system_tag_auto_apply_sweep(db_sync, mode="process")
assert res["n_applied"] == 1
assert res["n_flagged"] == 1
flag = db_sync.execute(
select(PresentationReview).where(
PresentationReview.image_record_id == img.id,
PresentationReview.tag_id == wip.id,
)
).scalar_one()
assert flag.mode == "process"
assert flag.conflict_tag_id == content.id
def test_process_auto_source_never_trains_head(db_sync):
# The runaway break: provisional wip tags (process sweep 'process_auto', soft
# title 'wip_title_soft') are NOT training positives; a HARD title-heuristic /
# manual one IS. So the head learns only from trusted labels, never its own
# output or the low-precision sketch/doodle tier (#1464 + #1474).
wip = _system_tag(db_sync, "wip")
auto_img = _img(db_sync, "f" * 64, _emb(0))
soft_img = _img(db_sync, "9" * 64, _emb(2))
title_img = _img(db_sync, "0" * 64, _emb(1))
db_sync.execute(image_tag.insert().values(
image_record_id=auto_img.id, tag_id=wip.id, source="process_auto"))
db_sync.execute(image_tag.insert().values(
image_record_id=soft_img.id, tag_id=wip.id, source="wip_title_soft"))
db_sync.execute(image_tag.insert().values(
image_record_id=title_img.id, tag_id=wip.id, source="wip_title"))
db_sync.commit()
positives = set(_ids_with_tag(db_sync, wip.id))
assert title_img.id in positives # trusted HARD label trains the head
assert auto_img.id not in positives # its own auto-applied output does NOT
assert soft_img.id not in positives # low-precision soft tier does NOT
def test_soft_wip_conflict_audit_flags_ring_loud(db_sync):
# A soft-tagged image (sketch/doodle title) that ALSO scores high on a content
# head is probably finished art mis-tagged — flagged for review; a quiet one is not.
from backend.app.services.ml.heads import soft_wip_conflict_audit
s = db_sync.execute(select(MLSettings).where(MLSettings.id == 1)).scalar_one()
s.process_conflict_threshold = 0.6
wip = _system_tag(db_sync, "wip")
content = Tag(name="looksreal", kind=TagKind.general)
db_sync.add(content)
db_sync.flush()
_head(db_sync, content.id, 0, weight=1.0) # sigmoid(1)=0.73 > 0.6 conflict
ring = _img(db_sync, "1" * 64, _emb(0)) # scores on the content head
quiet = _img(db_sync, "2" * 64, _emb(5)) # orthogonal → 0.5 < 0.6
for img in (ring, quiet):
db_sync.execute(image_tag.insert().values(
image_record_id=img.id, tag_id=wip.id, source="wip_title_soft"))
db_sync.commit()
res = soft_wip_conflict_audit(db_sync)
assert res["n_flagged"] == 1
flag = db_sync.execute(
select(PresentationReview).where(PresentationReview.image_record_id == ring.id)
).scalar_one()
assert flag.mode == "process"
assert flag.conflict_tag_id == content.id
assert db_sync.execute(
select(PresentationReview).where(PresentationReview.image_record_id == quiet.id)
).scalar_one_or_none() is None