feat: duplicates resolve themselves where Lidarr says it is safe (M498)
release / govulncheck (push) Successful in 45s
release / web (push) Successful in 1m23s
release / go (push) Successful in 1m39s
release / integration (push) Successful in 4m25s
release / android (push) Successful in 6m17s
release / Build signed APK (releases and dev) (push) Successful in 5m57s
release / Attach APK to the Release (tag releases only) (push) Skipped
release / Build + push container image (push) Successful in 1m54s
release / Verify release artifacts (tag releases only) (push) Skipped

The duplicate sweep proposed 4,197 groups and every one waited for the
operator. Most are safe to settle, and Lidarr defines what safe means: it
maps one file to each track of the release it monitors and downloads any
mapped file that disappears. Deleting a mapped copy opens exactly the hole
the operator saw Lidarr fill.

Classify (#5435)
- Migration 0075: duplicate_groups.class (same_release, cross_release,
  mismatch, review), resolve_note, resolved_automatically;
  duplicate_group_members.lidarr_state (tracked, unmapped);
  fingerprint_settings.auto_resolve; notification kind
  duplicates_resolved with both kind CHECKs swapped (rule 36).
- library.ClassifyDuplicateGroup, with MatchTitleKey dropping featuring
  credits, remaster notes and video-rip markers, and keeping live, demo,
  remix and instrumental. The rip markers move from api to library.

Choose the copy to keep (#5436)
- ProposeSurvivor ranks the copy Lidarr maps first, then tag fit (a
  clash-free track number, no rip marker in the name, an MBID), then the
  quality rules. File size picked the wrong Humanz copy in 6 of 21 groups.

Act (#5437)
- An hourly resolver pass reads Lidarr's unmapped files, matched by the
  last three path components, and records each copy's state.
- Same album, with at most one copy mapped: merged into the mapped copy.
  The merge is guarded, so a mapped copy can never be removed
  (MergeDuplicateGroupGuarded, ErrCopyTrackedByLidarr).
- Same album, every copy mapped: the monitored release lists the song
  twice (Humanz's 14x12" box set). The pass moves Lidarr to the release
  that lists each song once and best covers what is on disk. It never
  picks one covering less, and is capped at 10 albums per pass.
  - Fixed point (lesson #4183): the chosen release no longer repeats.
  - The album is left alone for 24h while Lidarr rescans, so "every copy
    unmapped" mid-rescan is never read as licence to merge.
- Both actions are audited with no actor and summarised to admins. The
  operator can switch them off in the Fingerprinting card (rule 25).
- Manual merges use the same guard: 409 copy_tracked_by_lidarr, or 503
  lidarr_unavailable when Lidarr cannot say.

Web
- Duplicates gets tabs: Needs review, Across releases, Resolved
  automatically. Each loads as you scroll (rule 172), replacing the
  pager.
- Each copy says whether Lidarr uses it.
- The resolver's note shows on each group.
- The merge confirm blocks, before sending, a merge that would remove
  the copy Lidarr uses.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-10-08 21:49:37 -04:00
co-authored by Claude Opus 5.5
parent 01e2294471
commit 4ecff52f19
48 changed files with 3068 additions and 254 deletions
+238 -7
View File
@@ -62,6 +62,29 @@ func (q *Queries) CountPendingDuplicateGroups(ctx context.Context) (int64, error
return column_1, err
}
const countPendingDuplicateGroupsInView = `-- name: CountPendingDuplicateGroupsInView :one
SELECT count(*)::bigint
FROM duplicate_groups g
WHERE g.status = 'pending'
AND (SELECT count(*) FROM duplicate_group_members m WHERE m.group_id = g.id) >= 2
AND CASE $1::text
WHEN 'cross_release' THEN g.class = 'cross_release'
ELSE g.class IS DISTINCT FROM 'cross_release'
END
`
// CountPendingDuplicateGroups narrowed to one view of the report (M498):
//
// review what the operator still has to decide: every class but
// cross_release, and groups not yet classified
// cross_release the same song on different releases, which is kept, not merged
func (q *Queries) CountPendingDuplicateGroupsInView(ctx context.Context, view string) (int64, error) {
row := q.db.QueryRow(ctx, countPendingDuplicateGroupsInView, view)
var column_1 int64
err := row.Scan(&column_1)
return column_1, err
}
const deleteStalePendingDuplicateGroups = `-- name: DeleteStalePendingDuplicateGroups :execrows
DELETE FROM duplicate_groups g
WHERE g.status = 'pending'
@@ -187,6 +210,32 @@ func (q *Queries) GetLatestFingerprintComputedAt(ctx context.Context) (pgtype.Ti
return latest, err
}
const listAlbumPresentTrackTitles = `-- name: ListAlbumPresentTrackTitles :many
SELECT title FROM tracks WHERE album_id = $1 AND missing_since IS NULL
`
// The titles an album has on disk, for choosing the Lidarr release that covers
// them best (M498 #5437).
func (q *Queries) ListAlbumPresentTrackTitles(ctx context.Context, albumID pgtype.UUID) ([]string, error) {
rows, err := q.db.Query(ctx, listAlbumPresentTrackTitles, albumID)
if err != nil {
return nil, err
}
defer rows.Close()
var items []string
for rows.Next() {
var title string
if err := rows.Scan(&title); err != nil {
return nil, err
}
items = append(items, title)
}
if err := rows.Err(); err != nil {
return nil, err
}
return items, nil
}
const listDismissedDuplicateMemberSets = `-- name: ListDismissedDuplicateMemberSets :many
SELECT g.id, array_agg(m.track_id ORDER BY m.track_id)::uuid[] AS track_ids
FROM duplicate_groups g
@@ -288,6 +337,98 @@ func (q *Queries) ListDuplicateCandidates(ctx context.Context, arg ListDuplicate
return items, nil
}
const listDuplicateGroupsForResolve = `-- name: ListDuplicateGroupsForResolve :many
SELECT g.id AS group_id,
g.tier,
t.id AS track_id,
t.album_id,
t.title,
artists.name AS artist_name,
t.file_path,
t.file_format,
t.file_size,
t.added_at,
t.disc_number,
t.track_number,
(t.mbid IS NOT NULL)::boolean AS has_mbid,
albums.release_group_mbid,
albums.title AS album_title,
EXISTS (SELECT 1 FROM tracks o
WHERE o.album_id = t.album_id AND o.id <> t.id AND o.missing_since IS NULL
AND t.track_number IS NOT NULL
AND o.disc_number IS NOT DISTINCT FROM t.disc_number
AND o.track_number = t.track_number)::boolean AS position_clash
FROM duplicate_groups g
JOIN duplicate_group_members m ON m.group_id = g.id
JOIN tracks t ON t.id = m.track_id
JOIN albums ON albums.id = t.album_id
JOIN artists ON artists.id = t.artist_id
WHERE g.status = 'pending'
AND t.missing_since IS NULL
ORDER BY g.id, t.id
`
type ListDuplicateGroupsForResolveRow struct {
GroupID pgtype.UUID
Tier string
TrackID pgtype.UUID
AlbumID pgtype.UUID
Title string
ArtistName string
FilePath string
FileFormat string
FileSize int64
AddedAt pgtype.Timestamptz
DiscNumber *int32
TrackNumber *int32
HasMbid bool
ReleaseGroupMbid *string
AlbumTitle string
PositionClash bool
}
// Every pending group's present copies, for the resolver (M498 #5435): what it
// needs to classify the group and to choose the copy to keep. One row per copy,
// grouped by group. position_clash is true when another present track on the
// same album claims the same disc and track number: tags that do not fit the
// album, which the survivor rule weighs against a copy.
func (q *Queries) ListDuplicateGroupsForResolve(ctx context.Context) ([]ListDuplicateGroupsForResolveRow, error) {
rows, err := q.db.Query(ctx, listDuplicateGroupsForResolve)
if err != nil {
return nil, err
}
defer rows.Close()
var items []ListDuplicateGroupsForResolveRow
for rows.Next() {
var i ListDuplicateGroupsForResolveRow
if err := rows.Scan(
&i.GroupID,
&i.Tier,
&i.TrackID,
&i.AlbumID,
&i.Title,
&i.ArtistName,
&i.FilePath,
&i.FileFormat,
&i.FileSize,
&i.AddedAt,
&i.DiscNumber,
&i.TrackNumber,
&i.HasMbid,
&i.ReleaseGroupMbid,
&i.AlbumTitle,
&i.PositionClash,
); err != nil {
return nil, err
}
items = append(items, i)
}
if err := rows.Err(); err != nil {
return nil, err
}
return items, nil
}
const listExactDuplicateHashes = `-- name: ListExactDuplicateHashes :many
SELECT f.audio_stream_sha256,
array_agg(t.id ORDER BY t.id)::uuid[] AS track_ids
@@ -329,17 +470,32 @@ func (q *Queries) ListExactDuplicateHashes(ctx context.Context, currentVersion i
const listPendingDuplicateGroupMembers = `-- name: ListPendingDuplicateGroupMembers :many
WITH page AS (
SELECT g.id, g.tier, g.worst_bit_error_rate, g.detected_at
SELECT g.id, g.tier, g.worst_bit_error_rate, g.detected_at, g.class, g.resolve_note
FROM duplicate_groups g
WHERE g.status = 'pending'
AND (SELECT count(*) FROM duplicate_group_members m WHERE m.group_id = g.id) >= 2
AND CASE $1::text
WHEN 'cross_release' THEN g.class = 'cross_release'
ELSE g.class IS DISTINCT FROM 'cross_release'
END
ORDER BY g.detected_at DESC, g.id
LIMIT $2 OFFSET $1
LIMIT $3 OFFSET $2
)
SELECT p.id AS group_id,
p.tier,
p.worst_bit_error_rate,
p.detected_at,
p.class,
p.resolve_note,
m.lidarr_state,
t.disc_number,
t.track_number,
(t.mbid IS NOT NULL)::boolean AS has_mbid,
EXISTS (SELECT 1 FROM tracks o
WHERE o.album_id = t.album_id AND o.id <> t.id AND o.missing_since IS NULL
AND t.track_number IS NOT NULL
AND o.disc_number IS NOT DISTINCT FROM t.disc_number
AND o.track_number = t.track_number)::boolean AS position_clash,
t.id AS track_id,
t.title,
artists.name AS artist_name,
@@ -361,6 +517,7 @@ SELECT p.id AS group_id,
`
type ListPendingDuplicateGroupMembersParams struct {
View string
PageOffset int32
PageLimit int32
}
@@ -370,6 +527,13 @@ type ListPendingDuplicateGroupMembersRow struct {
Tier string
WorstBitErrorRate *float32
DetectedAt pgtype.Timestamptz
Class *string
ResolveNote *string
LidarrState *string
DiscNumber *int32
TrackNumber *int32
HasMbid bool
PositionClash bool
TrackID pgtype.UUID
Title string
ArtistName string
@@ -384,12 +548,14 @@ type ListPendingDuplicateGroupMembersRow struct {
PlayCount int64
}
// One page of proposals, newest first, flattened to one row per member so the
// handler folds them without a query per group. What each copy carries — likes
// and plays from every user — is here because it is what the operator weighs
// when deciding which copy to keep.
// One page of proposals in one view (see CountPendingDuplicateGroupsInView),
// newest first, flattened to one row per member so the handler folds them
// without a query per group. What each copy carries — likes and plays from
// every user — is here because it is what the operator weighs when deciding
// which copy to keep; Lidarr's view of it, because a copy Lidarr tracks cannot
// be removed without Lidarr downloading it again.
func (q *Queries) ListPendingDuplicateGroupMembers(ctx context.Context, arg ListPendingDuplicateGroupMembersParams) ([]ListPendingDuplicateGroupMembersRow, error) {
rows, err := q.db.Query(ctx, listPendingDuplicateGroupMembers, arg.PageOffset, arg.PageLimit)
rows, err := q.db.Query(ctx, listPendingDuplicateGroupMembers, arg.View, arg.PageOffset, arg.PageLimit)
if err != nil {
return nil, err
}
@@ -402,6 +568,13 @@ func (q *Queries) ListPendingDuplicateGroupMembers(ctx context.Context, arg List
&i.Tier,
&i.WorstBitErrorRate,
&i.DetectedAt,
&i.Class,
&i.ResolveNote,
&i.LidarrState,
&i.DiscNumber,
&i.TrackNumber,
&i.HasMbid,
&i.PositionClash,
&i.TrackID,
&i.Title,
&i.ArtistName,
@@ -425,6 +598,64 @@ func (q *Queries) ListPendingDuplicateGroupMembers(ctx context.Context, arg List
return items, nil
}
const markDuplicateGroupResolvedAutomatically = `-- name: MarkDuplicateGroupResolvedAutomatically :exec
UPDATE duplicate_groups SET resolved_automatically = true WHERE id = $1
`
func (q *Queries) MarkDuplicateGroupResolvedAutomatically(ctx context.Context, id pgtype.UUID) error {
_, err := q.db.Exec(ctx, markDuplicateGroupResolvedAutomatically, id)
return err
}
const setDuplicateGroupClasses = `-- name: SetDuplicateGroupClasses :exec
UPDATE duplicate_groups g
SET class = v.class, resolve_note = NULLIF(v.note, '')
FROM (SELECT unnest($1::uuid[]) AS id,
unnest($2::text[]) AS class,
unnest($3::text[]) AS note) v
WHERE g.id = v.id
AND g.status = 'pending'
AND (g.class IS DISTINCT FROM v.class OR g.resolve_note IS DISTINCT FROM NULLIF(v.note, ''))
`
type SetDuplicateGroupClassesParams struct {
GroupIds []pgtype.UUID
Classes []string
Notes []string
}
// The resolver's verdicts, written in one statement. Only rows whose class or
// note actually changed are touched, so an hourly pass over thousands of
// unchanged groups writes nothing. An empty note clears it.
func (q *Queries) SetDuplicateGroupClasses(ctx context.Context, arg SetDuplicateGroupClassesParams) error {
_, err := q.db.Exec(ctx, setDuplicateGroupClasses, arg.GroupIds, arg.Classes, arg.Notes)
return err
}
const setDuplicateMemberLidarrStates = `-- name: SetDuplicateMemberLidarrStates :exec
UPDATE duplicate_group_members m
SET lidarr_state = NULLIF(v.state, '')
FROM (SELECT unnest($1::uuid[]) AS group_id,
unnest($2::uuid[]) AS track_id,
unnest($3::text[]) AS state) v
WHERE m.group_id = v.group_id
AND m.track_id = v.track_id
AND m.lidarr_state IS DISTINCT FROM NULLIF(v.state, '')
`
type SetDuplicateMemberLidarrStatesParams struct {
GroupIds []pgtype.UUID
TrackIds []pgtype.UUID
States []string
}
// Lidarr's view of each copy as of this pass; an empty state clears it (Lidarr
// not consulted). Unchanged rows are left alone.
func (q *Queries) SetDuplicateMemberLidarrStates(ctx context.Context, arg SetDuplicateMemberLidarrStatesParams) error {
_, err := q.db.Exec(ctx, setDuplicateMemberLidarrStates, arg.GroupIds, arg.TrackIds, arg.States)
return err
}
const startDuplicateSweep = `-- name: StartDuplicateSweep :one
INSERT INTO duplicate_sweeps DEFAULT VALUES RETURNING id, started_at
`