Files
minstrel/internal/api/admin_duplicates.go
T
bvandeusenandClaude Opus 5.5 5b372f61d7
release / govulncheck (push) Successful in 21s
release / web (push) Successful in 1m7s
release / go (push) Successful in 1m29s
release / integration (push) Successful in 4m11s
release / Attach APK to the Release (tag releases only) (push) Canceled after 0s
release / Build + push container image (push) Canceled after 0s
release / Verify release artifacts (tag releases only) (push) Canceled after 0s
release / android (push) Canceled after 5m58s
release / Build signed APK (releases and dev) (push) Canceled after 5m59s
feat: the same song on several releases counts as one song (M498 #5438)
A single and the album it is on stay two files, since each fulfils its own
release in Lidarr, but they are one song to the listener.

- tracks.song_id links copies; the generated song_key (song_id, else the
  track's own id) is what they share (migration 0076).
- The resolver links each cross-release group every pass (idempotent; only
  with auto-resolve on) and, when a link is new, shares existing likes across
  the song and logs them for sync.
- A like or unlike (web and Subsonic) reaches every copy; each change is
  logged and published so clients update every heart.
- The Liked list and its count show the song once; Shuffle and the mix writer
  take one copy per song.
- A merge keeps the removed copy's song link; "Not the same song" on the
  Across releases tab dismisses the group and undoes the link.

Shared plays ("heard via another copy") are left for a later step.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-08 22:07:55 -04:00

467 lines
18 KiB
Go

package api
import (
"context"
"encoding/json"
"errors"
"io"
"net/http"
"github.com/go-chi/chi/v5"
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgtype"
"git.fabledsword.com/bvandeusen/minstrel/internal/audit"
"git.fabledsword.com/bvandeusen/minstrel/internal/db/dbq"
"git.fabledsword.com/bvandeusen/minstrel/internal/library"
"git.fabledsword.com/bvandeusen/minstrel/internal/tracks"
)
// duplicateMemberView is one copy in a proposed duplicate group. LikeCount and
// PlayCount span every user: the report is admin-only, and what a copy carries
// is the fact the operator weighs when choosing which to keep.
type duplicateMemberView struct {
TrackID string `json:"track_id"`
Title string `json:"title"`
ArtistName string `json:"artist_name"`
AlbumID string `json:"album_id"`
AlbumTitle string `json:"album_title"`
FilePath string `json:"file_path"`
FileFormat string `json:"file_format"`
FileSize int64 `json:"file_size"`
DurationSec int32 `json:"duration_sec"`
AddedAt string `json:"added_at"`
LikeCount int64 `json:"like_count"`
PlayCount int64 `json:"play_count"`
// LidarrState is Lidarr's view of the file as of the resolver's last pass
// (M498): "tracked" (mapped to a track, so it cannot be removed without
// Lidarr downloading it again), "unmapped", or null when not asked.
LidarrState *string `json:"lidarr_state"`
DiscNumber *int32 `json:"disc_number"`
TrackNumber *int32 `json:"track_number"`
}
// duplicateGroupView is one proposal. SurvivorTrackID and SurvivorReason are
// the copy the report proposes keeping and the rule that chose it
// (library.ProposeSurvivor) — a default the merge (#3911) lets the operator
// override.
type duplicateGroupView struct {
ID string `json:"id"`
Tier string `json:"tier"`
WorstBitErrorRate *float32 `json:"worst_bit_error_rate"`
DetectedAt string `json:"detected_at"`
SurvivorTrackID string `json:"survivor_track_id"`
SurvivorReason string `json:"survivor_reason"`
// Class is the resolver's verdict (library.DuplicateClass), null until its
// first pass; ResolveNote is why it left the group for the operator.
Class *string `json:"class"`
ResolveNote *string `json:"resolve_note"`
Members []duplicateMemberView `json:"members"`
}
// duplicateSweepView is the latest sweep. State is "never" when none has run,
// which is what lets the page tell an empty report apart from a sweep that
// found nothing.
type duplicateSweepView struct {
State string `json:"state"`
StartedAt *string `json:"started_at"`
FinishedAt *string `json:"finished_at"`
Candidates *int32 `json:"candidates"`
GroupsFound *int32 `json:"groups_found"`
OversizeClusters *int32 `json:"oversize_clusters"`
ErrorMessage *string `json:"error_message"`
}
// duplicateViewCounts sizes each tab of the report.
type duplicateViewCounts struct {
Review int64 `json:"review"`
CrossRelease int64 `json:"cross_release"`
Resolved int64 `json:"resolved"`
}
// Report views (M498). cross_release groups are the same song on different
// releases, kept on purpose; everything else pending needs the operator.
const (
duplicateViewReview = "review"
duplicateViewCrossRelease = "cross_release"
)
// adminDuplicatesResponse is the paged report. Total counts groups in View.
type adminDuplicatesResponse struct {
Sweep duplicateSweepView `json:"sweep"`
Fingerprints fingerprintCoverageResp `json:"fingerprints"`
View string `json:"view"`
Counts duplicateViewCounts `json:"counts"`
Total int64 `json:"total"`
Limit int `json:"limit"`
Offset int `json:"offset"`
Groups []duplicateGroupView `json:"groups"`
}
// handleListDuplicates implements GET /api/admin/library/duplicates (#3912).
// ?view=review (the default) or cross_release picks the tab (M498).
//
// Read-only. The sweep's state and the fingerprint backfill's progress travel
// with the groups because an empty report means three different things — still
// fingerprinting, never swept, or swept and clean — and the page has to say which.
func (h *handlers) handleListDuplicates(w http.ResponseWriter, r *http.Request) {
limit, offset, err := parsePaging(r.URL.Query())
if err != nil {
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_paging")
return
}
view := r.URL.Query().Get("view")
if view == "" {
view = duplicateViewReview
}
if view != duplicateViewReview && view != duplicateViewCrossRelease {
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_view")
return
}
ctx := r.Context()
q := dbq.New(h.pool)
sweep := duplicateSweepView{State: "never"}
last, err := q.GetLatestDuplicateSweep(ctx)
switch {
case err == nil:
sweep = duplicateSweepViewOf(last)
case !errors.Is(err, pgx.ErrNoRows):
h.logger.Error("admin: latest duplicate sweep", "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
return
}
cov, err := h.fingerprintCoverage(ctx)
if err != nil {
h.logger.Error("admin: fingerprint coverage", "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
return
}
counts, err := duplicateCounts(ctx, q)
if err != nil {
h.logger.Error("admin: count duplicate groups", "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
return
}
total := counts.Review
if view == duplicateViewCrossRelease {
total = counts.CrossRelease
}
rows, err := q.ListPendingDuplicateGroupMembers(ctx, dbq.ListPendingDuplicateGroupMembersParams{
View: view, PageLimit: int32(limit), PageOffset: int32(offset),
})
if err != nil {
h.logger.Error("admin: list duplicate groups", "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
return
}
writeJSON(w, http.StatusOK, adminDuplicatesResponse{
Sweep: sweep,
Fingerprints: cov,
View: view,
Counts: counts,
Total: total,
Limit: limit,
Offset: offset,
Groups: foldDuplicateGroups(rows),
})
}
// resolvedActions are the audit actions the "Resolved automatically" tab lists.
var resolvedActions = []string{string(audit.ActionDuplicateMerge), string(audit.ActionLidarrReleaseChange)}
func duplicateCounts(ctx context.Context, q *dbq.Queries) (duplicateViewCounts, error) {
var c duplicateViewCounts
var err error
if c.Review, err = q.CountPendingDuplicateGroupsInView(ctx, duplicateViewReview); err != nil {
return c, err
}
if c.CrossRelease, err = q.CountPendingDuplicateGroupsInView(ctx, duplicateViewCrossRelease); err != nil {
return c, err
}
c.Resolved, err = q.CountAuditLogByActions(ctx, dbq.CountAuditLogByActionsParams{Actions: resolvedActions, SystemOnly: true})
return c, err
}
// resolvedActivityView is one thing the resolver did on its own: a merge
// (action duplicate_merge) or a Lidarr release change (lidarr_release_change).
// Details is the audit row's metadata, which names the files or releases.
type resolvedActivityView struct {
ID string `json:"id"`
Action string `json:"action"`
CreatedAt string `json:"created_at"`
Details json.RawMessage `json:"details"`
}
type resolvedActivityResponse struct {
Total int64 `json:"total"`
Limit int `json:"limit"`
Offset int `json:"offset"`
Items []resolvedActivityView `json:"items"`
}
// handleListResolvedDuplicates implements GET
// /api/admin/library/duplicates/resolved (M498): what the resolver did by
// itself, newest first, read from the audit log so it outlives the groups.
func (h *handlers) handleListResolvedDuplicates(w http.ResponseWriter, r *http.Request) {
limit, offset, err := parsePaging(r.URL.Query())
if err != nil {
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_paging")
return
}
q := dbq.New(h.pool)
total, err := q.CountAuditLogByActions(r.Context(), dbq.CountAuditLogByActionsParams{Actions: resolvedActions, SystemOnly: true})
if err != nil {
h.logger.Error("admin: count resolved duplicates", "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
return
}
rows, err := q.ListAuditLogByActions(r.Context(), dbq.ListAuditLogByActionsParams{
Actions: resolvedActions, SystemOnly: true, PageLimit: int32(limit), PageOffset: int32(offset),
})
if err != nil {
h.logger.Error("admin: list resolved duplicates", "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
return
}
items := make([]resolvedActivityView, 0, len(rows))
for _, row := range rows {
details := json.RawMessage(row.Metadata)
if len(details) == 0 {
details = json.RawMessage("{}")
}
items = append(items, resolvedActivityView{
ID: uuidToString(row.ID), Action: row.Action, CreatedAt: formatTimestamp(row.CreatedAt), Details: details,
})
}
writeJSON(w, http.StatusOK, resolvedActivityResponse{Total: total, Limit: limit, Offset: offset, Items: items})
}
func duplicateSweepViewOf(s dbq.DuplicateSweep) duplicateSweepView {
v := duplicateSweepView{
State: "running",
Candidates: s.Candidates,
GroupsFound: s.GroupsFound,
OversizeClusters: s.OversizeClusters,
ErrorMessage: s.ErrorMessage,
}
started := formatTimestamp(s.StartedAt)
v.StartedAt = &started
if s.FinishedAt.Valid {
finished := formatTimestamp(s.FinishedAt)
v.FinishedAt = &finished
v.State = "finished"
}
return v
}
// foldDuplicateGroups folds the one-row-per-member query result into groups and
// proposes each group's survivor. It relies on the query ordering members of a
// group together, so a run-length fold is enough and the page order holds.
func foldDuplicateGroups(rows []dbq.ListPendingDuplicateGroupMembersRow) []duplicateGroupView {
groups := make([]duplicateGroupView, 0, 8)
var candidates [][]library.SurvivorCandidate
for _, row := range rows {
id := uuidToString(row.GroupID)
if n := len(groups); n == 0 || groups[n-1].ID != id {
groups = append(groups, duplicateGroupView{
ID: id,
Tier: row.Tier,
WorstBitErrorRate: row.WorstBitErrorRate,
DetectedAt: formatTimestamp(row.DetectedAt),
Class: row.Class,
ResolveNote: row.ResolveNote,
})
candidates = append(candidates, nil)
}
n := len(groups) - 1
trackID := uuidToString(row.TrackID)
groups[n].Members = append(groups[n].Members, duplicateMemberView{
TrackID: trackID,
Title: row.Title,
ArtistName: row.ArtistName,
AlbumID: uuidToString(row.AlbumID),
AlbumTitle: row.AlbumTitle,
FilePath: row.FilePath,
FileFormat: row.FileFormat,
FileSize: row.FileSize,
DurationSec: row.DurationMs / 1000,
AddedAt: formatTimestamp(row.AddedAt),
LikeCount: row.LikeCount,
PlayCount: row.PlayCount,
LidarrState: row.LidarrState,
DiscNumber: row.DiscNumber,
TrackNumber: row.TrackNumber,
})
candidates[n] = append(candidates[n], library.SurvivorCandidate{
TrackID: trackID, FileFormat: row.FileFormat, FileSize: row.FileSize, AddedAt: row.AddedAt.Time,
LidarrTracked: row.LidarrState != nil && *row.LidarrState == "tracked",
TagFit: library.TagFitScore(row.TrackNumber != nil, row.PositionClash, row.FilePath, row.HasMbid),
})
}
for i := range groups {
groups[i].SurvivorTrackID, groups[i].SurvivorReason = library.ProposeSurvivor(candidates[i])
}
return groups
}
// handleRunDuplicateSweep implements POST /api/admin/library/duplicates/sweep:
// 202 when a sweep starts, 409 sweep_in_progress when one is already running.
// The sweep outlives the request, so it runs on a background context, as
// handleTriggerScan's scan does.
func (h *handlers) handleRunDuplicateSweep(w http.ResponseWriter, _ *http.Request) {
// Runs whatever the sweep interval says: the interval paces the automatic
// sweep, and an operator pressing the button has already decided.
started, err := library.TryStartDuplicateSweep(
context.Background(), h.pool, h.logger.With("source", "manual"), h.fingerprintSettings.Get(),
)
if err != nil {
h.logger.Error("admin: start duplicate sweep", "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
return
}
if !started {
writeAdminJSONErr(w, http.StatusConflict, "sweep_in_progress")
return
}
writeJSON(w, http.StatusAccepted, map[string]bool{"started": true})
}
// handleDismissDuplicateGroup implements POST
// /api/admin/library/duplicates/{id}/dismiss: "these are not duplicates". The
// sweep keeps the dismissal and will not propose that set of tracks again. On a
// cross-release group it also undoes the song link (M498 #5438): the copies
// stop sharing likes from here on. 404 duplicate_group_not_pending when the
// group was already resolved or is gone.
func (h *handlers) handleDismissDuplicateGroup(w http.ResponseWriter, r *http.Request) {
id, ok := parseUUID(chi.URLParam(r, "id"))
if !ok {
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_id")
return
}
n, err := h.dismissDuplicateGroup(r.Context(), id)
if err != nil {
h.logger.Error("admin: dismiss duplicate group", "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
return
}
if n == 0 {
writeAdminJSONErr(w, http.StatusNotFound, "duplicate_group_not_pending")
return
}
writeJSON(w, http.StatusOK, map[string]string{"status": "dismissed"})
}
// dismissDuplicateGroup dismisses the group and unlinks its copies as one song,
// together.
func (h *handlers) dismissDuplicateGroup(ctx context.Context, id pgtype.UUID) (int64, error) {
tx, err := h.pool.Begin(ctx)
if err != nil {
return 0, err
}
defer func() { _ = tx.Rollback(ctx) }()
tq := dbq.New(tx)
n, err := tq.DismissDuplicateGroup(ctx, id)
if err != nil || n == 0 {
return n, err
}
if err := tq.UnlinkDuplicateGroupSongs(ctx, id); err != nil {
return 0, err
}
return n, tx.Commit(ctx)
}
// mergeDuplicateRequest chooses the copy to keep. An empty survivor_track_id
// keeps the report's proposal.
type mergeDuplicateRequest struct {
SurvivorTrackID string `json:"survivor_track_id"`
Unmonitor bool `json:"unmonitor"`
}
// mergeDuplicateResponse reports what the merge removed. RemovedPaths are files
// deleted from disk; the operator reads them to know exactly what went.
type mergeDuplicateResponse struct {
SurvivorTrackID string `json:"survivor_track_id"`
RemovedPaths []string `json:"removed_paths"`
LidarrUnmonitorFailed *bool `json:"lidarr_unmonitor_failed,omitempty"`
}
// mergeRequestBodyLimit bounds the request body. It holds one id and a flag.
const mergeRequestBodyLimit = 1 << 16
// handleMergeDuplicateGroup implements POST /api/admin/library/duplicates/{id}/merge
// (#3911): keep one copy, move the others' likes, plays and playlist entries onto
// it, and delete the others' files and rows.
//
// Errors:
// - 409 library_not_writable / 500 file_delete_failed when a file could not be
// removed — nothing was changed (fileRemoveAPIError)
// - 409 copy_tracked_by_lidarr when a copy to remove is one Lidarr maps to a
// track: removing it would make Lidarr download it again (M498)
// - 503 lidarr_unavailable when Lidarr is on but did not answer, so that
// cannot be checked
// - 404 duplicate_group_not_pending when the group was already resolved
// - 400 survivor_not_in_group, invalid_id, invalid_body
func (h *handlers) handleMergeDuplicateGroup(w http.ResponseWriter, r *http.Request) {
admin, ok := requireUser(w, r)
if !ok {
return
}
groupID, ok := parseUUID(chi.URLParam(r, "id"))
if !ok {
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_id")
return
}
var body mergeDuplicateRequest
if err := json.NewDecoder(http.MaxBytesReader(w, r.Body, mergeRequestBodyLimit)).Decode(&body); err != nil && !errors.Is(err, io.EOF) {
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_body")
return
}
var survivorID pgtype.UUID // invalid: keep the proposal
if body.SurvivorTrackID != "" {
if survivorID, ok = parseUUID(body.SurvivorTrackID); !ok {
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_id")
return
}
}
res, unmonitorFailed, err := h.tracks.MergeDuplicates(r.Context(), groupID, survivorID, admin.ID, body.Unmonitor)
if err != nil {
if apiErr, ok := fileRemoveAPIError(err); ok {
logFileRemoveFailure(h.logger, apiErr, "group_id", uuidToString(groupID))
writeErr(w, apiErr)
return
}
switch {
case errors.Is(err, library.ErrDuplicateGroupNotPending):
writeAdminJSONErr(w, http.StatusNotFound, "duplicate_group_not_pending")
case errors.Is(err, library.ErrSurvivorNotInGroup):
writeAdminJSONErr(w, http.StatusBadRequest, "survivor_not_in_group")
case errors.Is(err, library.ErrCopyTrackedByLidarr):
writeAdminJSONErr(w, http.StatusConflict, "copy_tracked_by_lidarr")
case errors.Is(err, tracks.ErrLidarrUnavailable):
h.logger.Warn("admin: merge duplicate group: lidarr unavailable", "group_id", uuidToString(groupID), "err", err)
writeAdminJSONErr(w, http.StatusServiceUnavailable, "lidarr_unavailable")
default:
h.logger.Error("admin: merge duplicate group", "group_id", uuidToString(groupID), "err", err)
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
}
return
}
resp := mergeDuplicateResponse{
SurvivorTrackID: uuidToString(res.Survivor.TrackID),
RemovedPaths: make([]string, 0, len(res.Removed)),
}
for _, c := range res.Removed {
resp.RemovedPaths = append(resp.RemovedPaths, c.FilePath)
}
if body.Unmonitor && unmonitorFailed {
failed := true
resp.LidarrUnmonitorFailed = &failed
}
writeJSON(w, http.StatusOK, resp)
}