package api import ( "net/http" "path" "regexp" "strings" "git.fabledsword.com/bvandeusen/minstrel/internal/db/dbq" ) // sourceMarker is one sign that a file was ripped from a video rather than // taken from a release: words a video title carries and an album track does // not (#5410). // // Each pattern is used twice, so it is written in the subset that Postgres's // regex flavour and Go's RE2 read the same way: no \b (Postgres spells word // boundaries \y), no lookaround, and brackets written as (\(|\[) rather than // as a bracket expression. The SQL side matches it case-insensitively with ~*, // the Go side with (?i); [^a-zA-Z] is spelled out so neither has to fold case // inside a negated class. type sourceMarker struct { label string pattern string } // suspectSourceMarkers was calibrated against the operator's library on // 2026-10-08. Two candidates were left out on purpose: // - "live in"/"live at": 229 files, nearly all from real live albums. // - a bare "reaction": it caught Beck's "Chain Reaction". The marker below // wants the phrases a reaction video actually uses. var suspectSourceMarkers = []sourceMarker{ {"music video", `official[^a-zA-Z]*(music[^a-zA-Z]*)?video|music[^a-zA-Z]*video`}, {"official audio", `official[^a-zA-Z]*audio`}, {"lyric video", `lyrics?[^a-zA-Z]*video`}, {"visualiser", `visuali[sz]er`}, {"reaction", `reaction[^a-zA-Z]*(video|mashup)|reacts?[^a-zA-Z]+to[^a-zA-Z]|first[^a-zA-Z]*time[^a-zA-Z]*(hearing|listening)`}, {"MV", `(^|[^a-zA-Z])(mv|m/v)([^a-zA-Z]|$)`}, {"[Audio]", `(\(|\[)audio(\)|\])`}, {"[HD]", `(\(|\[)(hd|hq|4k)(\)|\])`}, } // suspectSourcePattern is every marker as one alternation, for the SQL filter. var suspectSourcePattern = func() string { parts := make([]string, len(suspectSourceMarkers)) for i, m := range suspectSourceMarkers { parts[i] = "(" + m.pattern + ")" } return strings.Join(parts, "|") }() var suspectSourceRegexps = func() []*regexp.Regexp { out := make([]*regexp.Regexp, len(suspectSourceMarkers)) for i, m := range suspectSourceMarkers { out[i] = regexp.MustCompile("(?i)" + m.pattern) } return out }() // sourceMarkersFor returns the labels of every marker the file's basename // carries, in list order. Empty for a file the SQL filter would not return. func sourceMarkersFor(filePath string) []string { base := path.Base(filePath) labels := make([]string, 0, 2) for i, re := range suspectSourceRegexps { if re.MatchString(base) { labels = append(labels, suspectSourceMarkers[i].label) } } return labels } // suspectTrackView is one flagged track. FilePath is the evidence: the // markers are read from its basename, and the operator needs to see it to // judge whether the flag is right. type suspectTrackView struct { TrackID string `json:"track_id"` Title string `json:"title"` ArtistID string `json:"artist_id"` ArtistName string `json:"artist_name"` AlbumID string `json:"album_id"` AlbumTitle string `json:"album_title"` FilePath string `json:"file_path"` DurationSec int32 `json:"duration_sec"` DiscNumber *int32 `json:"disc_number"` TrackNumber *int32 `json:"track_number"` Markers []string `json:"markers"` } // suspectGroupView is a folder's worth of flagged tracks. As on the // missing-files report, the folder is the unit of decision: Humanz was 62 rows // in one folder, which is one problem to look at, not 62. type suspectGroupView struct { Directory string `json:"directory"` Tracks []suspectTrackView `json:"tracks"` } // adminSuspectSourcesResponse is the paged envelope. Total counts tracks. type adminSuspectSourcesResponse struct { Total int64 `json:"total"` Limit int `json:"limit"` Offset int `json:"offset"` Groups []suspectGroupView `json:"groups"` } // handleListSuspectSources implements GET /api/admin/library/suspect-sources. // // Read-only. A marker is a reason to look, not proof: a band can title a song // "Music Video". What to do about a flagged file — merge it in Duplicates, // quarantine it, replace it in Lidarr — stays the operator's call. func (h *handlers) handleListSuspectSources(w http.ResponseWriter, r *http.Request) { limit, offset, err := parsePaging(r.URL.Query()) if err != nil { writeAdminJSONErr(w, http.StatusBadRequest, "invalid_paging") return } q := dbq.New(h.pool) total, err := q.CountSuspectSourceTracks(r.Context(), suspectSourcePattern) if err != nil { h.logger.Error("admin: count suspect-source tracks", "err", err) writeAdminJSONErr(w, http.StatusInternalServerError, "server_error") return } rows, err := q.ListSuspectSourceTracks(r.Context(), dbq.ListSuspectSourceTracksParams{ Pattern: suspectSourcePattern, PageLimit: int32(limit), PageOffset: int32(offset), }) if err != nil { h.logger.Error("admin: list suspect-source tracks", "err", err) writeAdminJSONErr(w, http.StatusInternalServerError, "server_error") return } writeJSON(w, http.StatusOK, adminSuspectSourcesResponse{ Total: total, Limit: limit, Offset: offset, Groups: groupSuspectByDirectory(rows), }) } // groupSuspectByDirectory folds the directory-ordered rows into per-folder // groups, the same run-length fold as groupMissingByDirectory. A page boundary // can split a folder; the web page joins the halves. func groupSuspectByDirectory(rows []dbq.ListSuspectSourceTracksRow) []suspectGroupView { groups := make([]suspectGroupView, 0, 8) for _, row := range rows { t := suspectTrackView{ TrackID: uuidToString(row.ID), Title: row.Title, ArtistID: uuidToString(row.ArtistID), ArtistName: row.ArtistName, AlbumID: uuidToString(row.AlbumID), AlbumTitle: row.AlbumTitle, FilePath: row.FilePath, DurationSec: row.DurationMs / 1000, DiscNumber: row.DiscNumber, TrackNumber: row.TrackNumber, Markers: sourceMarkersFor(row.FilePath), } if n := len(groups); n > 0 && groups[n-1].Directory == row.Directory { groups[n-1].Tracks = append(groups[n-1].Tracks, t) continue } groups = append(groups, suspectGroupView{ Directory: row.Directory, Tracks: []suspectTrackView{t}, }) } return groups }