release / govulncheck (push) Successful in 37s
release / go (push) Failing after 1m10s
release / web (push) Failing after 1m24s
release / integration (push) Successful in 4m38s
release / android (push) Successful in 5m5s
release / Attach APK to the Release (tag releases only) (push) Canceled after 0s
release / Build + push container image (push) Canceled after 0s
release / Verify release artifacts (tag releases only) (push) Canceled after 0s
release / Build signed APK (releases and dev) (push) Canceled after 5m18s
Humanz turned out to be YouTube rips: "(Official Video)", "Visualizer", a reaction video filed as a song, junk disc numbers (#5401). The library holds about 200 more files named the same way. This report lists them, grouped by folder like Missing files, with the markers each file name carries. - GET /api/admin/library/suspect-sources: one marker list in Go builds both the Postgres ~* filter and each row's labels, so they cannot drift. Basename only; missing files are left out. - Markers calibrated on the live library: "live in/at" dropped (real live albums), "reaction" narrowed (it caught "Chain Reaction"). - Admin tab "Suspect sources", read-only, loads as you scroll; a folder split across pages is joined back into one. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
174 lines
6.1 KiB
Go
174 lines
6.1 KiB
Go
package api
|
|
|
|
import (
|
|
"net/http"
|
|
"path"
|
|
"regexp"
|
|
"strings"
|
|
|
|
"git.fabledsword.com/bvandeusen/minstrel/internal/db/dbq"
|
|
)
|
|
|
|
// sourceMarker is one sign that a file was ripped from a video rather than
|
|
// taken from a release: words a video title carries and an album track does
|
|
// not (#5410).
|
|
//
|
|
// Each pattern is used twice, so it is written in the subset that Postgres's
|
|
// regex flavour and Go's RE2 read the same way: no \b (Postgres spells word
|
|
// boundaries \y), no lookaround, and brackets written as (\(|\[) rather than
|
|
// as a bracket expression. The SQL side matches it case-insensitively with ~*,
|
|
// the Go side with (?i); [^a-zA-Z] is spelled out so neither has to fold case
|
|
// inside a negated class.
|
|
type sourceMarker struct {
|
|
label string
|
|
pattern string
|
|
}
|
|
|
|
// suspectSourceMarkers was calibrated against the operator's library on
|
|
// 2026-10-08. Two candidates were left out on purpose:
|
|
// - "live in"/"live at": 229 files, nearly all from real live albums.
|
|
// - a bare "reaction": it caught Beck's "Chain Reaction". The marker below
|
|
// wants the phrases a reaction video actually uses.
|
|
var suspectSourceMarkers = []sourceMarker{
|
|
{"music video", `official[^a-zA-Z]*(music[^a-zA-Z]*)?video|music[^a-zA-Z]*video`},
|
|
{"official audio", `official[^a-zA-Z]*audio`},
|
|
{"lyric video", `lyrics?[^a-zA-Z]*video`},
|
|
{"visualiser", `visuali[sz]er`},
|
|
{"reaction", `reaction[^a-zA-Z]*(video|mashup)|reacts?[^a-zA-Z]+to[^a-zA-Z]|first[^a-zA-Z]*time[^a-zA-Z]*(hearing|listening)`},
|
|
{"MV", `(^|[^a-zA-Z])(mv|m/v)([^a-zA-Z]|$)`},
|
|
{"[Audio]", `(\(|\[)audio(\)|\])`},
|
|
{"[HD]", `(\(|\[)(hd|hq|4k)(\)|\])`},
|
|
}
|
|
|
|
// suspectSourcePattern is every marker as one alternation, for the SQL filter.
|
|
var suspectSourcePattern = func() string {
|
|
parts := make([]string, len(suspectSourceMarkers))
|
|
for i, m := range suspectSourceMarkers {
|
|
parts[i] = "(" + m.pattern + ")"
|
|
}
|
|
return strings.Join(parts, "|")
|
|
}()
|
|
|
|
var suspectSourceRegexps = func() []*regexp.Regexp {
|
|
out := make([]*regexp.Regexp, len(suspectSourceMarkers))
|
|
for i, m := range suspectSourceMarkers {
|
|
out[i] = regexp.MustCompile("(?i)" + m.pattern)
|
|
}
|
|
return out
|
|
}()
|
|
|
|
// sourceMarkersFor returns the labels of every marker the file's basename
|
|
// carries, in list order. Empty for a file the SQL filter would not return.
|
|
func sourceMarkersFor(filePath string) []string {
|
|
base := path.Base(filePath)
|
|
labels := make([]string, 0, 2)
|
|
for i, re := range suspectSourceRegexps {
|
|
if re.MatchString(base) {
|
|
labels = append(labels, suspectSourceMarkers[i].label)
|
|
}
|
|
}
|
|
return labels
|
|
}
|
|
|
|
// suspectTrackView is one flagged track. FilePath is the evidence: the
|
|
// markers are read from its basename, and the operator needs to see it to
|
|
// judge whether the flag is right.
|
|
type suspectTrackView struct {
|
|
TrackID string `json:"track_id"`
|
|
Title string `json:"title"`
|
|
ArtistID string `json:"artist_id"`
|
|
ArtistName string `json:"artist_name"`
|
|
AlbumID string `json:"album_id"`
|
|
AlbumTitle string `json:"album_title"`
|
|
FilePath string `json:"file_path"`
|
|
DurationSec int32 `json:"duration_sec"`
|
|
DiscNumber *int32 `json:"disc_number"`
|
|
TrackNumber *int32 `json:"track_number"`
|
|
Markers []string `json:"markers"`
|
|
}
|
|
|
|
// suspectGroupView is a folder's worth of flagged tracks. As on the
|
|
// missing-files report, the folder is the unit of decision: Humanz was 62 rows
|
|
// in one folder, which is one problem to look at, not 62.
|
|
type suspectGroupView struct {
|
|
Directory string `json:"directory"`
|
|
Tracks []suspectTrackView `json:"tracks"`
|
|
}
|
|
|
|
// adminSuspectSourcesResponse is the paged envelope. Total counts tracks.
|
|
type adminSuspectSourcesResponse struct {
|
|
Total int64 `json:"total"`
|
|
Limit int `json:"limit"`
|
|
Offset int `json:"offset"`
|
|
Groups []suspectGroupView `json:"groups"`
|
|
}
|
|
|
|
// handleListSuspectSources implements GET /api/admin/library/suspect-sources.
|
|
//
|
|
// Read-only. A marker is a reason to look, not proof: a band can title a song
|
|
// "Music Video". What to do about a flagged file — merge it in Duplicates,
|
|
// quarantine it, replace it in Lidarr — stays the operator's call.
|
|
func (h *handlers) handleListSuspectSources(w http.ResponseWriter, r *http.Request) {
|
|
limit, offset, err := parsePaging(r.URL.Query())
|
|
if err != nil {
|
|
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_paging")
|
|
return
|
|
}
|
|
|
|
q := dbq.New(h.pool)
|
|
total, err := q.CountSuspectSourceTracks(r.Context(), suspectSourcePattern)
|
|
if err != nil {
|
|
h.logger.Error("admin: count suspect-source tracks", "err", err)
|
|
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
|
|
return
|
|
}
|
|
rows, err := q.ListSuspectSourceTracks(r.Context(), dbq.ListSuspectSourceTracksParams{
|
|
Pattern: suspectSourcePattern,
|
|
PageLimit: int32(limit),
|
|
PageOffset: int32(offset),
|
|
})
|
|
if err != nil {
|
|
h.logger.Error("admin: list suspect-source tracks", "err", err)
|
|
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
|
|
return
|
|
}
|
|
|
|
writeJSON(w, http.StatusOK, adminSuspectSourcesResponse{
|
|
Total: total,
|
|
Limit: limit,
|
|
Offset: offset,
|
|
Groups: groupSuspectByDirectory(rows),
|
|
})
|
|
}
|
|
|
|
// groupSuspectByDirectory folds the directory-ordered rows into per-folder
|
|
// groups, the same run-length fold as groupMissingByDirectory. A page boundary
|
|
// can split a folder; the web page joins the halves.
|
|
func groupSuspectByDirectory(rows []dbq.ListSuspectSourceTracksRow) []suspectGroupView {
|
|
groups := make([]suspectGroupView, 0, 8)
|
|
for _, row := range rows {
|
|
t := suspectTrackView{
|
|
TrackID: uuidToString(row.ID),
|
|
Title: row.Title,
|
|
ArtistID: uuidToString(row.ArtistID),
|
|
ArtistName: row.ArtistName,
|
|
AlbumID: uuidToString(row.AlbumID),
|
|
AlbumTitle: row.AlbumTitle,
|
|
FilePath: row.FilePath,
|
|
DurationSec: row.DurationMs / 1000,
|
|
DiscNumber: row.DiscNumber,
|
|
TrackNumber: row.TrackNumber,
|
|
Markers: sourceMarkersFor(row.FilePath),
|
|
}
|
|
if n := len(groups); n > 0 && groups[n-1].Directory == row.Directory {
|
|
groups[n-1].Tracks = append(groups[n-1].Tracks, t)
|
|
continue
|
|
}
|
|
groups = append(groups, suspectGroupView{
|
|
Directory: row.Directory,
|
|
Tracks: []suspectTrackView{t},
|
|
})
|
|
}
|
|
return groups
|
|
}
|