feat: admin Suspect sources report, for files named like video rips (#5410)
release / govulncheck (push) Successful in 37s
release / go (push) Failing after 1m10s
release / web (push) Failing after 1m24s
release / integration (push) Successful in 4m38s
release / android (push) Successful in 5m5s
release / Attach APK to the Release (tag releases only) (push) Canceled after 0s
release / Build + push container image (push) Canceled after 0s
release / Verify release artifacts (tag releases only) (push) Canceled after 0s
release / Build signed APK (releases and dev) (push) Canceled after 5m18s
release / govulncheck (push) Successful in 37s
release / go (push) Failing after 1m10s
release / web (push) Failing after 1m24s
release / integration (push) Successful in 4m38s
release / android (push) Successful in 5m5s
release / Attach APK to the Release (tag releases only) (push) Canceled after 0s
release / Build + push container image (push) Canceled after 0s
release / Verify release artifacts (tag releases only) (push) Canceled after 0s
release / Build signed APK (releases and dev) (push) Canceled after 5m18s
Humanz turned out to be YouTube rips: "(Official Video)", "Visualizer", a reaction video filed as a song, junk disc numbers (#5401). The library holds about 200 more files named the same way. This report lists them, grouped by folder like Missing files, with the markers each file name carries. - GET /api/admin/library/suspect-sources: one marker list in Go builds both the Postgres ~* filter and each row's labels, so they cannot drift. Basename only; missing files are left out. - Markers calibrated on the live library: "live in/at" dropped (real live albums), "reaction" narrowed (it caught "Chain Reaction"). - Admin tab "Suspect sources", read-only, loads as you scroll; a folder split across pages is joined back into one. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,173 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"path"
|
||||
"regexp"
|
||||
"strings"
|
||||
|
||||
"git.fabledsword.com/bvandeusen/minstrel/internal/db/dbq"
|
||||
)
|
||||
|
||||
// sourceMarker is one sign that a file was ripped from a video rather than
|
||||
// taken from a release: words a video title carries and an album track does
|
||||
// not (#5410).
|
||||
//
|
||||
// Each pattern is used twice, so it is written in the subset that Postgres's
|
||||
// regex flavour and Go's RE2 read the same way: no \b (Postgres spells word
|
||||
// boundaries \y), no lookaround, and brackets written as (\(|\[) rather than
|
||||
// as a bracket expression. The SQL side matches it case-insensitively with ~*,
|
||||
// the Go side with (?i); [^a-zA-Z] is spelled out so neither has to fold case
|
||||
// inside a negated class.
|
||||
type sourceMarker struct {
|
||||
label string
|
||||
pattern string
|
||||
}
|
||||
|
||||
// suspectSourceMarkers was calibrated against the operator's library on
|
||||
// 2026-10-08. Two candidates were left out on purpose:
|
||||
// - "live in"/"live at": 229 files, nearly all from real live albums.
|
||||
// - a bare "reaction": it caught Beck's "Chain Reaction". The marker below
|
||||
// wants the phrases a reaction video actually uses.
|
||||
var suspectSourceMarkers = []sourceMarker{
|
||||
{"music video", `official[^a-zA-Z]*(music[^a-zA-Z]*)?video|music[^a-zA-Z]*video`},
|
||||
{"official audio", `official[^a-zA-Z]*audio`},
|
||||
{"lyric video", `lyrics?[^a-zA-Z]*video`},
|
||||
{"visualiser", `visuali[sz]er`},
|
||||
{"reaction", `reaction[^a-zA-Z]*(video|mashup)|reacts?[^a-zA-Z]+to[^a-zA-Z]|first[^a-zA-Z]*time[^a-zA-Z]*(hearing|listening)`},
|
||||
{"MV", `(^|[^a-zA-Z])(mv|m/v)([^a-zA-Z]|$)`},
|
||||
{"[Audio]", `(\(|\[)audio(\)|\])`},
|
||||
{"[HD]", `(\(|\[)(hd|hq|4k)(\)|\])`},
|
||||
}
|
||||
|
||||
// suspectSourcePattern is every marker as one alternation, for the SQL filter.
|
||||
var suspectSourcePattern = func() string {
|
||||
parts := make([]string, len(suspectSourceMarkers))
|
||||
for i, m := range suspectSourceMarkers {
|
||||
parts[i] = "(" + m.pattern + ")"
|
||||
}
|
||||
return strings.Join(parts, "|")
|
||||
}()
|
||||
|
||||
var suspectSourceRegexps = func() []*regexp.Regexp {
|
||||
out := make([]*regexp.Regexp, len(suspectSourceMarkers))
|
||||
for i, m := range suspectSourceMarkers {
|
||||
out[i] = regexp.MustCompile("(?i)" + m.pattern)
|
||||
}
|
||||
return out
|
||||
}()
|
||||
|
||||
// sourceMarkersFor returns the labels of every marker the file's basename
|
||||
// carries, in list order. Empty for a file the SQL filter would not return.
|
||||
func sourceMarkersFor(filePath string) []string {
|
||||
base := path.Base(filePath)
|
||||
labels := make([]string, 0, 2)
|
||||
for i, re := range suspectSourceRegexps {
|
||||
if re.MatchString(base) {
|
||||
labels = append(labels, suspectSourceMarkers[i].label)
|
||||
}
|
||||
}
|
||||
return labels
|
||||
}
|
||||
|
||||
// suspectTrackView is one flagged track. FilePath is the evidence: the
|
||||
// markers are read from its basename, and the operator needs to see it to
|
||||
// judge whether the flag is right.
|
||||
type suspectTrackView struct {
|
||||
TrackID string `json:"track_id"`
|
||||
Title string `json:"title"`
|
||||
ArtistID string `json:"artist_id"`
|
||||
ArtistName string `json:"artist_name"`
|
||||
AlbumID string `json:"album_id"`
|
||||
AlbumTitle string `json:"album_title"`
|
||||
FilePath string `json:"file_path"`
|
||||
DurationSec int32 `json:"duration_sec"`
|
||||
DiscNumber *int32 `json:"disc_number"`
|
||||
TrackNumber *int32 `json:"track_number"`
|
||||
Markers []string `json:"markers"`
|
||||
}
|
||||
|
||||
// suspectGroupView is a folder's worth of flagged tracks. As on the
|
||||
// missing-files report, the folder is the unit of decision: Humanz was 62 rows
|
||||
// in one folder, which is one problem to look at, not 62.
|
||||
type suspectGroupView struct {
|
||||
Directory string `json:"directory"`
|
||||
Tracks []suspectTrackView `json:"tracks"`
|
||||
}
|
||||
|
||||
// adminSuspectSourcesResponse is the paged envelope. Total counts tracks.
|
||||
type adminSuspectSourcesResponse struct {
|
||||
Total int64 `json:"total"`
|
||||
Limit int `json:"limit"`
|
||||
Offset int `json:"offset"`
|
||||
Groups []suspectGroupView `json:"groups"`
|
||||
}
|
||||
|
||||
// handleListSuspectSources implements GET /api/admin/library/suspect-sources.
|
||||
//
|
||||
// Read-only. A marker is a reason to look, not proof: a band can title a song
|
||||
// "Music Video". What to do about a flagged file — merge it in Duplicates,
|
||||
// quarantine it, replace it in Lidarr — stays the operator's call.
|
||||
func (h *handlers) handleListSuspectSources(w http.ResponseWriter, r *http.Request) {
|
||||
limit, offset, err := parsePaging(r.URL.Query())
|
||||
if err != nil {
|
||||
writeAdminJSONErr(w, http.StatusBadRequest, "invalid_paging")
|
||||
return
|
||||
}
|
||||
|
||||
q := dbq.New(h.pool)
|
||||
total, err := q.CountSuspectSourceTracks(r.Context(), suspectSourcePattern)
|
||||
if err != nil {
|
||||
h.logger.Error("admin: count suspect-source tracks", "err", err)
|
||||
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
|
||||
return
|
||||
}
|
||||
rows, err := q.ListSuspectSourceTracks(r.Context(), dbq.ListSuspectSourceTracksParams{
|
||||
Pattern: suspectSourcePattern,
|
||||
PageLimit: int32(limit),
|
||||
PageOffset: int32(offset),
|
||||
})
|
||||
if err != nil {
|
||||
h.logger.Error("admin: list suspect-source tracks", "err", err)
|
||||
writeAdminJSONErr(w, http.StatusInternalServerError, "server_error")
|
||||
return
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, adminSuspectSourcesResponse{
|
||||
Total: total,
|
||||
Limit: limit,
|
||||
Offset: offset,
|
||||
Groups: groupSuspectByDirectory(rows),
|
||||
})
|
||||
}
|
||||
|
||||
// groupSuspectByDirectory folds the directory-ordered rows into per-folder
|
||||
// groups, the same run-length fold as groupMissingByDirectory. A page boundary
|
||||
// can split a folder; the web page joins the halves.
|
||||
func groupSuspectByDirectory(rows []dbq.ListSuspectSourceTracksRow) []suspectGroupView {
|
||||
groups := make([]suspectGroupView, 0, 8)
|
||||
for _, row := range rows {
|
||||
t := suspectTrackView{
|
||||
TrackID: uuidToString(row.ID),
|
||||
Title: row.Title,
|
||||
ArtistID: uuidToString(row.ArtistID),
|
||||
ArtistName: row.ArtistName,
|
||||
AlbumID: uuidToString(row.AlbumID),
|
||||
AlbumTitle: row.AlbumTitle,
|
||||
FilePath: row.FilePath,
|
||||
DurationSec: row.DurationMs / 1000,
|
||||
DiscNumber: row.DiscNumber,
|
||||
TrackNumber: row.TrackNumber,
|
||||
Markers: sourceMarkersFor(row.FilePath),
|
||||
}
|
||||
if n := len(groups); n > 0 && groups[n-1].Directory == row.Directory {
|
||||
groups[n-1].Tracks = append(groups[n-1].Tracks, t)
|
||||
continue
|
||||
}
|
||||
groups = append(groups, suspectGroupView{
|
||||
Directory: row.Directory,
|
||||
Tracks: []suspectTrackView{t},
|
||||
})
|
||||
}
|
||||
return groups
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"slices"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// Real basenames from the operator's library (2026-10-08), each with the
|
||||
// labels it should carry. The negatives are the near misses that shaped the
|
||||
// list: a song called "Chain Reaction", a live album, an ordinary track.
|
||||
func TestSourceMarkersFor(t *testing.T) {
|
||||
cases := []struct {
|
||||
base string
|
||||
want []string
|
||||
}{
|
||||
{"Daft Punk - Random Access Memories - 06 - Daft Punk - Doin' It Right (Music Video) ft. Panda Bear.mp3", []string{"music video"}},
|
||||
{"Gorillaz - Humanz - 03 - Gorillaz - Saturnz Barz (Official Video).mp3", []string{"music video"}},
|
||||
{"Gorillaz - Humanz - 05 - Gorillaz - Andromeda (Official Audio).mp3", []string{"official audio"}},
|
||||
{"Gorillaz - Gorillaz - 06 - Gorillaz - P45 (Visualizer).mp3", []string{"visualiser"}},
|
||||
{"Gorillaz - Demon Days - 16 - Gorillaz - Don Quixote's Christmas Bonanza (Visualiser).mp3", []string{"visualiser"}},
|
||||
{"Artist - Album - 01 - Artist - Song (Lyric Video).mp3", []string{"lyric video"}},
|
||||
{"Gorillaz - Humanz - 02 - FIRST TIME HEARING Gorillaz - Ascension REACTION.mp3", []string{"reaction"}},
|
||||
{"Watsky - INTENTION - 05 - MANIAC Reacts to Watsky - AWW SHiT.mp3", []string{"reaction"}},
|
||||
{"米津玄師 - diorama - 09 - 【MV】米津玄師 - 恋と病熱.mp3", []string{"MV"}},
|
||||
{"Andora - Ego - 01 - Andora - Ego (feat. Will Stetson) MV.mp3", []string{"MV"}},
|
||||
{"Jimmy Eat World - Something(s) Loud - 05 - Jimmy Eat World - Call to Love (Audio).mp3", []string{"[Audio]"}},
|
||||
{"Record Heat - World War IV - 03 - Record Heat - Front Seat Feelin' [Audio].mp3", []string{"[Audio]"}},
|
||||
{"Aphex Twin - Come To Daddy - 03 - Aphex Twin - Bucephalus Bouncing Ball (HQ).mp3", []string{"[HD]"}},
|
||||
{"Artist - Album - 01 - Artist - Song (Official Lyric Video) [4K].mp3", []string{"lyric video", "[HD]"}},
|
||||
|
||||
{"Beck - Guero - 15 - Beck - Chain Reaction.mp3", nil},
|
||||
{"Nirvana - MTV Unplugged in New York - 01 - About a Girl (live in New York).flac", nil},
|
||||
{"Boards of Canada - Music Has the Right to Children - 05 - Roygbiv.flac", nil},
|
||||
{"Artist - Album - 01 - Mvula.flac", nil},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got := sourceMarkersFor("/music/x/" + c.base)
|
||||
if !slices.Equal(got, c.want) && !(len(got) == 0 && len(c.want) == 0) {
|
||||
t.Errorf("%s:\n got %v\n want %v", c.base, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The folder is not evidence: a marker word in a directory name must not
|
||||
// flag the files inside it.
|
||||
func TestSourceMarkersFor_ReadsOnlyTheBasename(t *testing.T) {
|
||||
if got := sourceMarkersFor("/music/Official Video Collection/01 - Song.flac"); len(got) != 0 {
|
||||
t.Errorf("directory name flagged the file: %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// The integration test runs the same alternation through Postgres's ~*, so the
|
||||
// one pattern is exercised in both regex dialects it has to work in.
|
||||
func TestHandleListSuspectSources_Integration(t *testing.T) {
|
||||
h, pool := testHandlers(t)
|
||||
truncateLibrary(t, pool)
|
||||
admin := seedUser(t, pool, "suspect-admin", "pw", true)
|
||||
|
||||
artist := seedArtist(t, pool, "Gorillaz")
|
||||
humanz := seedAlbum(t, pool, artist.ID, "Humanz", 2017)
|
||||
seedTrack(t, pool, humanz.ID, artist.ID, "Saturnz Barz (Official Video)", 1, 180_000)
|
||||
seedTrack(t, pool, humanz.ID, artist.ID, "Momentz (Visualizer)", 2, 200_000)
|
||||
seedTrack(t, pool, humanz.ID, artist.ID, "Andromeda [Audio]", 3, 190_000)
|
||||
seedTrack(t, pool, humanz.ID, artist.ID, "Chain Reaction", 4, 190_000)
|
||||
seedTrack(t, pool, humanz.ID, artist.ID, "Busted and Blue", 5, 190_000)
|
||||
gone := seedTrack(t, pool, humanz.ID, artist.ID, "Ascension (Lyric Video)", 6, 156_000)
|
||||
if _, err := pool.Exec(context.Background(),
|
||||
`UPDATE tracks SET missing_since = now() WHERE id = $1`, gone.ID); err != nil {
|
||||
t.Fatalf("mark missing: %v", err)
|
||||
}
|
||||
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/admin/library/suspect-sources?limit=50", nil)
|
||||
req = withUser(req, admin)
|
||||
w := httptest.NewRecorder()
|
||||
h.handleListSuspectSources(w, req)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("status %d: %s", w.Code, w.Body.String())
|
||||
}
|
||||
|
||||
var resp adminSuspectSourcesResponse
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
|
||||
t.Fatalf("decode: %v", err)
|
||||
}
|
||||
// Three present files carry a marker. "Chain Reaction" and "Busted and
|
||||
// Blue" carry none, and the lyric video's file is missing.
|
||||
if resp.Total != 3 {
|
||||
t.Errorf("total = %d, want 3", resp.Total)
|
||||
}
|
||||
if len(resp.Groups) != 1 {
|
||||
t.Fatalf("groups = %d, want 1 (one folder)", len(resp.Groups))
|
||||
}
|
||||
got := map[string][]string{}
|
||||
for _, tr := range resp.Groups[0].Tracks {
|
||||
got[tr.Title] = tr.Markers
|
||||
}
|
||||
want := map[string][]string{
|
||||
"Saturnz Barz (Official Video)": {"music video"},
|
||||
"Momentz (Visualizer)": {"visualiser"},
|
||||
"Andromeda [Audio]": {"[Audio]"},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("tracks = %v, want %v", got, want)
|
||||
}
|
||||
for title, labels := range want {
|
||||
if !slices.Equal(got[title], labels) {
|
||||
t.Errorf("%s: markers %v, want %v", title, got[title], labels)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -252,6 +252,7 @@ func Mount(r chi.Router, pool *pgxpool.Pool, logger *slog.Logger, events *playev
|
||||
// and because the destructive /tracks/{id} route above must
|
||||
// not be mistaken for it (#2527).
|
||||
admin.Get("/library/missing", h.handleListMissingTracks)
|
||||
admin.Get("/library/suspect-sources", h.handleListSuspectSources)
|
||||
|
||||
admin.Get("/library/coverage", h.handleGetLibraryCoverage)
|
||||
admin.Get("/library/fingerprints", h.handleGetFingerprintCoverage)
|
||||
|
||||
@@ -70,6 +70,20 @@ func (q *Queries) CountMissingTracks(ctx context.Context) (int64, error) {
|
||||
return count, err
|
||||
}
|
||||
|
||||
const countSuspectSourceTracks = `-- name: CountSuspectSourceTracks :one
|
||||
SELECT COUNT(*) FROM tracks t
|
||||
WHERE t.missing_since IS NULL
|
||||
AND regexp_replace(t.file_path, '^.*/', '') ~* $1::text
|
||||
`
|
||||
|
||||
// Total for the report's paging; the same filter as the list above.
|
||||
func (q *Queries) CountSuspectSourceTracks(ctx context.Context, pattern string) (int64, error) {
|
||||
row := q.db.QueryRow(ctx, countSuspectSourceTracks, pattern)
|
||||
var count int64
|
||||
err := row.Scan(&count)
|
||||
return count, err
|
||||
}
|
||||
|
||||
const countTracksByAlbum = `-- name: CountTracksByAlbum :one
|
||||
SELECT count(*) FROM tracks WHERE album_id = $1
|
||||
`
|
||||
@@ -576,6 +590,89 @@ func (q *Queries) ListRandomTracksForUser(ctx context.Context, arg ListRandomTra
|
||||
return items, nil
|
||||
}
|
||||
|
||||
const listSuspectSourceTracks = `-- name: ListSuspectSourceTracks :many
|
||||
SELECT t.id,
|
||||
t.title,
|
||||
t.file_path,
|
||||
regexp_replace(t.file_path, '/[^/]*$', '') AS directory,
|
||||
t.duration_ms,
|
||||
t.disc_number,
|
||||
t.track_number,
|
||||
albums.id AS album_id,
|
||||
albums.title AS album_title,
|
||||
artists.id AS artist_id,
|
||||
artists.name AS artist_name
|
||||
FROM tracks t
|
||||
JOIN albums ON albums.id = t.album_id
|
||||
JOIN artists ON artists.id = t.artist_id
|
||||
WHERE t.missing_since IS NULL
|
||||
AND regexp_replace(t.file_path, '^.*/', '') ~* $1::text
|
||||
ORDER BY directory, t.disc_number NULLS FIRST, t.track_number NULLS FIRST, t.title
|
||||
LIMIT $3 OFFSET $2
|
||||
`
|
||||
|
||||
type ListSuspectSourceTracksParams struct {
|
||||
Pattern string
|
||||
PageOffset int32
|
||||
PageLimit int32
|
||||
}
|
||||
|
||||
type ListSuspectSourceTracksRow struct {
|
||||
ID pgtype.UUID
|
||||
Title string
|
||||
FilePath string
|
||||
Directory string
|
||||
DurationMs int32
|
||||
DiscNumber *int32
|
||||
TrackNumber *int32
|
||||
AlbumID pgtype.UUID
|
||||
AlbumTitle string
|
||||
ArtistID pgtype.UUID
|
||||
ArtistName string
|
||||
}
|
||||
|
||||
// The admin report of present tracks whose filename looks like a video rip
|
||||
// (#5410): "(Official Video)", "[Audio]", "Visualizer" and the like.
|
||||
//
|
||||
// pattern is built in Go (internal/api/admin_suspect_sources.go) from the
|
||||
// same marker list that labels each row, so the filter and the labels cannot
|
||||
// drift. It is matched against the basename only: a folder called "Reaction
|
||||
// Sessions" says nothing about how a file was sourced.
|
||||
//
|
||||
// Ordered by directory, like ListMissingTracks, so the handler's fold gets
|
||||
// each folder as one contiguous run.
|
||||
func (q *Queries) ListSuspectSourceTracks(ctx context.Context, arg ListSuspectSourceTracksParams) ([]ListSuspectSourceTracksRow, error) {
|
||||
rows, err := q.db.Query(ctx, listSuspectSourceTracks, arg.Pattern, arg.PageOffset, arg.PageLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var items []ListSuspectSourceTracksRow
|
||||
for rows.Next() {
|
||||
var i ListSuspectSourceTracksRow
|
||||
if err := rows.Scan(
|
||||
&i.ID,
|
||||
&i.Title,
|
||||
&i.FilePath,
|
||||
&i.Directory,
|
||||
&i.DurationMs,
|
||||
&i.DiscNumber,
|
||||
&i.TrackNumber,
|
||||
&i.AlbumID,
|
||||
&i.AlbumTitle,
|
||||
&i.ArtistID,
|
||||
&i.ArtistName,
|
||||
); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
items = append(items, i)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return items, nil
|
||||
}
|
||||
|
||||
const listTrackPathsForReconcile = `-- name: ListTrackPathsForReconcile :many
|
||||
SELECT id, file_path, missing_since FROM tracks
|
||||
`
|
||||
|
||||
@@ -263,3 +263,39 @@ SELECT t.id,
|
||||
-- Total for the admin surface's badge and paging. Uses the same partial index
|
||||
-- (tracks_missing_since_idx) as the list above.
|
||||
SELECT COUNT(*) FROM tracks WHERE missing_since IS NOT NULL;
|
||||
|
||||
-- name: ListSuspectSourceTracks :many
|
||||
-- The admin report of present tracks whose filename looks like a video rip
|
||||
-- (#5410): "(Official Video)", "[Audio]", "Visualizer" and the like.
|
||||
--
|
||||
-- pattern is built in Go (internal/api/admin_suspect_sources.go) from the
|
||||
-- same marker list that labels each row, so the filter and the labels cannot
|
||||
-- drift. It is matched against the basename only: a folder called "Reaction
|
||||
-- Sessions" says nothing about how a file was sourced.
|
||||
--
|
||||
-- Ordered by directory, like ListMissingTracks, so the handler's fold gets
|
||||
-- each folder as one contiguous run.
|
||||
SELECT t.id,
|
||||
t.title,
|
||||
t.file_path,
|
||||
regexp_replace(t.file_path, '/[^/]*$', '') AS directory,
|
||||
t.duration_ms,
|
||||
t.disc_number,
|
||||
t.track_number,
|
||||
albums.id AS album_id,
|
||||
albums.title AS album_title,
|
||||
artists.id AS artist_id,
|
||||
artists.name AS artist_name
|
||||
FROM tracks t
|
||||
JOIN albums ON albums.id = t.album_id
|
||||
JOIN artists ON artists.id = t.artist_id
|
||||
WHERE t.missing_since IS NULL
|
||||
AND regexp_replace(t.file_path, '^.*/', '') ~* sqlc.arg(pattern)::text
|
||||
ORDER BY directory, t.disc_number NULLS FIRST, t.track_number NULLS FIRST, t.title
|
||||
LIMIT sqlc.arg(page_limit) OFFSET sqlc.arg(page_offset);
|
||||
|
||||
-- name: CountSuspectSourceTracks :one
|
||||
-- Total for the report's paging; the same filter as the list above.
|
||||
SELECT COUNT(*) FROM tracks t
|
||||
WHERE t.missing_since IS NULL
|
||||
AND regexp_replace(t.file_path, '^.*/', '') ~* sqlc.arg(pattern)::text;
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
import { createQuery } from '@tanstack/svelte-query';
|
||||
import { createInfiniteQuery, createQuery } from '@tanstack/svelte-query';
|
||||
import { api } from './client';
|
||||
import { qk } from './queries';
|
||||
import type {
|
||||
ActionResult,
|
||||
AdminMissingResponse,
|
||||
AdminSuspectResponse,
|
||||
AdminDuplicatesResponse,
|
||||
MergeDuplicateResult,
|
||||
AdminPlaybackError,
|
||||
@@ -896,6 +897,34 @@ export function createMissingFilesQuery(offset: number = 0, limit: number = 50)
|
||||
});
|
||||
}
|
||||
|
||||
// Suspect sources (#5410) --------------------------------------------------
|
||||
|
||||
export const SUSPECT_SOURCES_PAGE_SIZE = 50;
|
||||
|
||||
export async function listSuspectSources(offset: number = 0): Promise<AdminSuspectResponse> {
|
||||
return api.get<AdminSuspectResponse>(
|
||||
`/api/admin/library/suspect-sources?limit=${SUSPECT_SOURCES_PAGE_SIZE}&offset=${offset}`
|
||||
);
|
||||
}
|
||||
|
||||
// The next offset counts tracks, not groups: limit and total are in tracks.
|
||||
export function suspectSourcesNextOffset(last: AdminSuspectResponse): number | undefined {
|
||||
const loaded = last.offset + last.groups.reduce((n, g) => n + g.tracks.length, 0);
|
||||
return loaded >= last.total ? undefined : loaded;
|
||||
}
|
||||
|
||||
// Loads as the page scrolls (rule 172). Changes only when files are added or
|
||||
// removed, so it is not re-fetched on every focus.
|
||||
export function createSuspectSourcesQuery() {
|
||||
return createInfiniteQuery({
|
||||
queryKey: qk.adminSuspectSources(),
|
||||
queryFn: ({ pageParam }) => listSuspectSources(pageParam as number),
|
||||
initialPageParam: 0,
|
||||
getNextPageParam: suspectSourcesNextOffset,
|
||||
staleTime: 120_000
|
||||
});
|
||||
}
|
||||
|
||||
// Missing-file re-acquisition (#2527 / milestone #290) ----------------------
|
||||
|
||||
export type ReacquisitionSettings = {
|
||||
|
||||
@@ -58,6 +58,7 @@ export const qk = {
|
||||
['adminMissingFiles', { offset: offset ?? 0 }] as const,
|
||||
adminDuplicates: (offset?: number) =>
|
||||
['adminDuplicates', { offset: offset ?? 0 }] as const,
|
||||
adminSuspectSources: () => ['adminSuspectSources'] as const,
|
||||
adminDiagnosticDevices: (userId?: string) =>
|
||||
['adminDiagnosticDevices', { userId: userId ?? 'all' }] as const,
|
||||
smtpConfig: () => ['smtpConfig'] as const,
|
||||
|
||||
@@ -419,6 +419,37 @@ export type AdminMissingResponse = {
|
||||
groups: AdminMissingGroup[];
|
||||
};
|
||||
|
||||
// Suspect sources (#5410) --------------------------------------------------
|
||||
|
||||
// A present track whose filename carries a video-rip marker. markers are the
|
||||
// reasons, read from the file's basename: "music video", "[Audio]", "MV"...
|
||||
export type AdminSuspectTrack = {
|
||||
track_id: string;
|
||||
title: string;
|
||||
artist_id: string;
|
||||
artist_name: string;
|
||||
album_id: string;
|
||||
album_title: string;
|
||||
file_path: string;
|
||||
duration_sec: number;
|
||||
disc_number: number | null;
|
||||
track_number: number | null;
|
||||
markers: string[];
|
||||
};
|
||||
|
||||
// A folder's worth of flagged tracks; the server orders and groups by folder.
|
||||
export type AdminSuspectGroup = {
|
||||
directory: string;
|
||||
tracks: AdminSuspectTrack[];
|
||||
};
|
||||
|
||||
export type AdminSuspectResponse = {
|
||||
total: number;
|
||||
limit: number;
|
||||
offset: number;
|
||||
groups: AdminSuspectGroup[];
|
||||
};
|
||||
|
||||
// Duplicates report (#3912) -------------------------------------------------
|
||||
|
||||
// One copy in a proposed duplicate group. like_count and play_count cover every
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
{ href: '/admin/quarantine', label: 'Quarantine' },
|
||||
{ href: '/admin/missing-files', label: 'Missing files' },
|
||||
{ href: '/admin/duplicates', label: 'Duplicates' },
|
||||
{ href: '/admin/suspect-sources', label: 'Suspect sources' },
|
||||
{ href: '/admin/playback-errors', label: 'Playback errors' },
|
||||
{ href: '/admin/diagnostics', label: 'Diagnostics' },
|
||||
{ href: '/admin/tuning', label: 'Tuning' },
|
||||
|
||||
@@ -52,7 +52,7 @@ describe('AdminTabs', () => {
|
||||
);
|
||||
});
|
||||
|
||||
test('renders all ten tabs in order', () => {
|
||||
test('renders all eleven tabs in order', () => {
|
||||
state.pageUrl = new URL('http://localhost/admin');
|
||||
render(AdminTabs);
|
||||
const links = screen.getAllByRole('link');
|
||||
@@ -67,6 +67,9 @@ describe('AdminTabs', () => {
|
||||
// Duplicates follows Missing files: both are library-health reports on
|
||||
// what the library holds, rather than a queue of user reports.
|
||||
'Duplicates',
|
||||
// Suspect sources is the third library-health report: files whose
|
||||
// names say they were ripped from a video.
|
||||
'Suspect sources',
|
||||
'Playback errors',
|
||||
'Diagnostics',
|
||||
'Tuning',
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
<script lang="ts">
|
||||
import { pageTitle } from '#lib/branding.js';
|
||||
import { FileCheck, Music2 } from 'lucide-svelte';
|
||||
import { createSuspectSourcesQuery } from '#lib/api/admin.js';
|
||||
import { coverUrl } from '#lib/media/covers.js';
|
||||
import { formatDuration } from '#lib/media/duration.js';
|
||||
import InfiniteScrollSentinel from '#lib/components/InfiniteScrollSentinel.svelte';
|
||||
import type { AdminSuspectGroup, AdminSuspectTrack } from '#lib/api/types.js';
|
||||
|
||||
// Files whose names say they were ripped from a video (#5410): "(Official
|
||||
// Video)", "[Audio]", "Visualizer", a reaction. Humanz was 62 of them in one
|
||||
// folder, with junk disc numbers and a reaction video filed as a song.
|
||||
// Read-only: a marker is a reason to look, not proof, and the fix lives
|
||||
// elsewhere — Duplicates, Quarantine, or a better release in Lidarr.
|
||||
|
||||
const queryStore = createSuspectSourcesQuery();
|
||||
const query = $derived($queryStore);
|
||||
const total = $derived(query.data?.pages?.[0]?.total ?? 0);
|
||||
|
||||
// The server groups each page by folder, so a folder that straddles a page
|
||||
// boundary arrives as two groups. Join them back into one.
|
||||
const groups = $derived.by(() => {
|
||||
const out: AdminSuspectGroup[] = [];
|
||||
for (const page of query.data?.pages ?? []) {
|
||||
for (const g of page.groups) {
|
||||
const last = out.at(-1);
|
||||
if (last && last.directory === g.directory) {
|
||||
out[out.length - 1] = { ...last, tracks: [...last.tracks, ...g.tracks] };
|
||||
} else {
|
||||
out.push(g);
|
||||
}
|
||||
}
|
||||
}
|
||||
return out;
|
||||
});
|
||||
|
||||
function trackCountLabel(n: number): string {
|
||||
return n === 1 ? '1 track' : `${n} tracks`;
|
||||
}
|
||||
|
||||
function basename(path: string): string {
|
||||
return path.slice(path.lastIndexOf('/') + 1);
|
||||
}
|
||||
|
||||
// Disc and track are shown because junk numbering is the other half of
|
||||
// the symptom: Humanz's rips were spread over discs 1 to 15.
|
||||
function position(t: AdminSuspectTrack): string {
|
||||
const parts: string[] = [];
|
||||
if (t.disc_number != null) parts.push(`disc ${t.disc_number}`);
|
||||
if (t.track_number != null) parts.push(`track ${t.track_number}`);
|
||||
return parts.join(', ');
|
||||
}
|
||||
</script>
|
||||
|
||||
<svelte:head><title>{pageTitle('Admin · Suspect sources')}</title></svelte:head>
|
||||
|
||||
<div class="space-y-6">
|
||||
<header class="space-y-1">
|
||||
<div class="flex items-center gap-2">
|
||||
<h2 class="font-display text-2xl font-medium text-text-primary">Suspect sources</h2>
|
||||
{#if total > 0}
|
||||
<span
|
||||
class="inline-flex items-center rounded-full bg-accent-tint px-2 py-0.5 text-xs text-accent-fg"
|
||||
data-testid="suspect-count-pill"
|
||||
>
|
||||
{total}
|
||||
</span>
|
||||
{/if}
|
||||
</div>
|
||||
<p class="text-text-secondary">
|
||||
Tracks whose file name reads like a video title rather than an album track,
|
||||
such as "(Official Video)", "[Audio]" or "Visualizer". These usually came from
|
||||
a video site, and can be the wrong audio: a live take, a music-video edit, or
|
||||
something else entirely. Nothing here is changed automatically.
|
||||
</p>
|
||||
</header>
|
||||
|
||||
{#if query.isPending}
|
||||
<p class="text-text-secondary">Reading file names…</p>
|
||||
{:else if query.isError}
|
||||
<p class="text-error-fg">Couldn't load the suspect-sources list.</p>
|
||||
{:else if groups.length === 0}
|
||||
<div class="rounded-lg border border-border bg-surface p-6 text-center">
|
||||
<FileCheck size={28} strokeWidth={1} class="mx-auto text-text-muted" />
|
||||
<p class="mt-3 text-text-primary">No file names look like video rips.</p>
|
||||
<p class="mt-1 text-sm text-text-secondary">
|
||||
A track lands here when its file name carries a video-title marker, such as
|
||||
"Official Video", "Lyric Video", "[Audio]" or "MV".
|
||||
</p>
|
||||
</div>
|
||||
{:else}
|
||||
<ul class="space-y-4">
|
||||
{#each groups as group (group.directory)}
|
||||
<li class="overflow-hidden rounded-lg border border-border bg-surface">
|
||||
<div class="flex items-baseline justify-between gap-4 border-b border-border px-4 py-3">
|
||||
<h3 class="truncate font-mono text-sm text-text-primary" title={group.directory}>
|
||||
{group.directory}
|
||||
</h3>
|
||||
<span class="shrink-0 text-xs text-text-secondary">
|
||||
{trackCountLabel(group.tracks.length)}
|
||||
</span>
|
||||
</div>
|
||||
|
||||
<ul class="divide-y divide-border">
|
||||
{#each group.tracks as t (t.track_id)}
|
||||
<li class="flex items-center gap-3 px-4 py-2" data-testid="suspect-track-row">
|
||||
<div
|
||||
class="flex h-10 w-10 shrink-0 items-center justify-center rounded bg-surface-hover"
|
||||
aria-hidden="true"
|
||||
>
|
||||
{#if t.album_id}
|
||||
<img
|
||||
src={coverUrl(t.album_id)}
|
||||
alt=""
|
||||
class="h-full w-full rounded object-cover"
|
||||
loading="lazy"
|
||||
/>
|
||||
{:else}
|
||||
<Music2 size={18} strokeWidth={1} class="text-text-muted" />
|
||||
{/if}
|
||||
</div>
|
||||
|
||||
<div class="min-w-0 flex-1 space-y-0.5">
|
||||
<div class="truncate text-sm text-text-primary">{t.title}</div>
|
||||
<div class="truncate text-xs text-text-secondary">
|
||||
{t.artist_name} ·
|
||||
<a href="/albums/{t.album_id}" class="hover:text-text-primary hover:underline"
|
||||
>{t.album_title}</a
|
||||
>{#if position(t)}<span class="text-text-muted"> · {position(t)}</span>{/if}
|
||||
</div>
|
||||
<!-- The file name is the evidence; the markers are read from it. -->
|
||||
<div class="truncate font-mono text-xs text-text-muted" title={t.file_path}>
|
||||
{basename(t.file_path)}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="hidden shrink-0 flex-wrap justify-end gap-1 sm:flex">
|
||||
{#each t.markers as m (m)}
|
||||
<span
|
||||
class="rounded-full border border-border px-2 py-0.5 text-xs text-text-secondary"
|
||||
data-testid="suspect-marker">{m}</span
|
||||
>
|
||||
{/each}
|
||||
</div>
|
||||
<span class="shrink-0 text-xs text-text-muted">{formatDuration(t.duration_sec)}</span>
|
||||
</li>
|
||||
{/each}
|
||||
</ul>
|
||||
</li>
|
||||
{/each}
|
||||
</ul>
|
||||
|
||||
{#if query.hasNextPage}
|
||||
<InfiniteScrollSentinel
|
||||
enabled={!query.isFetchingNextPage}
|
||||
onIntersect={() => query.fetchNextPage()}
|
||||
/>
|
||||
{#if query.isFetchingNextPage}
|
||||
<p class="py-2 text-center text-sm text-text-secondary">Loading more…</p>
|
||||
{/if}
|
||||
{/if}
|
||||
{/if}
|
||||
</div>
|
||||
@@ -0,0 +1,125 @@
|
||||
import { afterEach, describe, expect, test, vi } from 'vitest';
|
||||
import { render, screen } from '@testing-library/svelte';
|
||||
import { readable } from 'svelte/store';
|
||||
import type { AdminSuspectResponse, AdminSuspectTrack } from '#lib/api/types.js';
|
||||
|
||||
vi.mock('#lib/api/admin.js', () => ({
|
||||
createSuspectSourcesQuery: vi.fn()
|
||||
}));
|
||||
|
||||
import SuspectSourcesPage from './+page.svelte';
|
||||
import { createSuspectSourcesQuery } from '#lib/api/admin.js';
|
||||
import { suspectSourcesNextOffset } from '#lib/api/admin.js';
|
||||
|
||||
const mocked = createSuspectSourcesQuery as ReturnType<typeof vi.fn>;
|
||||
|
||||
const HUMANZ = '/music/Gorillaz/Humanz (2017)';
|
||||
|
||||
function track(title: string, file: string, markers: string[], disc: number | null = 15): AdminSuspectTrack {
|
||||
return {
|
||||
track_id: `t-${title}`,
|
||||
title,
|
||||
artist_id: 'ar-1',
|
||||
artist_name: 'Gorillaz',
|
||||
album_id: 'al-1',
|
||||
album_title: 'Humanz',
|
||||
file_path: `${HUMANZ}/${file}`,
|
||||
duration_sec: 180,
|
||||
disc_number: disc,
|
||||
track_number: 4,
|
||||
markers
|
||||
};
|
||||
}
|
||||
|
||||
function page(offset: number, total: number, groups: AdminSuspectResponse['groups']): AdminSuspectResponse {
|
||||
return { total, limit: 50, offset, groups };
|
||||
}
|
||||
|
||||
// The page reads only these fields of the infinite query.
|
||||
function infinite(pages: AdminSuspectResponse[], opts: { hasNextPage?: boolean; isPending?: boolean } = {}) {
|
||||
return readable({
|
||||
data: { pages },
|
||||
isPending: opts.isPending ?? false,
|
||||
isError: false,
|
||||
hasNextPage: opts.hasNextPage ?? false,
|
||||
isFetchingNextPage: false,
|
||||
fetchNextPage: () => {}
|
||||
});
|
||||
}
|
||||
|
||||
afterEach(() => vi.clearAllMocks());
|
||||
|
||||
describe('Suspect sources page', () => {
|
||||
test('shows each flagged track with its file name and markers', () => {
|
||||
mocked.mockReturnValue(
|
||||
infinite([
|
||||
page(0, 2, [
|
||||
{
|
||||
directory: HUMANZ,
|
||||
tracks: [
|
||||
track('Strobelite', 'Gorillaz - Strobelite (Official Video).mp3', ['music video']),
|
||||
track('Andromeda', 'Gorillaz - Andromeda [Audio] [HQ].mp3', ['[Audio]', '[HD]'])
|
||||
]
|
||||
}
|
||||
])
|
||||
])
|
||||
);
|
||||
render(SuspectSourcesPage);
|
||||
|
||||
expect(screen.getByTestId('suspect-count-pill')).toHaveTextContent('2');
|
||||
expect(screen.getByText(HUMANZ)).toBeInTheDocument();
|
||||
expect(screen.getByText('Gorillaz - Strobelite (Official Video).mp3')).toBeInTheDocument();
|
||||
expect(screen.getAllByTestId('suspect-marker').map((m) => m.textContent)).toEqual([
|
||||
'music video',
|
||||
'[Audio]',
|
||||
'[HD]'
|
||||
]);
|
||||
expect(screen.getAllByText(/disc 15, track 4/)).toHaveLength(2);
|
||||
});
|
||||
|
||||
// A folder split across two pages is still one folder on screen.
|
||||
test('joins a folder that straddles a page boundary', () => {
|
||||
mocked.mockReturnValue(
|
||||
infinite([
|
||||
page(0, 3, [{ directory: HUMANZ, tracks: [track('Strobelite', 'a (Official Video).mp3', ['music video'])] }]),
|
||||
page(1, 3, [
|
||||
{ directory: HUMANZ, tracks: [track('Momentz', 'b (Visualizer).mp3', ['visualiser'])] },
|
||||
{ directory: '/music/Daft Punk/RAM', tracks: [track('Doin It Right', 'c (Music Video).mp3', ['music video'])] }
|
||||
])
|
||||
])
|
||||
);
|
||||
render(SuspectSourcesPage);
|
||||
|
||||
expect(screen.getAllByText(HUMANZ)).toHaveLength(1);
|
||||
expect(screen.getAllByText('2 tracks')).toHaveLength(1);
|
||||
expect(screen.getAllByTestId('suspect-track-row')).toHaveLength(3);
|
||||
});
|
||||
|
||||
test('empty library explains what would land here', () => {
|
||||
mocked.mockReturnValue(infinite([page(0, 0, [])]));
|
||||
render(SuspectSourcesPage);
|
||||
expect(screen.getByText(/no file names look like video rips/i)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
test('loads more as you scroll, with no button', () => {
|
||||
mocked.mockReturnValue(
|
||||
infinite([page(0, 80, [{ directory: HUMANZ, tracks: [track('Strobelite', 'a (Official Video).mp3', ['music video'])] }])], {
|
||||
hasNextPage: true
|
||||
})
|
||||
);
|
||||
const { container } = render(SuspectSourcesPage);
|
||||
expect(container.querySelector('div[aria-hidden="true"].h-px')).not.toBeNull();
|
||||
expect(screen.queryByRole('button', { name: /more/i })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('suspectSourcesNextOffset', () => {
|
||||
test('counts tracks, not groups, and stops at the total', () => {
|
||||
const groups = [
|
||||
{ directory: 'a', tracks: [track('1', '1.mp3', []), track('2', '2.mp3', [])] },
|
||||
{ directory: 'b', tracks: [track('3', '3.mp3', [])] }
|
||||
];
|
||||
expect(suspectSourcesNextOffset(page(0, 10, groups))).toBe(3);
|
||||
expect(suspectSourcesNextOffset(page(7, 10, groups))).toBeUndefined();
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user