package library import ( "sort" "strings" "time" ) // SurvivorCandidate is what choosing which copy to keep needs to know about one // member of a duplicate group. type SurvivorCandidate struct { TrackID string FileFormat string FileSize int64 AddedAt time.Time // LidarrTracked: Lidarr maps this file to a track of the release it // monitors. Removing it opens a hole Lidarr downloads again (M498), so a // tracked copy outranks everything else. False when Lidarr was not asked. LidarrTracked bool // TagFit is TagFitScore: how well the copy's tags and name fit the album. TagFit int } // TagFitScore counts the signs that a copy belongs where it is filed (M498 // #5436), one point each: // - it has a track number that no other track on its album also claims // - its file name carries no video-rip marker (SourceMarkersFor) // - it has a MusicBrainz recording id // // Two copies of one song differ in these far more often than in anything a // listener hears: the Humanz rips differ by a few bytes, and file size picked // the wrong one in 6 of 21 groups. func TagFitScore(hasTrackNumber, positionClash bool, filePath string, hasMBID bool) int { score := 0 if hasTrackNumber && !positionClash { score++ } if len(SourceMarkersFor(filePath)) == 0 { score++ } if hasMBID { score++ } return score } // losslessFormats are the scanned extensions that are lossless by definition. // m4a is left out on purpose: it holds either ALAC or AAC, and the scanner // records only the extension, so calling it lossless would sometimes prefer an // AAC copy over a FLAC one. var losslessFormats = map[string]bool{"flac": true, "wav": true} // ProposeSurvivor picks which copy of a duplicate group to keep, and gives the // reason in words the operator reads beside it. It is a default, not a verdict: // the report shows it and the merge (#3911) lets the operator choose another. // // In order: // 0. the copy Lidarr maps, then the copy whose tags fit the album best // (TagFitScore). Both are zero for every copy when the caller does not // know them, and the order falls through to the quality rules below // 1. lossless over lossy — the one difference no later step can recover // 2. the larger file — for one recording at one duration that is the higher // bitrate. The scanner does not record bitrate (tracks.bitrate is never // filled), so file size is the signal that actually exists // 3. the copy in the library longest — the one most likely to carry the play // history and likes, so the merge moves the least // 4. the lowest track id, so the choice is stable between page loads func ProposeSurvivor(cands []SurvivorCandidate) (trackID, reason string) { if len(cands) == 0 { return "", "" } ranked := append([]SurvivorCandidate(nil), cands...) sort.SliceStable(ranked, func(i, j int) bool { return survivorBefore(ranked[i], ranked[j]) }) best := ranked[0] if len(ranked) == 1 { return best.TrackID, "the only copy" } // The reason names the first rule that separated the best copy from the // runner-up — the rule that actually decided, not every rule it passed. next := ranked[1] switch { case best.LidarrTracked != next.LidarrTracked: return best.TrackID, "the copy Lidarr tracks" case best.TagFit != next.TagFit: return best.TrackID, "tags fit the album" case isLossless(best) != isLossless(next): return best.TrackID, "lossless (" + strings.ToLower(best.FileFormat) + ")" case best.FileSize != next.FileSize: return best.TrackID, "largest file" case !best.AddedAt.Equal(next.AddedAt): return best.TrackID, "in the library longest" default: return best.TrackID, "copies are otherwise identical" } } func survivorBefore(a, b SurvivorCandidate) bool { if a.LidarrTracked != b.LidarrTracked { return a.LidarrTracked } if a.TagFit != b.TagFit { return a.TagFit > b.TagFit } if isLossless(a) != isLossless(b) { return isLossless(a) } if a.FileSize != b.FileSize { return a.FileSize > b.FileSize } if !a.AddedAt.Equal(b.AddedAt) { return a.AddedAt.Before(b.AddedAt) } return a.TrackID < b.TrackID } func isLossless(c SurvivorCandidate) bool { return losslessFormats[strings.ToLower(c.FileFormat)] }