test-go / test (push) Successful in 1m10s
test-go / integration (push) Successful in 3m35s
release / Build signed APK (releases and dev) (push) Successful in 5m6s
release / Build + push container image (push) Successful in 16s
release / Verify release artifacts (tag releases only) (push) Skipped
Fixes the integration failure from f367eeaa:
"same-day rebuild produced different track lists".
Cutting RandomFill 30→10 for Songs-like broke
TestBuildSystemPlaylists_DailyNonceDeterminism, and the reason is worth
stating because the number is not the bug.
`likes_overlap` and `random_fill` end in a bare `ORDER BY random()` with no
daily seed (recommendation.sql:118, :161). Such an arm returns a STABLE set
only while its LIMIT exceeds the rows eligible for it — then it returns all
of them, and the random order stops mattering because scoreAndSortCandidates
sorts by track id before drawing jitter. Below that threshold the arm
returns a random SUBSET, and two builds on the same day draw different ones.
So the test was green by accident. It seeds ~20 tracks against a default
RandomFill of 30; the limit exceeded the library, so the arm returned
everything. Determinism held for a reason unrelated to the code being right.
Which means it does NOT hold in production. Any real library is larger than
30, so same-day rebuilds have been drawing different mixes since that arm
was written — invisible, because a mix changing after a refresh looks like a
feature. Filed as #3889; the fix is a seeded ordering per (user, day), which
needs a .sql change and sqlc regeneration and so cannot land from here.
The correction: grow an arm freely, never shrink one whose ordering is
unseeded random. LikesOverlap and RandomFill go back to the defaults;
TasteOverlap stays halved because it sorts by `tpa.weight DESC, t.id` and is
genuinely deterministic. Guarded by a test that names the reasoning, so the
next person to trim these has to read why first — and it should be DELETED
once #3889 lands rather than worked around.
The cost is honest: the seed-independent share of the Songs-like pool falls
from 29% to 20% instead of the intended cut. That matters less than it
sounds. The pool only biases the draw; the songs_like WEIGHTS are what
actually demote sim_score-0 candidates, and they are untouched here — a
perfect match still scores 5.00 against an unrelated favourite's 2.00.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SQ31KQpYbStyK5y58UmPLH
327 lines
12 KiB
Go
327 lines
12 KiB
Go
package recommendation
|
||
|
||
import (
|
||
"context"
|
||
"encoding/json"
|
||
"time"
|
||
|
||
"github.com/jackc/pgx/v5/pgtype"
|
||
|
||
"git.fabledsword.com/bvandeusen/minstrel/internal/db/dbq"
|
||
"git.fabledsword.com/bvandeusen/minstrel/internal/mood"
|
||
)
|
||
|
||
// LoadCandidates fetches the candidate pool for radio scoring. Combines
|
||
// the existing track+stats query with a one-shot bulk fetch of the user's
|
||
// active contextual_likes, mapping each candidate to its max similarity
|
||
// against currentVector. Pass currentVector with Seed=true to short-circuit
|
||
// the contextual term to 0 (cold-start path).
|
||
func LoadCandidates(
|
||
ctx context.Context,
|
||
q *dbq.Queries,
|
||
userID, seedID pgtype.UUID,
|
||
recentlyPlayedHours int,
|
||
currentVector SessionVector,
|
||
) ([]Candidate, error) {
|
||
rows, err := q.LoadRadioCandidates(ctx, dbq.LoadRadioCandidatesParams{
|
||
UserID: userID,
|
||
ID: seedID,
|
||
Column3: float64(recentlyPlayedHours),
|
||
})
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
likes, err := loadContextualLikesByTrack(ctx, q, userID)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
profile, err := LoadTasteProfile(ctx, q, userID)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
affinity, err := LoadContextAffinity(ctx, q, userID, currentVector.DeviceClass)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
out := make([]Candidate, 0, len(rows))
|
||
for _, r := range rows {
|
||
var lpt *time.Time
|
||
if r.LastPlayedAt.Valid {
|
||
t := r.LastPlayedAt.Time
|
||
lpt = &t
|
||
}
|
||
ctxScore := ContextualMatchScore(currentVector, likes[r.Track.ID], DefaultSimilarityWeights)
|
||
out = append(out, Candidate{
|
||
Track: r.Track,
|
||
Inputs: ScoringInputs{
|
||
IsGeneralLiked: r.IsLiked,
|
||
LastPlayedAt: lpt,
|
||
PlayCount: int(r.PlayCount),
|
||
SkipCount: int(r.SkipCount),
|
||
ContextualMatchScore: ctxScore,
|
||
// Fallback path: mood is scored only in the primary
|
||
// (similarity) loader — loading per-candidate tags over this
|
||
// near-whole-library pool isn't worth it (nil moods → 0).
|
||
TasteMatchScore: profile.Match(r.Track.ArtistID, r.Track.Genre, r.ReleaseDate, nil),
|
||
ContextAffinityScore: affinity.Affinity(r.Track.ArtistID),
|
||
},
|
||
})
|
||
}
|
||
return out, nil
|
||
}
|
||
|
||
// CandidateSourceLimits controls per-source K values for the M4c
|
||
// similarity-driven pool. Defaults via DefaultCandidateSourceLimits().
|
||
type CandidateSourceLimits struct {
|
||
LBSimilar int
|
||
SimilarArtist int
|
||
TagOverlap int
|
||
LikesOverlap int
|
||
RandomFill int
|
||
// TasteOverlap (#796 phase 2b): tracks by the user's top positively-
|
||
// weighted taste-profile artists. 0 disables the arm (e.g. cold-start
|
||
// users have an empty profile, so it contributes nothing anyway).
|
||
TasteOverlap int
|
||
// UserCoplay (#1533): tracks by artists co-played across the instance
|
||
// with the seed's artist (source='user_cooccurrence'). Empty on
|
||
// single-user servers, so it contributes nothing there.
|
||
UserCoplay int
|
||
}
|
||
|
||
// DefaultCandidateSourceLimits returns the v1 hardcoded constants per spec.
|
||
func DefaultCandidateSourceLimits() CandidateSourceLimits {
|
||
return CandidateSourceLimits{
|
||
LBSimilar: 30,
|
||
SimilarArtist: 30,
|
||
TagOverlap: 20,
|
||
LikesOverlap: 20,
|
||
RandomFill: 30,
|
||
TasteOverlap: 20,
|
||
UserCoplay: 20,
|
||
}
|
||
}
|
||
|
||
// SongsLikeCandidateSourceLimits is the pool shape for "Songs like {X}"
|
||
// (#3881). Same total size as the default (~170) — the composition is what
|
||
// changes, shifted hard toward arms that actually measure distance from the
|
||
// seed.
|
||
//
|
||
// The surface answers "what sounds like THIS", and it shared the default
|
||
// pool with For-You, which answers the much broader "what will they enjoy
|
||
// today". Under the default, 50 of ~170 candidates carried sim_score = 0 by
|
||
// construction — `taste_overlap` (tracks by the user's top taste artists) and
|
||
// `random_fill` (literally any track not already in the pool), both of which
|
||
// are seed-INDEPENDENT. Nearly a third of the pool had no relationship to the
|
||
// seed at all, and the operator saw it: "I was getting a seeming wide variety
|
||
// of music from each one when I was hoping to stay in a certain neighborhood."
|
||
//
|
||
// TIERED, per rule 131 — a system mix degrades, it never vanishes:
|
||
//
|
||
// tier 1 lb_similar real track-level similarity. The exact promise.
|
||
// tier 2 similar_artist, tag_overlap, coplay, likes_overlap
|
||
// seed-RELATED but weaker signal.
|
||
// tier 3 taste_overlap, random_fill
|
||
// seed-independent. The floor, and nothing more.
|
||
//
|
||
// The tiering is enforced by SCORE rather than by a fallback ladder: tier-3
|
||
// arms carry sim_score 0, so under SongsLike weights (SimilarityWeight 4.0,
|
||
// everything seed-independent demoted) they rank below any real match and
|
||
// surface only when tiers 1–2 cannot fill the mix. That is the rule's
|
||
// "fill from tier 1 first, reach down only when a tier cannot fill".
|
||
//
|
||
// Which is exactly why tier 3 is REDUCED rather than removed. Zeroing those
|
||
// two arms was the first instinct and it is the vanish-or-nothing shape rule
|
||
// 131 exists to forbid: a seed whose artist has thin ListenBrainz coverage
|
||
// would produce a short mix or none at all, and "no playlist" is a worse
|
||
// answer than "a few tracks further from the seed than we would like".
|
||
//
|
||
// DO NOT SHRINK AN ARM ORDERED BY UNSEEDED random(). This is the constraint
|
||
// that shapes the numbers below, and it is not obvious from reading them.
|
||
//
|
||
// `likes_overlap` and `random_fill` both end in a bare `ORDER BY random()`
|
||
// (recommendation.sql:118, :161) with no daily seed. Such an arm returns a
|
||
// STABLE set only while its LIMIT exceeds the rows eligible for it — at that
|
||
// point it returns all of them and the random order is irrelevant, because
|
||
// the caller sorts by id before scoring. Drop the limit below the eligible
|
||
// count and the arm starts returning a random SUBSET, which differs between
|
||
// two builds on the same day.
|
||
//
|
||
// That is a real defect (#3889) rather than a quirk of this function, and it
|
||
// bit here: cutting RandomFill to 10 broke
|
||
// TestBuildSystemPlaylists_DailyNonceDeterminism, whose library is smaller
|
||
// than the default limit and whose determinism was therefore accidental.
|
||
// Growing an arm is always safe; only shrinking one is.
|
||
//
|
||
// So the seed-independent arms are trimmed only where the ordering is
|
||
// deterministic: `taste_overlap` sorts by `tpa.weight DESC, t.id` and can be
|
||
// cut, `random_fill` cannot. The reduction is consequently modest — and it
|
||
// matters less than it looks, because the WEIGHTS are what demote sim_score-0
|
||
// candidates now. The pool change biases the draw; the songs_like profile is
|
||
// what actually keeps unrelated tracks out of the result.
|
||
//
|
||
// One arm is left alone that arguably should not be: `likes_overlap` assigns
|
||
// a FLAT 0.6 sim_score (recommendation.sql:108) rather than measuring
|
||
// anything — a collaborative signal wearing similarity's clothes, which a
|
||
// raised SimilarityWeight amplifies. If real ListenBrainz scores commonly
|
||
// land below 0.6 it will outrank genuine matches. It cannot be trimmed here
|
||
// without the determinism fix landing first; the honest repair is to stop it
|
||
// claiming a similarity score it never computed (#3879).
|
||
func SongsLikeCandidateSourceLimits() CandidateSourceLimits {
|
||
return CandidateSourceLimits{
|
||
LBSimilar: 60, // tier 1 — doubled; the only arm that measures the seed
|
||
SimilarArtist: 40, // tier 2 — raised; growing is always safe
|
||
TagOverlap: 20, // tier 2
|
||
UserCoplay: 20, // tier 2
|
||
LikesOverlap: 20, // tier 2 — NOT trimmed: unseeded random(), see above
|
||
TasteOverlap: 10, // tier 3 floor — halved; deterministic ordering, safe
|
||
RandomFill: 30, // tier 3 floor — NOT trimmed: unseeded random(), see above
|
||
}
|
||
}
|
||
|
||
// LoadCandidatesFromSimilarity is M4c's primary candidate-pool loader.
|
||
// 5-way SQL UNION (LB-similar / similar-artist tracks / MB-tag overlap /
|
||
// likes-overlap / random fill) + dedup-by-max sim_score. Returns
|
||
// []Candidate (same shape as LoadCandidates) so Shuffle is unchanged.
|
||
//
|
||
// Caller (radio handler) falls back to LoadCandidates on error.
|
||
func LoadCandidatesFromSimilarity(
|
||
ctx context.Context,
|
||
q *dbq.Queries,
|
||
userID, seedID pgtype.UUID,
|
||
recentlyPlayedHours int,
|
||
currentVector SessionVector,
|
||
exclude []pgtype.UUID,
|
||
limits CandidateSourceLimits,
|
||
) ([]Candidate, error) {
|
||
if exclude == nil {
|
||
exclude = []pgtype.UUID{}
|
||
}
|
||
rows, err := q.LoadRadioCandidatesV2(ctx, dbq.LoadRadioCandidatesV2Params{
|
||
UserID: userID,
|
||
ID: seedID,
|
||
Column3: int64(recentlyPlayedHours),
|
||
Column4: exclude,
|
||
Limit: int32(limits.LBSimilar),
|
||
Limit_2: int32(limits.SimilarArtist),
|
||
Limit_3: int32(limits.TagOverlap),
|
||
Limit_4: int32(limits.LikesOverlap),
|
||
Limit_5: int32(limits.RandomFill),
|
||
Limit_6: int32(limits.TasteOverlap),
|
||
Limit_7: int32(limits.UserCoplay),
|
||
})
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
likes, err := loadContextualLikesByTrack(ctx, q, userID)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
profile, err := LoadTasteProfile(ctx, q, userID)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
affinity, err := LoadContextAffinity(ctx, q, userID, currentVector.DeviceClass)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
trackIDs := make([]pgtype.UUID, 0, len(rows))
|
||
for _, r := range rows {
|
||
trackIDs = append(trackIDs, r.Track.ID)
|
||
}
|
||
moods, err := loadCandidateMoods(ctx, q, trackIDs)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
|
||
out := make([]Candidate, 0, len(rows))
|
||
for _, r := range rows {
|
||
var lpt *time.Time
|
||
if r.LastPlayedAt.Valid {
|
||
t := r.LastPlayedAt.Time
|
||
lpt = &t
|
||
}
|
||
// sqlc returns SimilarityScore as interface{} (couldn't infer the
|
||
// type through max(...) over a UNION). Type-assert; default to 0
|
||
// on the (impossible-but-defensive) case where it's nil/wrong type.
|
||
var simScore float64
|
||
if v, ok := r.SimilarityScore.(float64); ok {
|
||
simScore = v
|
||
}
|
||
ctxScore := ContextualMatchScore(currentVector, likes[r.Track.ID], DefaultSimilarityWeights)
|
||
out = append(out, Candidate{
|
||
Track: r.Track,
|
||
Inputs: ScoringInputs{
|
||
IsGeneralLiked: r.IsLiked,
|
||
LastPlayedAt: lpt,
|
||
PlayCount: int(r.PlayCount),
|
||
SkipCount: int(r.SkipCount),
|
||
ContextualMatchScore: ctxScore,
|
||
SimilarityScore: simScore,
|
||
TasteMatchScore: profile.Match(
|
||
r.Track.ArtistID, r.Track.Genre, r.ReleaseDate, moods[r.Track.ID]),
|
||
ContextAffinityScore: affinity.Affinity(r.Track.ArtistID),
|
||
},
|
||
})
|
||
}
|
||
return out, nil
|
||
}
|
||
|
||
// loadCandidateMoods fetches the enriched tags for the given candidate tracks
|
||
// and reduces each to its canonical mood buckets (internal/mood, #1534), so the
|
||
// scorer can apply the mood facet per candidate. Tracks with no mood-word tags
|
||
// are absent from the map (→ no mood signal). Empty input short-circuits.
|
||
func loadCandidateMoods(
|
||
ctx context.Context, q *dbq.Queries, trackIDs []pgtype.UUID,
|
||
) (map[pgtype.UUID][]string, error) {
|
||
if len(trackIDs) == 0 {
|
||
return map[pgtype.UUID][]string{}, nil
|
||
}
|
||
rows, err := q.ListTrackTagsForTracks(ctx, trackIDs)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
tagsByTrack := make(map[pgtype.UUID][]string)
|
||
for _, r := range rows {
|
||
tagsByTrack[r.TrackID] = append(tagsByTrack[r.TrackID], r.Tag)
|
||
}
|
||
out := make(map[pgtype.UUID][]string, len(tagsByTrack))
|
||
for id, tags := range tagsByTrack {
|
||
if m := mood.Of(tags); len(m) > 0 {
|
||
out[id] = m
|
||
}
|
||
}
|
||
return out, nil
|
||
}
|
||
|
||
// loadContextualLikesByTrack fetches the user's active contextual_likes in
|
||
// one query and groups them by track_id. Rows whose session_vector fails
|
||
// to unmarshal are skipped with no error (don't poison scoring over one
|
||
// bad row); the SQL query already filters NULL vectors.
|
||
func loadContextualLikesByTrack(
|
||
ctx context.Context,
|
||
q *dbq.Queries,
|
||
userID pgtype.UUID,
|
||
) (map[pgtype.UUID][]SessionVector, error) {
|
||
rows, err := q.ListActiveContextualLikesForUser(ctx, userID)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
out := make(map[pgtype.UUID][]SessionVector, len(rows))
|
||
for _, r := range rows {
|
||
var v SessionVector
|
||
if err := json.Unmarshal(r.SessionVector, &v); err != nil {
|
||
continue
|
||
}
|
||
out[r.TrackID] = append(out[r.TrackID], v)
|
||
}
|
||
return out, nil
|
||
}
|