Recommendation metrics: publish the margin of error on every delta #122

Merged
bvandeusen merged 4 commits from dev into main 2026-08-06 21:50:42 -04:00
11 changed files with 538 additions and 50 deletions
+46 -3
View File
@@ -39,6 +39,14 @@ type surfaceMetric struct {
SkipRate float64 `json:"skip_rate"` // skips / plays, [0,1]
AvgCompletion float64 `json:"avg_completion"` // mean completion ratio, [0,1]
LowConfidence bool `json:"low_confidence"` // plays < recMetricsLowVolume
// SkipDelta / CompletionDelta are this row's difference from the manual
// baseline WITH its margin of error (#2495). nil on the baseline row
// itself, and whenever the samples are too thin for a margin to mean
// anything. Computed server-side so both clients read the same arithmetic
// instead of each re-deriving it — and so `low_confidence` is no longer
// mistaken for a decision threshold, which it never was.
SkipDelta *metricDelta `json:"skip_delta,omitempty"`
CompletionDelta *metricDelta `json:"completion_delta,omitempty"`
// Breakdown splits the family into the pick-kind populations its
// builder stamped (#1249, generalized #1270): For You's taste/fresh,
// Discover's buckets, tier1-3 for tiered mixes — plus earlier plays
@@ -119,6 +127,10 @@ type familyAccum struct {
// completionSum is avg*count re-expanded, so merging N raw rows
// reduces to a single weighted division at the end.
completionSum float64
// completionSqSum is the sum of squared completion ratios, which is what
// makes the variance mergeable across raw source rows (#2495). Standard
// deviations cannot be combined; sums of squares add exactly.
completionSqSum float64
}
func (a *familyAccum) add(row dbq.RecommendationSourceMetricsForUserRow) {
@@ -126,6 +138,7 @@ func (a *familyAccum) add(row dbq.RecommendationSourceMetricsForUserRow) {
a.skips += row.Skips
a.completionN += row.CompletionN
a.completionSum += row.AvgCompletion * float64(row.CompletionN)
a.completionSqSum += row.CompletionSqsum
}
func (a *familyAccum) metric() surfaceMetric {
@@ -145,6 +158,33 @@ func (a *familyAccum) metric() surfaceMetric {
return m
}
// completionVariance is the sample variance of this family's completion ratios.
func (a *familyAccum) completionVariance() float64 {
return sampleVariance(a.completionSum, a.completionSqSum, a.completionN)
}
// applyDeltas attaches baseline-relative deltas + margins to a metric.
// Split out so every row — parent surfaces and breakdown rows alike — goes
// through the identical arithmetic; a breakdown arm is exactly where the old
// card was most misleading, because those are the thinnest samples on screen.
func applyDeltas(m *surfaceMetric, acc *familyAccum, baseline *familyAccum) {
if baseline == nil || baseline.plays == 0 {
return
}
m.SkipDelta = proportionDelta(
m.SkipRate, acc.plays,
float64(baseline.skips)/float64(baseline.plays), baseline.plays,
)
baseMean := 0.0
if baseline.completionN > 0 {
baseMean = baseline.completionSum / float64(baseline.completionN)
}
m.CompletionDelta = meanDelta(
m.AvgCompletion, acc.completionVariance(), acc.completionN,
baseMean, baseline.completionVariance(), baseline.completionN,
)
}
// handleGetRecommendationMetrics implements GET /api/me/recommendation-metrics.
// Bucketed per-surface-family outcomes for the caller over the last `days`
// (default 30, capped at 365), grouped by surface intent and anchored by the
@@ -214,7 +254,7 @@ func pickKindFamily(parent recFamily, kind string) recFamily {
// Breakdown rows. Attached only when at least one attributed play
// exists — an all-unattributed breakdown would just repeat the parent
// row, and families that never stamp (radio, direct plays) stay flat.
func pickKindBreakdown(picks map[string]*familyAccum) []surfaceMetric {
func pickKindBreakdown(picks map[string]*familyAccum, baseline *familyAccum) []surfaceMetric {
attributed := int64(0)
for kind, acc := range picks {
if kind != "" {
@@ -227,7 +267,9 @@ func pickKindBreakdown(picks map[string]*familyAccum) []surfaceMetric {
out := make([]surfaceMetric, 0, len(picks))
for _, kind := range pickKindOrder {
if acc, ok := picks[kind]; ok && acc.plays > 0 {
out = append(out, acc.metric())
m := acc.metric()
applyDeltas(&m, acc, baseline)
out = append(out, m)
}
}
return out
@@ -287,7 +329,8 @@ func bucketMetricsResponse(
for _, acc := range families {
if acc.fam.intent == g.intent {
m := acc.metric()
m.Breakdown = pickKindBreakdown(picks[acc.fam.key])
applyDeltas(&m, acc, baseline)
m.Breakdown = pickKindBreakdown(picks[acc.fam.key], baseline)
group.Surfaces = append(group.Surfaces, m)
}
}
+117
View File
@@ -0,0 +1,117 @@
package api
import "math"
// Uncertainty on the deltas the recommendation-metrics card shows (#2495).
//
// Why this exists: the card had exactly one volume threshold,
// recMetricsLowVolume = 20, and it was doing two jobs. Twenty plays is enough to
// be worth DISPLAYING — below that a skip rate is anecdote — but it is nowhere
// near enough to ACT on. Detecting the ~13pp differences that actually matter
// needs roughly 133 plays per arm for 80% power at α=0.05.
//
// So Discover's taste-matched (59 plays) and random-unheard (70) both rendered as
// full-confidence rows with a bold delta beside them, and that comparison sits at
// p ≈ 0.06. The card said "signal"; the arithmetic said "maybe". It led directly
// to a recommendation the data didn't support, and any reader with the same
// numbers would have made the same call.
//
// The fix is to publish the margin of error next to the delta and flag when the
// delta is smaller than it — i.e. not distinguishable from zero. Computed here,
// server-side, so both clients agree rather than each re-deriving it.
//
// recMetricsLowVolume stays exactly as it was. This is a second, independent
// signal, not a replacement: "too thin to show" and "too thin to act on" are
// different questions and deserve different answers.
// deltaZ is the two-sided 95% normal critical value. Normal rather than
// Student's t: at the sample sizes where a delta is worth acting on (n in the
// hundreds) the difference is immaterial, and a household dashboard does not
// need a t-table.
const deltaZ = 1.96
// metricDelta is a difference from the baseline, with its uncertainty.
//
// Both figures are in PERCENTAGE POINTS, matching how the card reads them out —
// a skip rate of 0.153 against a baseline of 0.270 is "-11.7", not "-0.117".
type metricDelta struct {
// DeltaPP is surface minus baseline. Negative skip is better; negative
// completion is worse. The client owns that colouring.
DeltaPP float64 `json:"delta_pp"`
// MarginPP is the 95% margin of error on DeltaPP. Read the delta as
// DeltaPP ± MarginPP.
MarginPP float64 `json:"margin_pp"`
// Distinguishable reports |DeltaPP| >= MarginPP: the interval excludes
// zero, so the difference is worth reading as a difference. When false the
// number may be pure noise no matter how large it looks.
Distinguishable bool `json:"distinguishable"`
}
// proportionDelta compares two rates (skips/plays) as a two-proportion
// difference. Returns nil when either sample is empty, or when either rate is
// degenerate (0 or 1) — a rate with no observed variation has an SE of 0 on its
// side, which would report a spuriously narrow margin rather than an honest one.
func proportionDelta(rate1 float64, n1 int64, rate2 float64, n2 int64) *metricDelta {
if n1 <= 0 || n2 <= 0 {
return nil
}
v1 := rate1 * (1 - rate1) / float64(n1)
v2 := rate2 * (1 - rate2) / float64(n2)
se := math.Sqrt(v1 + v2)
if se <= 0 {
// Both rates are 0 or both are 1. The delta is exactly zero and the
// margin is meaningless; reporting nothing is more honest than
// reporting certainty.
return nil
}
return newDelta((rate1-rate2)*100, deltaZ*se*100)
}
// meanDelta compares two means (average completion ratio) using Welch's
// standard error, which does not assume equal variances between the two groups.
//
// Note the margins here are wider than intuition suggests, and that is correct:
// completion is strongly bimodal — a play is either abandoned early (≈0.05) or
// finished (≈1.0), with little in between — so its standard deviation is large
// (~0.4) even though the mean looks stable.
func meanDelta(mean1 float64, variance1 float64, n1 int64, mean2 float64, variance2 float64, n2 int64) *metricDelta {
// Two observations minimum per side: a sample variance needs n-1 > 0.
if n1 < 2 || n2 < 2 {
return nil
}
se := math.Sqrt(variance1/float64(n1) + variance2/float64(n2))
if se <= 0 || math.IsNaN(se) || math.IsInf(se, 0) {
return nil
}
return newDelta((mean1-mean2)*100, deltaZ*se*100)
}
func newDelta(deltaPP, marginPP float64) *metricDelta {
return &metricDelta{
DeltaPP: deltaPP,
MarginPP: marginPP,
// >= rather than >: a delta exactly equal to its margin sits on the
// boundary, and calling the boundary "distinguishable" is the
// conventional reading of a 95% interval that just excludes zero.
Distinguishable: math.Abs(deltaPP) >= marginPP,
}
}
// sampleVariance recovers the sample variance from the aggregates the SQL
// returns. sum is mean×n rather than a selected column, which keeps the query to
// one extra expression.
//
// The subtraction can go very slightly negative through floating-point
// cancellation when every observation is identical, so the result is clamped —
// a negative variance would produce NaN downstream.
func sampleVariance(sum, sqSum float64, n int64) float64 {
if n < 2 {
return 0
}
nf := float64(n)
v := (sqSum - (sum * sum / nf)) / (nf - 1)
if v < 0 {
return 0
}
return v
}
+152
View File
@@ -0,0 +1,152 @@
package api
import (
"math"
"testing"
)
func TestProportionDelta_ReproducesTheDiscoverCase(t *testing.T) {
// The comparison that motivated #2495: Discover taste-matched (59 plays,
// 15.3% skip) vs random-unheard (70 plays, 28.6%). A 13.3pp gap that the old
// card rendered as a confident coloured number, sitting at p ≈ 0.06.
d := proportionDelta(0.153, 59, 0.286, 70)
if d == nil {
t.Fatal("expected a delta for two real samples")
}
if math.Abs(d.DeltaPP-(-13.3)) > 0.1 {
t.Errorf("DeltaPP = %.2f, want ≈ -13.3", d.DeltaPP)
}
// This is the assertion the whole task exists for: at these sample sizes the
// margin swallows the difference.
if d.Distinguishable {
t.Errorf("13.3pp on n=59/70 reported as distinguishable (margin %.2f) — "+
"this is exactly the false confidence #2495 set out to remove", d.MarginPP)
}
if d.MarginPP <= 13.3 {
t.Errorf("MarginPP = %.2f, expected it to exceed the 13.3pp delta", d.MarginPP)
}
}
// Same effect size, ~10x the volume: now it is real. Proves the flag tracks
// sample size rather than just the size of the gap.
func TestProportionDelta_SameGapBecomesDistinguishableWithVolume(t *testing.T) {
d := proportionDelta(0.153, 600, 0.286, 700)
if d == nil {
t.Fatal("expected a delta")
}
if !d.Distinguishable {
t.Errorf("13.3pp on n=600/700 should be distinguishable (margin %.2f)", d.MarginPP)
}
}
func TestProportionDelta_SignAndDirection(t *testing.T) {
// Surface skips MORE than baseline -> positive delta (worse for skip rate).
worse := proportionDelta(0.40, 500, 0.25, 500)
if worse == nil || worse.DeltaPP <= 0 {
t.Fatalf("expected a positive delta, got %+v", worse)
}
better := proportionDelta(0.10, 500, 0.25, 500)
if better == nil || better.DeltaPP >= 0 {
t.Fatalf("expected a negative delta, got %+v", better)
}
}
func TestProportionDelta_EmptySamples(t *testing.T) {
if d := proportionDelta(0.2, 0, 0.3, 100); d != nil {
t.Errorf("n1=0 produced a delta: %+v", d)
}
if d := proportionDelta(0.2, 100, 0.3, 0); d != nil {
t.Errorf("n2=0 produced a delta: %+v", d)
}
}
// Two degenerate rates have zero standard error, which would report a margin of
// 0 and therefore "distinguishable" for a delta of exactly 0. Reporting nothing
// is the honest answer.
func TestProportionDelta_DegenerateRates(t *testing.T) {
if d := proportionDelta(0, 50, 0, 50); d != nil {
t.Errorf("both rates 0 produced a delta: %+v", d)
}
if d := proportionDelta(1, 50, 1, 50); d != nil {
t.Errorf("both rates 1 produced a delta: %+v", d)
}
// One degenerate side is still informative — the other side carries variance.
if d := proportionDelta(0, 200, 0.3, 200); d == nil {
t.Error("one degenerate rate should still yield a delta")
}
}
func TestMeanDelta(t *testing.T) {
// Completion is bimodal, so ~0.16 variance (sd ≈ 0.4) is realistic.
const v = 0.16
thin := meanDelta(0.82, v, 59, 0.54, v, 70)
if thin == nil {
t.Fatal("expected a delta")
}
if math.Abs(thin.DeltaPP-28.0) > 0.1 {
t.Errorf("DeltaPP = %.2f, want ≈ 28.0", thin.DeltaPP)
}
// 28pp is large enough to survive even a wide margin at this n.
if !thin.Distinguishable {
t.Errorf("28pp on n=59/70 with sd 0.4 should be distinguishable (margin %.2f)", thin.MarginPP)
}
// A small completion gap at the same volume should not be.
small := meanDelta(0.56, v, 59, 0.54, v, 70)
if small == nil {
t.Fatal("expected a delta")
}
if small.Distinguishable {
t.Errorf("2pp on n=59/70 reported as distinguishable (margin %.2f)", small.MarginPP)
}
}
// A sample variance needs at least two observations per side.
func TestMeanDelta_NeedsTwoObservations(t *testing.T) {
if d := meanDelta(0.8, 0.1, 1, 0.5, 0.1, 100); d != nil {
t.Errorf("n1=1 produced a delta: %+v", d)
}
if d := meanDelta(0.8, 0.1, 100, 0.5, 0.1, 1); d != nil {
t.Errorf("n2=1 produced a delta: %+v", d)
}
}
func TestMeanDelta_ZeroVarianceBothSides(t *testing.T) {
if d := meanDelta(0.8, 0, 50, 0.5, 0, 50); d != nil {
t.Errorf("zero variance on both sides produced a delta: %+v", d)
}
}
func TestSampleVariance(t *testing.T) {
// Observations 0, 1: mean 0.5, sample variance 0.5.
if got := sampleVariance(1.0, 1.0, 2); math.Abs(got-0.5) > 1e-9 {
t.Errorf("sampleVariance = %v, want 0.5", got)
}
// Identical observations -> zero variance, and must not go negative through
// floating-point cancellation.
if got := sampleVariance(4.0, 4.0, 4); got != 0 {
t.Errorf("identical observations gave variance %v, want 0", got)
}
if got := sampleVariance(0, 0, 1); got != 0 {
t.Errorf("n=1 gave variance %v, want 0", got)
}
}
// Clamping matters: a negative variance would become NaN in the square root and
// propagate into the JSON as a null-ish number.
func TestSampleVariance_NeverNegative(t *testing.T) {
// sqSum slightly below sum²/n, as cancellation can produce.
if got := sampleVariance(10.0, 24.999999999, 4); got < 0 {
t.Errorf("variance went negative: %v", got)
}
}
func TestNewDelta_BoundaryCountsAsDistinguishable(t *testing.T) {
d := newDelta(5.0, 5.0)
if !d.Distinguishable {
t.Error("a delta exactly equal to its margin should count as distinguishable")
}
d = newDelta(4.999, 5.0)
if d.Distinguishable {
t.Error("a delta just inside its margin should not count as distinguishable")
}
}
+17 -7
View File
@@ -18,7 +18,8 @@ SELECT
count(*)::bigint AS plays,
count(*) FILTER (WHERE pe.was_skipped)::bigint AS skips,
count(pe.completion_ratio)::bigint AS completion_n,
COALESCE(avg(pe.completion_ratio), 0)::float8 AS avg_completion
COALESCE(avg(pe.completion_ratio), 0)::float8 AS avg_completion,
COALESCE(sum(pe.completion_ratio * pe.completion_ratio), 0)::float8 AS completion_sqsum
FROM play_events pe
WHERE pe.user_id = $1
AND pe.started_at > now() - ($2::float8 * INTERVAL '1 day')
@@ -32,18 +33,26 @@ type RecommendationSourceMetricsForUserParams struct {
}
type RecommendationSourceMetricsForUserRow struct {
Source *string
PickKind *string
Plays int64
Skips int64
CompletionN int64
AvgCompletion float64
Source *string
PickKind *string
Plays int64
Skips int64
CompletionN int64
AvgCompletion float64
CompletionSqsum float64
}
// $1 user_id, $2 window_days. plays/skips are counts; avg_completion is the
// mean completion ratio over the completion_n plays that recorded one.
// pick_kind splits For You plays into taste/fresh/unattributed (#1249);
// it is NULL for every other source, so those still group to one row.
//
// completion_sqsum carries the sum of SQUARED completion ratios so the Go
// handler can compute a variance — needed for the margin of error on a
// completion delta (#2495). It is the sum rather than `stddev_samp` on purpose:
// raw source rows get merged into surface families in Go, and sums of squares
// add across groups exactly, whereas standard deviations cannot be combined
// without them. Variance = (sqsum - sum²/n) / (n-1), with sum = avg × n.
func (q *Queries) RecommendationSourceMetricsForUser(ctx context.Context, arg RecommendationSourceMetricsForUserParams) ([]RecommendationSourceMetricsForUserRow, error) {
rows, err := q.db.Query(ctx, recommendationSourceMetricsForUser, arg.UserID, arg.Column2)
if err != nil {
@@ -60,6 +69,7 @@ func (q *Queries) RecommendationSourceMetricsForUser(ctx context.Context, arg Re
&i.Skips,
&i.CompletionN,
&i.AvgCompletion,
&i.CompletionSqsum,
); err != nil {
return nil, err
}
@@ -44,13 +44,21 @@ ORDER BY 1, 2;
-- mean completion ratio over the completion_n plays that recorded one.
-- pick_kind splits For You plays into taste/fresh/unattributed (#1249);
-- it is NULL for every other source, so those still group to one row.
--
-- completion_sqsum carries the sum of SQUARED completion ratios so the Go
-- handler can compute a variance — needed for the margin of error on a
-- completion delta (#2495). It is the sum rather than `stddev_samp` on purpose:
-- raw source rows get merged into surface families in Go, and sums of squares
-- add across groups exactly, whereas standard deviations cannot be combined
-- without them. Variance = (sqsum - sum²/n) / (n-1), with sum = avg × n.
SELECT
pe.source,
pe.pick_kind,
count(*)::bigint AS plays,
count(*) FILTER (WHERE pe.was_skipped)::bigint AS skips,
count(pe.completion_ratio)::bigint AS completion_n,
COALESCE(avg(pe.completion_ratio), 0)::float8 AS avg_completion
COALESCE(avg(pe.completion_ratio), 0)::float8 AS avg_completion,
COALESCE(sum(pe.completion_ratio * pe.completion_ratio), 0)::float8 AS completion_sqsum
FROM play_events pe
WHERE pe.user_id = $1
AND pe.started_at > now() - ($2::float8 * INTERVAL '1 day')
+15 -2
View File
@@ -414,8 +414,21 @@ func (s *Scanner) resolveArtist(ctx context.Context, q *dbq.Queries, name, mbid
ID: existing.ID,
Mbid: &m,
}); uerr != nil {
s.logger.Warn("library scan: heal artist mbid failed",
"artist_id", existing.ID, "err", uerr)
if isUniqueViolation(uerr) {
// Another artist row already owns this MBID — two rows that
// should be merged (usually two spellings of one name).
// Expected, not a fault: leave NULL and let the operator
// merge. Mirrors resolveAlbum, which has always handled it
// this way — without this branch the identical benign
// condition logged a generic warning plus a Postgres ERROR
// line on every scan, which teaches an operator to ignore
// database errors (#2524).
s.logger.Info("library scan: duplicate artist mbid (canonical row already owns it)",
"artist_id", existing.ID, "artist", name, "mbid", mbid)
} else {
s.logger.Warn("library scan: heal artist mbid failed",
"artist_id", existing.ID, "err", uerr)
}
} else {
existing.Mbid = &m
}
+18
View File
@@ -4,6 +4,20 @@ import { api } from './client';
// Mirrors internal/api/me_recommendation_metrics.go: raw play sources are
// bucketed server-side into stable surface families, grouped by intent, and
// anchored by the manual-plays baseline (milestone #127).
// A difference from the baseline, with its uncertainty (#2495). Both figures
// are already in percentage points — the server does the arithmetic so both
// clients read the same numbers.
//
// `distinguishable: false` means |delta_pp| < margin_pp: the delta cannot be
// told apart from zero, however large it looks. That distinction is the whole
// point of this type — `low_confidence` answers "is this worth showing?", which
// is a much lower bar than "is this worth acting on?".
export type MetricDelta = {
delta_pp: number;
margin_pp: number;
distinguishable: boolean;
};
export type SurfaceMetric = {
key: string;
label: string;
@@ -12,6 +26,10 @@ export type SurfaceMetric = {
skip_rate: number;
avg_completion: number;
low_confidence: boolean;
// Absent on the baseline row itself, and whenever the samples are too thin
// for a margin to mean anything.
skip_delta?: MetricDelta;
completion_delta?: MetricDelta;
// Present when the surface's builder stamps pick-kind provenance and
// the window holds attributed plays (#1249, generalized #1270): For
// You's taste/fresh split, Discover's candidate buckets, the tiered
+21 -7
View File
@@ -216,9 +216,17 @@
return plays > 0 ? hits / plays : 0;
}
function latest(s: TrendSeries): { skip: number; completion: number } {
// The "latest" columns are ONE WEEK while the Plays column is the whole
// window, which is a trap: a 40% skip rate off 17 plays sat next to a
// four-figure Plays total and read as a solid signal. It isn't — I misread
// exactly this and briefly concluded Deep cuts was the worst surface, when
// over 180 days it's one of the best (#2495). So the week's own play count
// comes back with the rates and is rendered beside them.
function latest(s: TrendSeries): { skip: number; completion: number; plays: number } {
const last = s.points[s.points.length - 1];
return last ? { skip: last.skip_rate, completion: last.avg_completion } : { skip: 0, completion: 0 };
return last
? { skip: last.skip_rate, completion: last.avg_completion, plays: last.plays }
: { skip: 0, completion: 0, plays: 0 };
}
function pct(v: number): string {
@@ -423,6 +431,9 @@
Skip rate per surface over the last {trends?.weeks ?? 12} weeks (lower is better; all
users aggregated, rates only). Dashed ticks mark tuning changes. Taste hit is the share
of plays whose artist fits the current taste profile.
<span class="font-medium">The skip and completion columns show the most recent week
alone</span>, not the whole window — the figure after the skip rate is that week's
play count, so a rate drawn from a handful of listens reads as what it is.
</p>
</div>
{#if trendsFailed}
@@ -439,10 +450,10 @@
<tr class="text-left text-text-secondary">
<th class="py-1 font-medium">Surface</th>
<th class="py-1 font-medium">Skip rate by week</th>
<th class="py-1 text-right font-medium">Plays</th>
<th class="py-1 text-right font-medium">Latest skip</th>
<th class="py-1 text-right font-medium">Latest completion</th>
<th class="py-1 text-right font-medium">Taste hit</th>
<th class="py-1 text-right font-medium">Plays<span class="font-normal text-xs"> (window)</span></th>
<th class="py-1 text-right font-medium">Skip<span class="font-normal text-xs"> (last wk)</span></th>
<th class="py-1 text-right font-medium">Completion<span class="font-normal text-xs"> (last wk)</span></th>
<th class="py-1 text-right font-medium">Taste hit<span class="font-normal text-xs"> (window)</span></th>
</tr>
</thead>
<tbody>
@@ -493,7 +504,10 @@
</svg>
</td>
<td class="py-1.5 text-right tabular-nums">{s.plays}</td>
<td class="py-1.5 text-right tabular-nums">{pct(latest(s).skip)}</td>
<td class="py-1.5 text-right tabular-nums">
{pct(latest(s).skip)}
<span class="text-xs text-text-secondary">/{latest(s).plays}</span>
</td>
<td class="py-1.5 text-right tabular-nums">{pct(latest(s).completion)}</td>
<td class="py-1.5 text-right tabular-nums">{pct(windowTasteHitRate(s))}</td>
</tr>
+23 -3
View File
@@ -195,9 +195,29 @@ describe('Admin tuning page', () => {
await waitFor(() => expect(screen.getByText('Weekly trends')).toBeInTheDocument());
expect(screen.getByTestId('sparkline-radio')).toBeInTheDocument();
expect(screen.getByTestId('sparkline-discover')).toBeInTheDocument();
// Latest skip rate column for radio = 40% (also discover's latest
// completion, hence getAllBy).
expect(screen.getAllByText('40%').length).toBeGreaterThan(0);
// The skip column is the LAST WEEK's rate, now carrying that week's play
// count so a rate off a handful of plays reads as what it is (#2495).
// Radio's latest week: 40% skip over 15 plays; Discover's: 60% over 5.
expect(screen.getByText('/15')).toBeInTheDocument();
expect(screen.getByText('/5')).toBeInTheDocument();
// Completion columns are unchanged and still bare percentages — radio 70%.
expect(screen.getByText('70%')).toBeInTheDocument();
// '40%' is now genuinely ambiguous: radio's latest SKIP rate and discover's
// latest COMPLETION are both 40%. testing-library matches an element's own
// direct text nodes, so the skip cell still matches despite its trailing
// play-count span. Assert the count rather than pretending it's unique.
expect(screen.getAllByText('40%')).toHaveLength(2);
// The window/last-week distinction has to be visible in the headers, or the
// Plays total reads as the denominator of the skip rate. That misreading is
// what #2495 was filed over.
//
// Queried as column headers rather than by text: the caption below also
// mentions "Plays", and matching on the word finds the prose too.
// Exact accessible names: /^Skip/ also matches the "Skip rate by week"
// sparkline column, and /Plays/ matched the caption prose before that.
expect(screen.getByRole('columnheader', { name: 'Plays (window)' })).toBeInTheDocument();
expect(screen.getByRole('columnheader', { name: 'Skip (last wk)' })).toBeInTheDocument();
expect(screen.getByRole('columnheader', { name: 'Completion (last wk)' })).toBeInTheDocument();
// The knob turn is listed under the chart AND tooltipped on each
// sparkline's marker tick, hence getAllBy.
expect(
+52 -27
View File
@@ -12,7 +12,7 @@
createRecommendationMetricsQuery,
type RecommendationMetrics,
type SurfaceIntent,
type SurfaceMetric
type MetricDelta
} from '$lib/api/metrics';
import { theme, setTheme, type ThemePreference } from '$lib/stores/theme.svelte';
import { player, setCrossfade } from '$lib/player/store.svelte';
@@ -45,20 +45,39 @@
return `${(v * 100).toFixed(0)}%`;
}
// Delta in percentage points vs the baseline, signed ("+12" / "5").
function deltaPts(value: number, baseline: number): string {
const pts = Math.round((value - baseline) * 100);
return pts > 0 ? `+${pts}` : `${pts}`;
// Deltas come from the server with their margin of error (#2495). The client
// no longer subtracts rates itself: the margin needs the sample sizes and
// variances, and having both clients re-derive it invites them to disagree.
//
// A delta that isn't distinguishable from zero is prefixed "≈" and dimmed.
// That is the point of this whole change — the card used to render a 12 on
// 59 plays exactly as boldly as a 6 on 400, and the first of those is noise.
function deltaText(d: MetricDelta | undefined): string {
if (!d) return '';
const pts = Math.round(d.delta_pp);
const signed = pts > 0 ? `+${pts}` : `${pts}`;
return d.distinguishable ? signed : `≈${signed}`;
}
// A surface's skip delta is "worse" when it skips more than the
// baseline; completion delta is "worse" when it completes less.
function skipDeltaClass(m: SurfaceMetric, baseline: SurfaceMetric): string {
return m.skip_rate > baseline.skip_rate ? 'text-danger' : 'text-text-secondary';
function deltaTitle(d: MetricDelta | undefined): string | undefined {
if (!d) return undefined;
const range = `${d.delta_pp.toFixed(1)} ± ${d.margin_pp.toFixed(1)} points vs baseline`;
return d.distinguishable
? `${range} (95% confidence)`
: `${range} — not distinguishable from zero at 95% confidence, so read this as no measured difference.`;
}
function completionDeltaClass(m: SurfaceMetric, baseline: SurfaceMetric): string {
return m.avg_completion < baseline.avg_completion ? 'text-danger' : 'text-text-secondary';
// A skip delta is "worse" above the baseline; a completion delta is "worse"
// below it. Neither gets a colour unless it's distinguishable — colouring
// noise red is what made the old card misleading.
function skipDeltaClass(d: MetricDelta | undefined): string {
if (!d?.distinguishable) return 'text-text-secondary opacity-60';
return d.delta_pp > 0 ? 'text-danger' : 'text-text-secondary';
}
function completionDeltaClass(d: MetricDelta | undefined): string {
if (!d?.distinguishable) return 'text-text-secondary opacity-60';
return d.delta_pp < 0 ? 'text-danger' : 'text-text-secondary';
}
// Pick-kind breakdowns are collapsed by default (#1270): with every
@@ -384,18 +403,18 @@
<td class="py-1 text-right tabular-nums">{m.plays}</td>
<td class="py-1 text-right tabular-nums">
{pct(m.skip_rate)}
{#if baseline}
<span class="ml-1 text-xs {skipDeltaClass(m, baseline)}">
{deltaPts(m.skip_rate, baseline.skip_rate)}
</span>
{#if m.skip_delta}
<span class="ml-1 text-xs {skipDeltaClass(m.skip_delta)}"
data-testid="skip-delta-{m.key}"
title={deltaTitle(m.skip_delta)}>{deltaText(m.skip_delta)}</span>
{/if}
</td>
<td class="py-1 text-right tabular-nums">
{pct(m.avg_completion)}
{#if baseline}
<span class="ml-1 text-xs {completionDeltaClass(m, baseline)}">
{deltaPts(m.avg_completion, baseline.avg_completion)}
</span>
{#if m.completion_delta}
<span class="ml-1 text-xs {completionDeltaClass(m.completion_delta)}"
data-testid="completion-delta-{m.key}"
title={deltaTitle(m.completion_delta)}>{deltaText(m.completion_delta)}</span>
{/if}
</td>
</tr>
@@ -420,18 +439,18 @@
<td class="py-1 text-right text-xs tabular-nums">{b.plays}</td>
<td class="py-1 text-right text-xs tabular-nums">
{pct(b.skip_rate)}
{#if baseline}
<span class="ml-1 {skipDeltaClass(b, baseline)}">
{deltaPts(b.skip_rate, baseline.skip_rate)}
</span>
{#if b.skip_delta}
<span class="ml-1 {skipDeltaClass(b.skip_delta)}"
data-testid="skip-delta-{b.key}"
title={deltaTitle(b.skip_delta)}>{deltaText(b.skip_delta)}</span>
{/if}
</td>
<td class="py-1 text-right text-xs tabular-nums">
{pct(b.avg_completion)}
{#if baseline}
<span class="ml-1 {completionDeltaClass(b, baseline)}">
{deltaPts(b.avg_completion, baseline.avg_completion)}
</span>
{#if b.completion_delta}
<span class="ml-1 {completionDeltaClass(b.completion_delta)}"
data-testid="completion-delta-{b.key}"
title={deltaTitle(b.completion_delta)}>{deltaText(b.completion_delta)}</span>
{/if}
</td>
</tr>
@@ -442,6 +461,12 @@
</table>
</div>
{/each}
<p class="text-xs text-text-secondary">
Deltas compare each surface with your manual plays. A delta marked
<span class="opacity-60"></span> is smaller than its own margin of error at this
sample size — it can't be told apart from no difference, however big it looks.
Hover any delta for its range.
</p>
{:else}
<p class="text-sm text-text-secondary">
No plays recorded yet. Play something from For You, Discover, or a mix.
+68
View File
@@ -241,6 +241,74 @@ describe('Settings page — Recommendation metrics card', () => {
expect(screen.queryByText(/Taste picks/)).not.toBeInTheDocument();
});
// #2495: the card used to render a delta computed client-side with no notion
// of uncertainty, so a -12 on 59 plays looked exactly as solid as a -6 on 400.
// Deltas now arrive from the server with a margin, and an indistinguishable
// one is marked with "≈" and dimmed rather than coloured.
test('a delta smaller than its margin is marked as indistinguishable', async () => {
setupPage();
metricsMock.data = {
window_days: 30,
baseline: metric('manual', 'Manual library plays', { plays: 400, skip_rate: 0.27 }),
groups: [
{
intent: 'discovery',
label: 'Discovery mixes',
surfaces: [
metric('discover', 'Discover', {
plays: 59,
skip_rate: 0.153,
// 13.3pp gap, but the margin at n=59 is wider than the gap.
skip_delta: { delta_pp: -13.3, margin_pp: 14.5, distinguishable: false },
completion_delta: { delta_pp: 28.0, margin_pp: 12.1, distinguishable: true }
})
]
}
]
};
render(SettingsPage);
await waitFor(() => expect(screen.getByText('Discover')).toBeInTheDocument());
// Targeted by test id rather than text: the legend below the table also
// contains a "≈", so matching on the glyph finds the explanation instead of
// the delta. (It did, on the first attempt at this test.)
const skip = screen.getByTestId('skip-delta-discover');
expect(skip).toHaveTextContent('≈-13');
expect(skip).toHaveAttribute('title', expect.stringContaining('not distinguishable from zero'));
// It must NOT be coloured as a real regression/improvement.
expect(skip.className).toContain('opacity-60');
// The completion delta clears its margin, so it renders plainly.
const completion = screen.getByTestId('completion-delta-discover');
expect(completion).toHaveTextContent('+28');
expect(completion.className).not.toContain('opacity-60');
// And the legend explains the glyph rather than leaving it a mystery.
expect(screen.getByText(/smaller than its own margin of error/i)).toBeInTheDocument();
});
// A delta is omitted entirely when the samples are too thin for a margin to
// mean anything — the server decides that, and the cell must simply show the
// rate rather than a bare "0".
test('a surface with no delta shows its rate and nothing else', async () => {
setupPage();
metricsMock.data = {
window_days: 30,
baseline: metric('manual', 'Manual library plays', { plays: 400 }),
groups: [
{
intent: 'go_to',
label: 'Go-to surfaces',
surfaces: [metric('radio', 'Radio', { plays: 1, skip_rate: 0 })]
}
]
};
render(SettingsPage);
await waitFor(() => expect(screen.getByText('Radio')).toBeInTheDocument());
expect(screen.queryByTestId('skip-delta-radio')).not.toBeInTheDocument();
expect(screen.queryByTestId('completion-delta-radio')).not.toBeInTheDocument();
});
test('surfaces without a breakdown render no toggle and no sub-rows', async () => {
setupPage();
metricsMock.data = {