Recommendation metrics: publish the margin of error on every delta #122
@@ -39,6 +39,14 @@ type surfaceMetric struct {
|
|||||||
SkipRate float64 `json:"skip_rate"` // skips / plays, [0,1]
|
SkipRate float64 `json:"skip_rate"` // skips / plays, [0,1]
|
||||||
AvgCompletion float64 `json:"avg_completion"` // mean completion ratio, [0,1]
|
AvgCompletion float64 `json:"avg_completion"` // mean completion ratio, [0,1]
|
||||||
LowConfidence bool `json:"low_confidence"` // plays < recMetricsLowVolume
|
LowConfidence bool `json:"low_confidence"` // plays < recMetricsLowVolume
|
||||||
|
// SkipDelta / CompletionDelta are this row's difference from the manual
|
||||||
|
// baseline WITH its margin of error (#2495). nil on the baseline row
|
||||||
|
// itself, and whenever the samples are too thin for a margin to mean
|
||||||
|
// anything. Computed server-side so both clients read the same arithmetic
|
||||||
|
// instead of each re-deriving it — and so `low_confidence` is no longer
|
||||||
|
// mistaken for a decision threshold, which it never was.
|
||||||
|
SkipDelta *metricDelta `json:"skip_delta,omitempty"`
|
||||||
|
CompletionDelta *metricDelta `json:"completion_delta,omitempty"`
|
||||||
// Breakdown splits the family into the pick-kind populations its
|
// Breakdown splits the family into the pick-kind populations its
|
||||||
// builder stamped (#1249, generalized #1270): For You's taste/fresh,
|
// builder stamped (#1249, generalized #1270): For You's taste/fresh,
|
||||||
// Discover's buckets, tier1-3 for tiered mixes — plus earlier plays
|
// Discover's buckets, tier1-3 for tiered mixes — plus earlier plays
|
||||||
@@ -119,6 +127,10 @@ type familyAccum struct {
|
|||||||
// completionSum is avg*count re-expanded, so merging N raw rows
|
// completionSum is avg*count re-expanded, so merging N raw rows
|
||||||
// reduces to a single weighted division at the end.
|
// reduces to a single weighted division at the end.
|
||||||
completionSum float64
|
completionSum float64
|
||||||
|
// completionSqSum is the sum of squared completion ratios, which is what
|
||||||
|
// makes the variance mergeable across raw source rows (#2495). Standard
|
||||||
|
// deviations cannot be combined; sums of squares add exactly.
|
||||||
|
completionSqSum float64
|
||||||
}
|
}
|
||||||
|
|
||||||
func (a *familyAccum) add(row dbq.RecommendationSourceMetricsForUserRow) {
|
func (a *familyAccum) add(row dbq.RecommendationSourceMetricsForUserRow) {
|
||||||
@@ -126,6 +138,7 @@ func (a *familyAccum) add(row dbq.RecommendationSourceMetricsForUserRow) {
|
|||||||
a.skips += row.Skips
|
a.skips += row.Skips
|
||||||
a.completionN += row.CompletionN
|
a.completionN += row.CompletionN
|
||||||
a.completionSum += row.AvgCompletion * float64(row.CompletionN)
|
a.completionSum += row.AvgCompletion * float64(row.CompletionN)
|
||||||
|
a.completionSqSum += row.CompletionSqsum
|
||||||
}
|
}
|
||||||
|
|
||||||
func (a *familyAccum) metric() surfaceMetric {
|
func (a *familyAccum) metric() surfaceMetric {
|
||||||
@@ -145,6 +158,33 @@ func (a *familyAccum) metric() surfaceMetric {
|
|||||||
return m
|
return m
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// completionVariance is the sample variance of this family's completion ratios.
|
||||||
|
func (a *familyAccum) completionVariance() float64 {
|
||||||
|
return sampleVariance(a.completionSum, a.completionSqSum, a.completionN)
|
||||||
|
}
|
||||||
|
|
||||||
|
// applyDeltas attaches baseline-relative deltas + margins to a metric.
|
||||||
|
// Split out so every row — parent surfaces and breakdown rows alike — goes
|
||||||
|
// through the identical arithmetic; a breakdown arm is exactly where the old
|
||||||
|
// card was most misleading, because those are the thinnest samples on screen.
|
||||||
|
func applyDeltas(m *surfaceMetric, acc *familyAccum, baseline *familyAccum) {
|
||||||
|
if baseline == nil || baseline.plays == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
m.SkipDelta = proportionDelta(
|
||||||
|
m.SkipRate, acc.plays,
|
||||||
|
float64(baseline.skips)/float64(baseline.plays), baseline.plays,
|
||||||
|
)
|
||||||
|
baseMean := 0.0
|
||||||
|
if baseline.completionN > 0 {
|
||||||
|
baseMean = baseline.completionSum / float64(baseline.completionN)
|
||||||
|
}
|
||||||
|
m.CompletionDelta = meanDelta(
|
||||||
|
m.AvgCompletion, acc.completionVariance(), acc.completionN,
|
||||||
|
baseMean, baseline.completionVariance(), baseline.completionN,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
// handleGetRecommendationMetrics implements GET /api/me/recommendation-metrics.
|
// handleGetRecommendationMetrics implements GET /api/me/recommendation-metrics.
|
||||||
// Bucketed per-surface-family outcomes for the caller over the last `days`
|
// Bucketed per-surface-family outcomes for the caller over the last `days`
|
||||||
// (default 30, capped at 365), grouped by surface intent and anchored by the
|
// (default 30, capped at 365), grouped by surface intent and anchored by the
|
||||||
@@ -214,7 +254,7 @@ func pickKindFamily(parent recFamily, kind string) recFamily {
|
|||||||
// Breakdown rows. Attached only when at least one attributed play
|
// Breakdown rows. Attached only when at least one attributed play
|
||||||
// exists — an all-unattributed breakdown would just repeat the parent
|
// exists — an all-unattributed breakdown would just repeat the parent
|
||||||
// row, and families that never stamp (radio, direct plays) stay flat.
|
// row, and families that never stamp (radio, direct plays) stay flat.
|
||||||
func pickKindBreakdown(picks map[string]*familyAccum) []surfaceMetric {
|
func pickKindBreakdown(picks map[string]*familyAccum, baseline *familyAccum) []surfaceMetric {
|
||||||
attributed := int64(0)
|
attributed := int64(0)
|
||||||
for kind, acc := range picks {
|
for kind, acc := range picks {
|
||||||
if kind != "" {
|
if kind != "" {
|
||||||
@@ -227,7 +267,9 @@ func pickKindBreakdown(picks map[string]*familyAccum) []surfaceMetric {
|
|||||||
out := make([]surfaceMetric, 0, len(picks))
|
out := make([]surfaceMetric, 0, len(picks))
|
||||||
for _, kind := range pickKindOrder {
|
for _, kind := range pickKindOrder {
|
||||||
if acc, ok := picks[kind]; ok && acc.plays > 0 {
|
if acc, ok := picks[kind]; ok && acc.plays > 0 {
|
||||||
out = append(out, acc.metric())
|
m := acc.metric()
|
||||||
|
applyDeltas(&m, acc, baseline)
|
||||||
|
out = append(out, m)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return out
|
return out
|
||||||
@@ -287,7 +329,8 @@ func bucketMetricsResponse(
|
|||||||
for _, acc := range families {
|
for _, acc := range families {
|
||||||
if acc.fam.intent == g.intent {
|
if acc.fam.intent == g.intent {
|
||||||
m := acc.metric()
|
m := acc.metric()
|
||||||
m.Breakdown = pickKindBreakdown(picks[acc.fam.key])
|
applyDeltas(&m, acc, baseline)
|
||||||
|
m.Breakdown = pickKindBreakdown(picks[acc.fam.key], baseline)
|
||||||
group.Surfaces = append(group.Surfaces, m)
|
group.Surfaces = append(group.Surfaces, m)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,117 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import "math"
|
||||||
|
|
||||||
|
// Uncertainty on the deltas the recommendation-metrics card shows (#2495).
|
||||||
|
//
|
||||||
|
// Why this exists: the card had exactly one volume threshold,
|
||||||
|
// recMetricsLowVolume = 20, and it was doing two jobs. Twenty plays is enough to
|
||||||
|
// be worth DISPLAYING — below that a skip rate is anecdote — but it is nowhere
|
||||||
|
// near enough to ACT on. Detecting the ~13pp differences that actually matter
|
||||||
|
// needs roughly 133 plays per arm for 80% power at α=0.05.
|
||||||
|
//
|
||||||
|
// So Discover's taste-matched (59 plays) and random-unheard (70) both rendered as
|
||||||
|
// full-confidence rows with a bold delta beside them, and that comparison sits at
|
||||||
|
// p ≈ 0.06. The card said "signal"; the arithmetic said "maybe". It led directly
|
||||||
|
// to a recommendation the data didn't support, and any reader with the same
|
||||||
|
// numbers would have made the same call.
|
||||||
|
//
|
||||||
|
// The fix is to publish the margin of error next to the delta and flag when the
|
||||||
|
// delta is smaller than it — i.e. not distinguishable from zero. Computed here,
|
||||||
|
// server-side, so both clients agree rather than each re-deriving it.
|
||||||
|
//
|
||||||
|
// recMetricsLowVolume stays exactly as it was. This is a second, independent
|
||||||
|
// signal, not a replacement: "too thin to show" and "too thin to act on" are
|
||||||
|
// different questions and deserve different answers.
|
||||||
|
|
||||||
|
// deltaZ is the two-sided 95% normal critical value. Normal rather than
|
||||||
|
// Student's t: at the sample sizes where a delta is worth acting on (n in the
|
||||||
|
// hundreds) the difference is immaterial, and a household dashboard does not
|
||||||
|
// need a t-table.
|
||||||
|
const deltaZ = 1.96
|
||||||
|
|
||||||
|
// metricDelta is a difference from the baseline, with its uncertainty.
|
||||||
|
//
|
||||||
|
// Both figures are in PERCENTAGE POINTS, matching how the card reads them out —
|
||||||
|
// a skip rate of 0.153 against a baseline of 0.270 is "-11.7", not "-0.117".
|
||||||
|
type metricDelta struct {
|
||||||
|
// DeltaPP is surface minus baseline. Negative skip is better; negative
|
||||||
|
// completion is worse. The client owns that colouring.
|
||||||
|
DeltaPP float64 `json:"delta_pp"`
|
||||||
|
// MarginPP is the 95% margin of error on DeltaPP. Read the delta as
|
||||||
|
// DeltaPP ± MarginPP.
|
||||||
|
MarginPP float64 `json:"margin_pp"`
|
||||||
|
// Distinguishable reports |DeltaPP| >= MarginPP: the interval excludes
|
||||||
|
// zero, so the difference is worth reading as a difference. When false the
|
||||||
|
// number may be pure noise no matter how large it looks.
|
||||||
|
Distinguishable bool `json:"distinguishable"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// proportionDelta compares two rates (skips/plays) as a two-proportion
|
||||||
|
// difference. Returns nil when either sample is empty, or when either rate is
|
||||||
|
// degenerate (0 or 1) — a rate with no observed variation has an SE of 0 on its
|
||||||
|
// side, which would report a spuriously narrow margin rather than an honest one.
|
||||||
|
func proportionDelta(rate1 float64, n1 int64, rate2 float64, n2 int64) *metricDelta {
|
||||||
|
if n1 <= 0 || n2 <= 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
v1 := rate1 * (1 - rate1) / float64(n1)
|
||||||
|
v2 := rate2 * (1 - rate2) / float64(n2)
|
||||||
|
se := math.Sqrt(v1 + v2)
|
||||||
|
if se <= 0 {
|
||||||
|
// Both rates are 0 or both are 1. The delta is exactly zero and the
|
||||||
|
// margin is meaningless; reporting nothing is more honest than
|
||||||
|
// reporting certainty.
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return newDelta((rate1-rate2)*100, deltaZ*se*100)
|
||||||
|
}
|
||||||
|
|
||||||
|
// meanDelta compares two means (average completion ratio) using Welch's
|
||||||
|
// standard error, which does not assume equal variances between the two groups.
|
||||||
|
//
|
||||||
|
// Note the margins here are wider than intuition suggests, and that is correct:
|
||||||
|
// completion is strongly bimodal — a play is either abandoned early (≈0.05) or
|
||||||
|
// finished (≈1.0), with little in between — so its standard deviation is large
|
||||||
|
// (~0.4) even though the mean looks stable.
|
||||||
|
func meanDelta(mean1 float64, variance1 float64, n1 int64, mean2 float64, variance2 float64, n2 int64) *metricDelta {
|
||||||
|
// Two observations minimum per side: a sample variance needs n-1 > 0.
|
||||||
|
if n1 < 2 || n2 < 2 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
se := math.Sqrt(variance1/float64(n1) + variance2/float64(n2))
|
||||||
|
if se <= 0 || math.IsNaN(se) || math.IsInf(se, 0) {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return newDelta((mean1-mean2)*100, deltaZ*se*100)
|
||||||
|
}
|
||||||
|
|
||||||
|
func newDelta(deltaPP, marginPP float64) *metricDelta {
|
||||||
|
return &metricDelta{
|
||||||
|
DeltaPP: deltaPP,
|
||||||
|
MarginPP: marginPP,
|
||||||
|
// >= rather than >: a delta exactly equal to its margin sits on the
|
||||||
|
// boundary, and calling the boundary "distinguishable" is the
|
||||||
|
// conventional reading of a 95% interval that just excludes zero.
|
||||||
|
Distinguishable: math.Abs(deltaPP) >= marginPP,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// sampleVariance recovers the sample variance from the aggregates the SQL
|
||||||
|
// returns. sum is mean×n rather than a selected column, which keeps the query to
|
||||||
|
// one extra expression.
|
||||||
|
//
|
||||||
|
// The subtraction can go very slightly negative through floating-point
|
||||||
|
// cancellation when every observation is identical, so the result is clamped —
|
||||||
|
// a negative variance would produce NaN downstream.
|
||||||
|
func sampleVariance(sum, sqSum float64, n int64) float64 {
|
||||||
|
if n < 2 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
nf := float64(n)
|
||||||
|
v := (sqSum - (sum * sum / nf)) / (nf - 1)
|
||||||
|
if v < 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
return v
|
||||||
|
}
|
||||||
@@ -0,0 +1,152 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"math"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestProportionDelta_ReproducesTheDiscoverCase(t *testing.T) {
|
||||||
|
// The comparison that motivated #2495: Discover taste-matched (59 plays,
|
||||||
|
// 15.3% skip) vs random-unheard (70 plays, 28.6%). A 13.3pp gap that the old
|
||||||
|
// card rendered as a confident coloured number, sitting at p ≈ 0.06.
|
||||||
|
d := proportionDelta(0.153, 59, 0.286, 70)
|
||||||
|
if d == nil {
|
||||||
|
t.Fatal("expected a delta for two real samples")
|
||||||
|
}
|
||||||
|
if math.Abs(d.DeltaPP-(-13.3)) > 0.1 {
|
||||||
|
t.Errorf("DeltaPP = %.2f, want ≈ -13.3", d.DeltaPP)
|
||||||
|
}
|
||||||
|
// This is the assertion the whole task exists for: at these sample sizes the
|
||||||
|
// margin swallows the difference.
|
||||||
|
if d.Distinguishable {
|
||||||
|
t.Errorf("13.3pp on n=59/70 reported as distinguishable (margin %.2f) — "+
|
||||||
|
"this is exactly the false confidence #2495 set out to remove", d.MarginPP)
|
||||||
|
}
|
||||||
|
if d.MarginPP <= 13.3 {
|
||||||
|
t.Errorf("MarginPP = %.2f, expected it to exceed the 13.3pp delta", d.MarginPP)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Same effect size, ~10x the volume: now it is real. Proves the flag tracks
|
||||||
|
// sample size rather than just the size of the gap.
|
||||||
|
func TestProportionDelta_SameGapBecomesDistinguishableWithVolume(t *testing.T) {
|
||||||
|
d := proportionDelta(0.153, 600, 0.286, 700)
|
||||||
|
if d == nil {
|
||||||
|
t.Fatal("expected a delta")
|
||||||
|
}
|
||||||
|
if !d.Distinguishable {
|
||||||
|
t.Errorf("13.3pp on n=600/700 should be distinguishable (margin %.2f)", d.MarginPP)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProportionDelta_SignAndDirection(t *testing.T) {
|
||||||
|
// Surface skips MORE than baseline -> positive delta (worse for skip rate).
|
||||||
|
worse := proportionDelta(0.40, 500, 0.25, 500)
|
||||||
|
if worse == nil || worse.DeltaPP <= 0 {
|
||||||
|
t.Fatalf("expected a positive delta, got %+v", worse)
|
||||||
|
}
|
||||||
|
better := proportionDelta(0.10, 500, 0.25, 500)
|
||||||
|
if better == nil || better.DeltaPP >= 0 {
|
||||||
|
t.Fatalf("expected a negative delta, got %+v", better)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestProportionDelta_EmptySamples(t *testing.T) {
|
||||||
|
if d := proportionDelta(0.2, 0, 0.3, 100); d != nil {
|
||||||
|
t.Errorf("n1=0 produced a delta: %+v", d)
|
||||||
|
}
|
||||||
|
if d := proportionDelta(0.2, 100, 0.3, 0); d != nil {
|
||||||
|
t.Errorf("n2=0 produced a delta: %+v", d)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Two degenerate rates have zero standard error, which would report a margin of
|
||||||
|
// 0 and therefore "distinguishable" for a delta of exactly 0. Reporting nothing
|
||||||
|
// is the honest answer.
|
||||||
|
func TestProportionDelta_DegenerateRates(t *testing.T) {
|
||||||
|
if d := proportionDelta(0, 50, 0, 50); d != nil {
|
||||||
|
t.Errorf("both rates 0 produced a delta: %+v", d)
|
||||||
|
}
|
||||||
|
if d := proportionDelta(1, 50, 1, 50); d != nil {
|
||||||
|
t.Errorf("both rates 1 produced a delta: %+v", d)
|
||||||
|
}
|
||||||
|
// One degenerate side is still informative — the other side carries variance.
|
||||||
|
if d := proportionDelta(0, 200, 0.3, 200); d == nil {
|
||||||
|
t.Error("one degenerate rate should still yield a delta")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMeanDelta(t *testing.T) {
|
||||||
|
// Completion is bimodal, so ~0.16 variance (sd ≈ 0.4) is realistic.
|
||||||
|
const v = 0.16
|
||||||
|
thin := meanDelta(0.82, v, 59, 0.54, v, 70)
|
||||||
|
if thin == nil {
|
||||||
|
t.Fatal("expected a delta")
|
||||||
|
}
|
||||||
|
if math.Abs(thin.DeltaPP-28.0) > 0.1 {
|
||||||
|
t.Errorf("DeltaPP = %.2f, want ≈ 28.0", thin.DeltaPP)
|
||||||
|
}
|
||||||
|
// 28pp is large enough to survive even a wide margin at this n.
|
||||||
|
if !thin.Distinguishable {
|
||||||
|
t.Errorf("28pp on n=59/70 with sd 0.4 should be distinguishable (margin %.2f)", thin.MarginPP)
|
||||||
|
}
|
||||||
|
// A small completion gap at the same volume should not be.
|
||||||
|
small := meanDelta(0.56, v, 59, 0.54, v, 70)
|
||||||
|
if small == nil {
|
||||||
|
t.Fatal("expected a delta")
|
||||||
|
}
|
||||||
|
if small.Distinguishable {
|
||||||
|
t.Errorf("2pp on n=59/70 reported as distinguishable (margin %.2f)", small.MarginPP)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A sample variance needs at least two observations per side.
|
||||||
|
func TestMeanDelta_NeedsTwoObservations(t *testing.T) {
|
||||||
|
if d := meanDelta(0.8, 0.1, 1, 0.5, 0.1, 100); d != nil {
|
||||||
|
t.Errorf("n1=1 produced a delta: %+v", d)
|
||||||
|
}
|
||||||
|
if d := meanDelta(0.8, 0.1, 100, 0.5, 0.1, 1); d != nil {
|
||||||
|
t.Errorf("n2=1 produced a delta: %+v", d)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMeanDelta_ZeroVarianceBothSides(t *testing.T) {
|
||||||
|
if d := meanDelta(0.8, 0, 50, 0.5, 0, 50); d != nil {
|
||||||
|
t.Errorf("zero variance on both sides produced a delta: %+v", d)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSampleVariance(t *testing.T) {
|
||||||
|
// Observations 0, 1: mean 0.5, sample variance 0.5.
|
||||||
|
if got := sampleVariance(1.0, 1.0, 2); math.Abs(got-0.5) > 1e-9 {
|
||||||
|
t.Errorf("sampleVariance = %v, want 0.5", got)
|
||||||
|
}
|
||||||
|
// Identical observations -> zero variance, and must not go negative through
|
||||||
|
// floating-point cancellation.
|
||||||
|
if got := sampleVariance(4.0, 4.0, 4); got != 0 {
|
||||||
|
t.Errorf("identical observations gave variance %v, want 0", got)
|
||||||
|
}
|
||||||
|
if got := sampleVariance(0, 0, 1); got != 0 {
|
||||||
|
t.Errorf("n=1 gave variance %v, want 0", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clamping matters: a negative variance would become NaN in the square root and
|
||||||
|
// propagate into the JSON as a null-ish number.
|
||||||
|
func TestSampleVariance_NeverNegative(t *testing.T) {
|
||||||
|
// sqSum slightly below sum²/n, as cancellation can produce.
|
||||||
|
if got := sampleVariance(10.0, 24.999999999, 4); got < 0 {
|
||||||
|
t.Errorf("variance went negative: %v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNewDelta_BoundaryCountsAsDistinguishable(t *testing.T) {
|
||||||
|
d := newDelta(5.0, 5.0)
|
||||||
|
if !d.Distinguishable {
|
||||||
|
t.Error("a delta exactly equal to its margin should count as distinguishable")
|
||||||
|
}
|
||||||
|
d = newDelta(4.999, 5.0)
|
||||||
|
if d.Distinguishable {
|
||||||
|
t.Error("a delta just inside its margin should not count as distinguishable")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -18,7 +18,8 @@ SELECT
|
|||||||
count(*)::bigint AS plays,
|
count(*)::bigint AS plays,
|
||||||
count(*) FILTER (WHERE pe.was_skipped)::bigint AS skips,
|
count(*) FILTER (WHERE pe.was_skipped)::bigint AS skips,
|
||||||
count(pe.completion_ratio)::bigint AS completion_n,
|
count(pe.completion_ratio)::bigint AS completion_n,
|
||||||
COALESCE(avg(pe.completion_ratio), 0)::float8 AS avg_completion
|
COALESCE(avg(pe.completion_ratio), 0)::float8 AS avg_completion,
|
||||||
|
COALESCE(sum(pe.completion_ratio * pe.completion_ratio), 0)::float8 AS completion_sqsum
|
||||||
FROM play_events pe
|
FROM play_events pe
|
||||||
WHERE pe.user_id = $1
|
WHERE pe.user_id = $1
|
||||||
AND pe.started_at > now() - ($2::float8 * INTERVAL '1 day')
|
AND pe.started_at > now() - ($2::float8 * INTERVAL '1 day')
|
||||||
@@ -32,18 +33,26 @@ type RecommendationSourceMetricsForUserParams struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
type RecommendationSourceMetricsForUserRow struct {
|
type RecommendationSourceMetricsForUserRow struct {
|
||||||
Source *string
|
Source *string
|
||||||
PickKind *string
|
PickKind *string
|
||||||
Plays int64
|
Plays int64
|
||||||
Skips int64
|
Skips int64
|
||||||
CompletionN int64
|
CompletionN int64
|
||||||
AvgCompletion float64
|
AvgCompletion float64
|
||||||
|
CompletionSqsum float64
|
||||||
}
|
}
|
||||||
|
|
||||||
// $1 user_id, $2 window_days. plays/skips are counts; avg_completion is the
|
// $1 user_id, $2 window_days. plays/skips are counts; avg_completion is the
|
||||||
// mean completion ratio over the completion_n plays that recorded one.
|
// mean completion ratio over the completion_n plays that recorded one.
|
||||||
// pick_kind splits For You plays into taste/fresh/unattributed (#1249);
|
// pick_kind splits For You plays into taste/fresh/unattributed (#1249);
|
||||||
// it is NULL for every other source, so those still group to one row.
|
// it is NULL for every other source, so those still group to one row.
|
||||||
|
//
|
||||||
|
// completion_sqsum carries the sum of SQUARED completion ratios so the Go
|
||||||
|
// handler can compute a variance — needed for the margin of error on a
|
||||||
|
// completion delta (#2495). It is the sum rather than `stddev_samp` on purpose:
|
||||||
|
// raw source rows get merged into surface families in Go, and sums of squares
|
||||||
|
// add across groups exactly, whereas standard deviations cannot be combined
|
||||||
|
// without them. Variance = (sqsum - sum²/n) / (n-1), with sum = avg × n.
|
||||||
func (q *Queries) RecommendationSourceMetricsForUser(ctx context.Context, arg RecommendationSourceMetricsForUserParams) ([]RecommendationSourceMetricsForUserRow, error) {
|
func (q *Queries) RecommendationSourceMetricsForUser(ctx context.Context, arg RecommendationSourceMetricsForUserParams) ([]RecommendationSourceMetricsForUserRow, error) {
|
||||||
rows, err := q.db.Query(ctx, recommendationSourceMetricsForUser, arg.UserID, arg.Column2)
|
rows, err := q.db.Query(ctx, recommendationSourceMetricsForUser, arg.UserID, arg.Column2)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -60,6 +69,7 @@ func (q *Queries) RecommendationSourceMetricsForUser(ctx context.Context, arg Re
|
|||||||
&i.Skips,
|
&i.Skips,
|
||||||
&i.CompletionN,
|
&i.CompletionN,
|
||||||
&i.AvgCompletion,
|
&i.AvgCompletion,
|
||||||
|
&i.CompletionSqsum,
|
||||||
); err != nil {
|
); err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -44,13 +44,21 @@ ORDER BY 1, 2;
|
|||||||
-- mean completion ratio over the completion_n plays that recorded one.
|
-- mean completion ratio over the completion_n plays that recorded one.
|
||||||
-- pick_kind splits For You plays into taste/fresh/unattributed (#1249);
|
-- pick_kind splits For You plays into taste/fresh/unattributed (#1249);
|
||||||
-- it is NULL for every other source, so those still group to one row.
|
-- it is NULL for every other source, so those still group to one row.
|
||||||
|
--
|
||||||
|
-- completion_sqsum carries the sum of SQUARED completion ratios so the Go
|
||||||
|
-- handler can compute a variance — needed for the margin of error on a
|
||||||
|
-- completion delta (#2495). It is the sum rather than `stddev_samp` on purpose:
|
||||||
|
-- raw source rows get merged into surface families in Go, and sums of squares
|
||||||
|
-- add across groups exactly, whereas standard deviations cannot be combined
|
||||||
|
-- without them. Variance = (sqsum - sum²/n) / (n-1), with sum = avg × n.
|
||||||
SELECT
|
SELECT
|
||||||
pe.source,
|
pe.source,
|
||||||
pe.pick_kind,
|
pe.pick_kind,
|
||||||
count(*)::bigint AS plays,
|
count(*)::bigint AS plays,
|
||||||
count(*) FILTER (WHERE pe.was_skipped)::bigint AS skips,
|
count(*) FILTER (WHERE pe.was_skipped)::bigint AS skips,
|
||||||
count(pe.completion_ratio)::bigint AS completion_n,
|
count(pe.completion_ratio)::bigint AS completion_n,
|
||||||
COALESCE(avg(pe.completion_ratio), 0)::float8 AS avg_completion
|
COALESCE(avg(pe.completion_ratio), 0)::float8 AS avg_completion,
|
||||||
|
COALESCE(sum(pe.completion_ratio * pe.completion_ratio), 0)::float8 AS completion_sqsum
|
||||||
FROM play_events pe
|
FROM play_events pe
|
||||||
WHERE pe.user_id = $1
|
WHERE pe.user_id = $1
|
||||||
AND pe.started_at > now() - ($2::float8 * INTERVAL '1 day')
|
AND pe.started_at > now() - ($2::float8 * INTERVAL '1 day')
|
||||||
|
|||||||
@@ -414,8 +414,21 @@ func (s *Scanner) resolveArtist(ctx context.Context, q *dbq.Queries, name, mbid
|
|||||||
ID: existing.ID,
|
ID: existing.ID,
|
||||||
Mbid: &m,
|
Mbid: &m,
|
||||||
}); uerr != nil {
|
}); uerr != nil {
|
||||||
s.logger.Warn("library scan: heal artist mbid failed",
|
if isUniqueViolation(uerr) {
|
||||||
"artist_id", existing.ID, "err", uerr)
|
// Another artist row already owns this MBID — two rows that
|
||||||
|
// should be merged (usually two spellings of one name).
|
||||||
|
// Expected, not a fault: leave NULL and let the operator
|
||||||
|
// merge. Mirrors resolveAlbum, which has always handled it
|
||||||
|
// this way — without this branch the identical benign
|
||||||
|
// condition logged a generic warning plus a Postgres ERROR
|
||||||
|
// line on every scan, which teaches an operator to ignore
|
||||||
|
// database errors (#2524).
|
||||||
|
s.logger.Info("library scan: duplicate artist mbid (canonical row already owns it)",
|
||||||
|
"artist_id", existing.ID, "artist", name, "mbid", mbid)
|
||||||
|
} else {
|
||||||
|
s.logger.Warn("library scan: heal artist mbid failed",
|
||||||
|
"artist_id", existing.ID, "err", uerr)
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
existing.Mbid = &m
|
existing.Mbid = &m
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,6 +4,20 @@ import { api } from './client';
|
|||||||
// Mirrors internal/api/me_recommendation_metrics.go: raw play sources are
|
// Mirrors internal/api/me_recommendation_metrics.go: raw play sources are
|
||||||
// bucketed server-side into stable surface families, grouped by intent, and
|
// bucketed server-side into stable surface families, grouped by intent, and
|
||||||
// anchored by the manual-plays baseline (milestone #127).
|
// anchored by the manual-plays baseline (milestone #127).
|
||||||
|
// A difference from the baseline, with its uncertainty (#2495). Both figures
|
||||||
|
// are already in percentage points — the server does the arithmetic so both
|
||||||
|
// clients read the same numbers.
|
||||||
|
//
|
||||||
|
// `distinguishable: false` means |delta_pp| < margin_pp: the delta cannot be
|
||||||
|
// told apart from zero, however large it looks. That distinction is the whole
|
||||||
|
// point of this type — `low_confidence` answers "is this worth showing?", which
|
||||||
|
// is a much lower bar than "is this worth acting on?".
|
||||||
|
export type MetricDelta = {
|
||||||
|
delta_pp: number;
|
||||||
|
margin_pp: number;
|
||||||
|
distinguishable: boolean;
|
||||||
|
};
|
||||||
|
|
||||||
export type SurfaceMetric = {
|
export type SurfaceMetric = {
|
||||||
key: string;
|
key: string;
|
||||||
label: string;
|
label: string;
|
||||||
@@ -12,6 +26,10 @@ export type SurfaceMetric = {
|
|||||||
skip_rate: number;
|
skip_rate: number;
|
||||||
avg_completion: number;
|
avg_completion: number;
|
||||||
low_confidence: boolean;
|
low_confidence: boolean;
|
||||||
|
// Absent on the baseline row itself, and whenever the samples are too thin
|
||||||
|
// for a margin to mean anything.
|
||||||
|
skip_delta?: MetricDelta;
|
||||||
|
completion_delta?: MetricDelta;
|
||||||
// Present when the surface's builder stamps pick-kind provenance and
|
// Present when the surface's builder stamps pick-kind provenance and
|
||||||
// the window holds attributed plays (#1249, generalized #1270): For
|
// the window holds attributed plays (#1249, generalized #1270): For
|
||||||
// You's taste/fresh split, Discover's candidate buckets, the tiered
|
// You's taste/fresh split, Discover's candidate buckets, the tiered
|
||||||
|
|||||||
@@ -216,9 +216,17 @@
|
|||||||
return plays > 0 ? hits / plays : 0;
|
return plays > 0 ? hits / plays : 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
function latest(s: TrendSeries): { skip: number; completion: number } {
|
// The "latest" columns are ONE WEEK while the Plays column is the whole
|
||||||
|
// window, which is a trap: a 40% skip rate off 17 plays sat next to a
|
||||||
|
// four-figure Plays total and read as a solid signal. It isn't — I misread
|
||||||
|
// exactly this and briefly concluded Deep cuts was the worst surface, when
|
||||||
|
// over 180 days it's one of the best (#2495). So the week's own play count
|
||||||
|
// comes back with the rates and is rendered beside them.
|
||||||
|
function latest(s: TrendSeries): { skip: number; completion: number; plays: number } {
|
||||||
const last = s.points[s.points.length - 1];
|
const last = s.points[s.points.length - 1];
|
||||||
return last ? { skip: last.skip_rate, completion: last.avg_completion } : { skip: 0, completion: 0 };
|
return last
|
||||||
|
? { skip: last.skip_rate, completion: last.avg_completion, plays: last.plays }
|
||||||
|
: { skip: 0, completion: 0, plays: 0 };
|
||||||
}
|
}
|
||||||
|
|
||||||
function pct(v: number): string {
|
function pct(v: number): string {
|
||||||
@@ -420,6 +428,10 @@
|
|||||||
<div>
|
<div>
|
||||||
<h2 class="text-lg font-semibold">Weekly trends</h2>
|
<h2 class="text-lg font-semibold">Weekly trends</h2>
|
||||||
<p class="text-xs text-text-secondary">
|
<p class="text-xs text-text-secondary">
|
||||||
|
<span class="font-medium">Skip and completion are the most recent week only</span> —
|
||||||
|
the number after the skip rate is that week's play count, so a rate off a
|
||||||
|
handful of plays reads as what it is. Plays and Taste hit cover the whole
|
||||||
|
window.
|
||||||
Skip rate per surface over the last {trends?.weeks ?? 12} weeks (lower is better; all
|
Skip rate per surface over the last {trends?.weeks ?? 12} weeks (lower is better; all
|
||||||
users aggregated, rates only). Dashed ticks mark tuning changes. Taste hit is the share
|
users aggregated, rates only). Dashed ticks mark tuning changes. Taste hit is the share
|
||||||
of plays whose artist fits the current taste profile.
|
of plays whose artist fits the current taste profile.
|
||||||
@@ -439,10 +451,10 @@
|
|||||||
<tr class="text-left text-text-secondary">
|
<tr class="text-left text-text-secondary">
|
||||||
<th class="py-1 font-medium">Surface</th>
|
<th class="py-1 font-medium">Surface</th>
|
||||||
<th class="py-1 font-medium">Skip rate by week</th>
|
<th class="py-1 font-medium">Skip rate by week</th>
|
||||||
<th class="py-1 text-right font-medium">Plays</th>
|
<th class="py-1 text-right font-medium">Plays<span class="font-normal text-xs"> (window)</span></th>
|
||||||
<th class="py-1 text-right font-medium">Latest skip</th>
|
<th class="py-1 text-right font-medium">Skip<span class="font-normal text-xs"> (last wk)</span></th>
|
||||||
<th class="py-1 text-right font-medium">Latest completion</th>
|
<th class="py-1 text-right font-medium">Completion<span class="font-normal text-xs"> (last wk)</span></th>
|
||||||
<th class="py-1 text-right font-medium">Taste hit</th>
|
<th class="py-1 text-right font-medium">Taste hit<span class="font-normal text-xs"> (window)</span></th>
|
||||||
</tr>
|
</tr>
|
||||||
</thead>
|
</thead>
|
||||||
<tbody>
|
<tbody>
|
||||||
@@ -493,7 +505,10 @@
|
|||||||
</svg>
|
</svg>
|
||||||
</td>
|
</td>
|
||||||
<td class="py-1.5 text-right tabular-nums">{s.plays}</td>
|
<td class="py-1.5 text-right tabular-nums">{s.plays}</td>
|
||||||
<td class="py-1.5 text-right tabular-nums">{pct(latest(s).skip)}</td>
|
<td class="py-1.5 text-right tabular-nums">
|
||||||
|
{pct(latest(s).skip)}
|
||||||
|
<span class="text-xs text-text-secondary">/{latest(s).plays}</span>
|
||||||
|
</td>
|
||||||
<td class="py-1.5 text-right tabular-nums">{pct(latest(s).completion)}</td>
|
<td class="py-1.5 text-right tabular-nums">{pct(latest(s).completion)}</td>
|
||||||
<td class="py-1.5 text-right tabular-nums">{pct(windowTasteHitRate(s))}</td>
|
<td class="py-1.5 text-right tabular-nums">{pct(windowTasteHitRate(s))}</td>
|
||||||
</tr>
|
</tr>
|
||||||
|
|||||||
@@ -195,9 +195,20 @@ describe('Admin tuning page', () => {
|
|||||||
await waitFor(() => expect(screen.getByText('Weekly trends')).toBeInTheDocument());
|
await waitFor(() => expect(screen.getByText('Weekly trends')).toBeInTheDocument());
|
||||||
expect(screen.getByTestId('sparkline-radio')).toBeInTheDocument();
|
expect(screen.getByTestId('sparkline-radio')).toBeInTheDocument();
|
||||||
expect(screen.getByTestId('sparkline-discover')).toBeInTheDocument();
|
expect(screen.getByTestId('sparkline-discover')).toBeInTheDocument();
|
||||||
// Latest skip rate column for radio = 40% (also discover's latest
|
// The skip column is the LAST WEEK's rate, now carrying that week's play
|
||||||
// completion, hence getAllBy).
|
// count so a rate off a handful of plays reads as what it is (#2495).
|
||||||
expect(screen.getAllByText('40%').length).toBeGreaterThan(0);
|
// Radio's latest week: 40% skip over 15 plays; Discover's: 60% over 5.
|
||||||
|
expect(screen.getByText('/15')).toBeInTheDocument();
|
||||||
|
expect(screen.getByText('/5')).toBeInTheDocument();
|
||||||
|
// Completion columns are unchanged and still bare percentages — radio 70%,
|
||||||
|
// discover 40%.
|
||||||
|
expect(screen.getByText('70%')).toBeInTheDocument();
|
||||||
|
expect(screen.getByText('40%')).toBeInTheDocument();
|
||||||
|
// The window/last-week distinction has to be visible in the headers, or the
|
||||||
|
// Plays total reads as the denominator of the skip rate. That misreading is
|
||||||
|
// what #2495 was filed over.
|
||||||
|
expect(screen.getByText(/Plays/)).toHaveTextContent('(window)');
|
||||||
|
expect(screen.getByText(/^Skip/)).toHaveTextContent('(last wk)');
|
||||||
// The knob turn is listed under the chart AND tooltipped on each
|
// The knob turn is listed under the chart AND tooltipped on each
|
||||||
// sparkline's marker tick, hence getAllBy.
|
// sparkline's marker tick, hence getAllBy.
|
||||||
expect(
|
expect(
|
||||||
|
|||||||
@@ -12,7 +12,7 @@
|
|||||||
createRecommendationMetricsQuery,
|
createRecommendationMetricsQuery,
|
||||||
type RecommendationMetrics,
|
type RecommendationMetrics,
|
||||||
type SurfaceIntent,
|
type SurfaceIntent,
|
||||||
type SurfaceMetric
|
type MetricDelta
|
||||||
} from '$lib/api/metrics';
|
} from '$lib/api/metrics';
|
||||||
import { theme, setTheme, type ThemePreference } from '$lib/stores/theme.svelte';
|
import { theme, setTheme, type ThemePreference } from '$lib/stores/theme.svelte';
|
||||||
import { player, setCrossfade } from '$lib/player/store.svelte';
|
import { player, setCrossfade } from '$lib/player/store.svelte';
|
||||||
@@ -45,20 +45,39 @@
|
|||||||
return `${(v * 100).toFixed(0)}%`;
|
return `${(v * 100).toFixed(0)}%`;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Delta in percentage points vs the baseline, signed ("+12" / "−5").
|
// Deltas come from the server with their margin of error (#2495). The client
|
||||||
function deltaPts(value: number, baseline: number): string {
|
// no longer subtracts rates itself: the margin needs the sample sizes and
|
||||||
const pts = Math.round((value - baseline) * 100);
|
// variances, and having both clients re-derive it invites them to disagree.
|
||||||
return pts > 0 ? `+${pts}` : `${pts}`;
|
//
|
||||||
|
// A delta that isn't distinguishable from zero is prefixed "≈" and dimmed.
|
||||||
|
// That is the point of this whole change — the card used to render a −12 on
|
||||||
|
// 59 plays exactly as boldly as a −6 on 400, and the first of those is noise.
|
||||||
|
function deltaText(d: MetricDelta | undefined): string {
|
||||||
|
if (!d) return '';
|
||||||
|
const pts = Math.round(d.delta_pp);
|
||||||
|
const signed = pts > 0 ? `+${pts}` : `${pts}`;
|
||||||
|
return d.distinguishable ? signed : `≈${signed}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
// A surface's skip delta is "worse" when it skips more than the
|
function deltaTitle(d: MetricDelta | undefined): string | undefined {
|
||||||
// baseline; completion delta is "worse" when it completes less.
|
if (!d) return undefined;
|
||||||
function skipDeltaClass(m: SurfaceMetric, baseline: SurfaceMetric): string {
|
const range = `${d.delta_pp.toFixed(1)} ± ${d.margin_pp.toFixed(1)} points vs baseline`;
|
||||||
return m.skip_rate > baseline.skip_rate ? 'text-danger' : 'text-text-secondary';
|
return d.distinguishable
|
||||||
|
? `${range} (95% confidence)`
|
||||||
|
: `${range} — not distinguishable from zero at 95% confidence, so read this as no measured difference.`;
|
||||||
}
|
}
|
||||||
|
|
||||||
function completionDeltaClass(m: SurfaceMetric, baseline: SurfaceMetric): string {
|
// A skip delta is "worse" above the baseline; a completion delta is "worse"
|
||||||
return m.avg_completion < baseline.avg_completion ? 'text-danger' : 'text-text-secondary';
|
// below it. Neither gets a colour unless it's distinguishable — colouring
|
||||||
|
// noise red is what made the old card misleading.
|
||||||
|
function skipDeltaClass(d: MetricDelta | undefined): string {
|
||||||
|
if (!d?.distinguishable) return 'text-text-secondary opacity-60';
|
||||||
|
return d.delta_pp > 0 ? 'text-danger' : 'text-text-secondary';
|
||||||
|
}
|
||||||
|
|
||||||
|
function completionDeltaClass(d: MetricDelta | undefined): string {
|
||||||
|
if (!d?.distinguishable) return 'text-text-secondary opacity-60';
|
||||||
|
return d.delta_pp < 0 ? 'text-danger' : 'text-text-secondary';
|
||||||
}
|
}
|
||||||
|
|
||||||
// Pick-kind breakdowns are collapsed by default (#1270): with every
|
// Pick-kind breakdowns are collapsed by default (#1270): with every
|
||||||
@@ -384,18 +403,16 @@
|
|||||||
<td class="py-1 text-right tabular-nums">{m.plays}</td>
|
<td class="py-1 text-right tabular-nums">{m.plays}</td>
|
||||||
<td class="py-1 text-right tabular-nums">
|
<td class="py-1 text-right tabular-nums">
|
||||||
{pct(m.skip_rate)}
|
{pct(m.skip_rate)}
|
||||||
{#if baseline}
|
{#if m.skip_delta}
|
||||||
<span class="ml-1 text-xs {skipDeltaClass(m, baseline)}">
|
<span class="ml-1 text-xs {skipDeltaClass(m.skip_delta)}"
|
||||||
{deltaPts(m.skip_rate, baseline.skip_rate)}
|
title={deltaTitle(m.skip_delta)}>{deltaText(m.skip_delta)}</span>
|
||||||
</span>
|
|
||||||
{/if}
|
{/if}
|
||||||
</td>
|
</td>
|
||||||
<td class="py-1 text-right tabular-nums">
|
<td class="py-1 text-right tabular-nums">
|
||||||
{pct(m.avg_completion)}
|
{pct(m.avg_completion)}
|
||||||
{#if baseline}
|
{#if m.completion_delta}
|
||||||
<span class="ml-1 text-xs {completionDeltaClass(m, baseline)}">
|
<span class="ml-1 text-xs {completionDeltaClass(m.completion_delta)}"
|
||||||
{deltaPts(m.avg_completion, baseline.avg_completion)}
|
title={deltaTitle(m.completion_delta)}>{deltaText(m.completion_delta)}</span>
|
||||||
</span>
|
|
||||||
{/if}
|
{/if}
|
||||||
</td>
|
</td>
|
||||||
</tr>
|
</tr>
|
||||||
@@ -420,18 +437,16 @@
|
|||||||
<td class="py-1 text-right text-xs tabular-nums">{b.plays}</td>
|
<td class="py-1 text-right text-xs tabular-nums">{b.plays}</td>
|
||||||
<td class="py-1 text-right text-xs tabular-nums">
|
<td class="py-1 text-right text-xs tabular-nums">
|
||||||
{pct(b.skip_rate)}
|
{pct(b.skip_rate)}
|
||||||
{#if baseline}
|
{#if b.skip_delta}
|
||||||
<span class="ml-1 {skipDeltaClass(b, baseline)}">
|
<span class="ml-1 {skipDeltaClass(b.skip_delta)}"
|
||||||
{deltaPts(b.skip_rate, baseline.skip_rate)}
|
title={deltaTitle(b.skip_delta)}>{deltaText(b.skip_delta)}</span>
|
||||||
</span>
|
|
||||||
{/if}
|
{/if}
|
||||||
</td>
|
</td>
|
||||||
<td class="py-1 text-right text-xs tabular-nums">
|
<td class="py-1 text-right text-xs tabular-nums">
|
||||||
{pct(b.avg_completion)}
|
{pct(b.avg_completion)}
|
||||||
{#if baseline}
|
{#if b.completion_delta}
|
||||||
<span class="ml-1 {completionDeltaClass(b, baseline)}">
|
<span class="ml-1 {completionDeltaClass(b.completion_delta)}"
|
||||||
{deltaPts(b.avg_completion, baseline.avg_completion)}
|
title={deltaTitle(b.completion_delta)}>{deltaText(b.completion_delta)}</span>
|
||||||
</span>
|
|
||||||
{/if}
|
{/if}
|
||||||
</td>
|
</td>
|
||||||
</tr>
|
</tr>
|
||||||
@@ -442,6 +457,12 @@
|
|||||||
</table>
|
</table>
|
||||||
</div>
|
</div>
|
||||||
{/each}
|
{/each}
|
||||||
|
<p class="text-xs text-text-secondary">
|
||||||
|
Deltas compare each surface with your manual plays. A delta marked
|
||||||
|
<span class="opacity-60">≈</span> is smaller than its own margin of error at this
|
||||||
|
sample size — it can't be told apart from no difference, however big it looks.
|
||||||
|
Hover any delta for its range.
|
||||||
|
</p>
|
||||||
{:else}
|
{:else}
|
||||||
<p class="text-sm text-text-secondary">
|
<p class="text-sm text-text-secondary">
|
||||||
No plays recorded yet. Play something from For You, Discover, or a mix.
|
No plays recorded yet. Play something from For You, Discover, or a mix.
|
||||||
|
|||||||
@@ -241,6 +241,71 @@ describe('Settings page — Recommendation metrics card', () => {
|
|||||||
expect(screen.queryByText(/Taste picks/)).not.toBeInTheDocument();
|
expect(screen.queryByText(/Taste picks/)).not.toBeInTheDocument();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// #2495: the card used to render a delta computed client-side with no notion
|
||||||
|
// of uncertainty, so a -12 on 59 plays looked exactly as solid as a -6 on 400.
|
||||||
|
// Deltas now arrive from the server with a margin, and an indistinguishable
|
||||||
|
// one is marked with "≈" and dimmed rather than coloured.
|
||||||
|
test('a delta smaller than its margin is marked as indistinguishable', async () => {
|
||||||
|
setupPage();
|
||||||
|
metricsMock.data = {
|
||||||
|
window_days: 30,
|
||||||
|
baseline: metric('manual', 'Manual library plays', { plays: 400, skip_rate: 0.27 }),
|
||||||
|
groups: [
|
||||||
|
{
|
||||||
|
intent: 'discovery',
|
||||||
|
label: 'Discovery mixes',
|
||||||
|
surfaces: [
|
||||||
|
metric('discover', 'Discover', {
|
||||||
|
plays: 59,
|
||||||
|
skip_rate: 0.153,
|
||||||
|
// 13.3pp gap, but the margin at n=59 is wider than the gap.
|
||||||
|
skip_delta: { delta_pp: -13.3, margin_pp: 14.5, distinguishable: false },
|
||||||
|
completion_delta: { delta_pp: 28.0, margin_pp: 12.1, distinguishable: true }
|
||||||
|
})
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
};
|
||||||
|
render(SettingsPage);
|
||||||
|
await waitFor(() => expect(screen.getByText('Discover')).toBeInTheDocument());
|
||||||
|
|
||||||
|
// The indistinguishable skip delta is prefixed and explained on hover.
|
||||||
|
const skip = screen.getByText('≈-13');
|
||||||
|
expect(skip).toBeInTheDocument();
|
||||||
|
expect(skip).toHaveAttribute('title', expect.stringContaining('not distinguishable from zero'));
|
||||||
|
// It must NOT be coloured as a real regression/improvement.
|
||||||
|
expect(skip.className).toContain('opacity-60');
|
||||||
|
|
||||||
|
// The completion delta clears its margin, so it renders plainly.
|
||||||
|
const completion = screen.getByText('+28');
|
||||||
|
expect(completion).toBeInTheDocument();
|
||||||
|
expect(completion.className).not.toContain('opacity-60');
|
||||||
|
|
||||||
|
// And the legend explains the glyph rather than leaving it a mystery.
|
||||||
|
expect(screen.getByText(/smaller than its own margin of error/i)).toBeInTheDocument();
|
||||||
|
});
|
||||||
|
|
||||||
|
// A delta is omitted entirely when the samples are too thin for a margin to
|
||||||
|
// mean anything — the server decides that, and the cell must simply show the
|
||||||
|
// rate rather than a bare "0".
|
||||||
|
test('a surface with no delta shows its rate and nothing else', async () => {
|
||||||
|
setupPage();
|
||||||
|
metricsMock.data = {
|
||||||
|
window_days: 30,
|
||||||
|
baseline: metric('manual', 'Manual library plays', { plays: 400 }),
|
||||||
|
groups: [
|
||||||
|
{
|
||||||
|
intent: 'go_to',
|
||||||
|
label: 'Go-to surfaces',
|
||||||
|
surfaces: [metric('radio', 'Radio', { plays: 1, skip_rate: 0 })]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
};
|
||||||
|
render(SettingsPage);
|
||||||
|
await waitFor(() => expect(screen.getByText('Radio')).toBeInTheDocument());
|
||||||
|
expect(screen.queryByText(/≈/)).not.toBeInTheDocument();
|
||||||
|
});
|
||||||
|
|
||||||
test('surfaces without a breakdown render no toggle and no sub-rows', async () => {
|
test('surfaces without a breakdown render no toggle and no sub-rows', async () => {
|
||||||
setupPage();
|
setupPage();
|
||||||
metricsMock.data = {
|
metricsMock.data = {
|
||||||
|
|||||||
Reference in New Issue
Block a user