diff --git a/src/proficiency_outcome.py b/src/proficiency_outcome.py index fd3a7fb..562deea 100644 --- a/src/proficiency_outcome.py +++ b/src/proficiency_outcome.py @@ -72,9 +72,10 @@ def recompute_category( For each row: 1. Re-compute benchmark via ``blend()`` on stored leaderboard / self_eval components. If neither component exists the row skips the conversion. - 2. Collect *all* rows with ``outcome_samples > 0`` and compute the - category-level *peer_rate* (mean outcome) and *peer_benchmark* (mean - benchmark_score). +2. Collect *all* rows with ``outcome_samples > 0`` and compute the + category-level *peer_rate* (sample-weighted pooled rate: + Σoutcome_samples × outcome_score / Σoutcome_samples) and + *peer_benchmark* (mean benchmark_score over the same set). 3. Call ``expected_success_rate`` per row with the benchmark, outcome, and peer statistics. Write the final ``blended_score`` and ``source``. """ @@ -127,14 +128,20 @@ def recompute_category( ] if outcome_rows: - peer_rate = ( - sum( - d["outcome_score"] - for d in outcome_rows - if d["outcome_score"] is not None + # Pooled peer_rate: Σ(outcome_samples × outcome_score) / Σ(outcome_samples) + # over all trafficked models in the category. The unweighted mean of + # per-model rates would let a 1-sample model at 100 % dominate the + # category signal — exactly the small-sample noise the pooled formula + # exists to suppress. Exclude a row from BOTH sides if its + # outcome_score is None (invariant from add_outcome, but be safe). + rates = [d for d in outcome_rows if d["outcome_score"] is not None] + if rates: + peer_rate = ( + sum(d["outcome_samples"] * d["outcome_score"] for d in rates) + / sum(d["outcome_samples"] for d in rates) ) - / len(outcome_rows) - ) + else: + peer_rate = None benchmark_scores = [ d["benchmark_score"] for d in outcome_rows diff --git a/tests/test_proficiency.py b/tests/test_proficiency.py index c5a63e4..758e975 100644 --- a/tests/test_proficiency.py +++ b/tests/test_proficiency.py @@ -867,3 +867,42 @@ def test_expected_success_rate_prior_falls_back_to_peer_rate_when_peer_benchmark ) assert source == "outcome_prior" assert score == pytest.approx(0.50) + + +def test_recompute_category_uses_pooled_peer_rate(tmp_path): + """Peer rate is sample-weighted Σ(n·rate)/Σn, not unweighted mean of rates. + + Pooled: (1×1.0 + 95×0.81) / (1+95) = 77.95/96 ≈ 0.812 + Unweighted: (1.0 + 0.81) / 2 = 0.905 + + Model A: 1 sample at 1.0. Model B: 95 samples at 0.81. + Model C: no outcomes, no benchmark — so prior = peer_rate directly. + Model C's blended_score must be ~0.812 (pooled), not ~0.905 (unweighted). + """ + from config import load_config + from proficiency_store import add_outcome, set_leaderboard + + cfg = load_config("config/config.yaml") + conn = _models_db(tmp_path, [ + ("model-a", "a", "standard", "default", "full"), + ("model-b", "b", "standard", "default", "full"), + ("model-c", "c", "standard", "default", "full"), + ]) + + # Model C gets a row with no benchmark so its prior = peer_rate directly. + set_leaderboard(conn, cfg, "model-c", "nw", "coding_general", None) + + # Model A: 1 sample at 1.0 + add_outcome(conn, cfg, "model-a", "nw", "coding_general", [1.0]) + # Model B: 95 samples at 0.81 + add_outcome(conn, cfg, "model-b", "nw", "coding_general", [0.81] * 95) + + # After the second add_outcome, recompute_category runs over all three: + # peer_rate is now pooled from A (1×1.0) + B (95×0.81). + c_row = _row(conn, "model-c") + expected_pooled = (1.0 + 95 * 0.81) / (1 + 95) # ≈ 0.812 + assert c_row["source"] == "outcome_prior" + assert c_row["blended_score"] == pytest.approx(expected_pooled, rel=1e-3) + # Confirm it's NOT the unweighted result + unweighted = (1.0 + 0.81) / 2 # ≈ 0.905 + assert c_row["blended_score"] < unweighted - 0.09