fix(proficiency): exposure-bias fix + backlog fold — empirical-Bayes scoring, pooled peer_rate, exploration live #62
@@ -72,9 +72,10 @@ def recompute_category(
|
||||
For each row:
|
||||
1. Re-compute benchmark via ``blend()`` on stored leaderboard / self_eval
|
||||
components. If neither component exists the row skips the conversion.
|
||||
2. Collect *all* rows with ``outcome_samples > 0`` and compute the
|
||||
category-level *peer_rate* (mean outcome) and *peer_benchmark* (mean
|
||||
benchmark_score).
|
||||
2. Collect *all* rows with ``outcome_samples > 0`` and compute the
|
||||
category-level *peer_rate* (sample-weighted pooled rate:
|
||||
Σoutcome_samples × outcome_score / Σoutcome_samples) and
|
||||
*peer_benchmark* (mean benchmark_score over the same set).
|
||||
3. Call ``expected_success_rate`` per row with the benchmark, outcome, and
|
||||
peer statistics. Write the final ``blended_score`` and ``source``.
|
||||
"""
|
||||
@@ -127,14 +128,20 @@ def recompute_category(
|
||||
]
|
||||
|
||||
if outcome_rows:
|
||||
peer_rate = (
|
||||
sum(
|
||||
d["outcome_score"]
|
||||
for d in outcome_rows
|
||||
if d["outcome_score"] is not None
|
||||
# Pooled peer_rate: Σ(outcome_samples × outcome_score) / Σ(outcome_samples)
|
||||
# over all trafficked models in the category. The unweighted mean of
|
||||
# per-model rates would let a 1-sample model at 100 % dominate the
|
||||
# category signal — exactly the small-sample noise the pooled formula
|
||||
# exists to suppress. Exclude a row from BOTH sides if its
|
||||
# outcome_score is None (invariant from add_outcome, but be safe).
|
||||
rates = [d for d in outcome_rows if d["outcome_score"] is not None]
|
||||
if rates:
|
||||
peer_rate = (
|
||||
sum(d["outcome_samples"] * d["outcome_score"] for d in rates)
|
||||
/ sum(d["outcome_samples"] for d in rates)
|
||||
)
|
||||
/ len(outcome_rows)
|
||||
)
|
||||
else:
|
||||
peer_rate = None
|
||||
benchmark_scores = [
|
||||
d["benchmark_score"]
|
||||
for d in outcome_rows
|
||||
|
||||
@@ -867,3 +867,42 @@ def test_expected_success_rate_prior_falls_back_to_peer_rate_when_peer_benchmark
|
||||
)
|
||||
assert source == "outcome_prior"
|
||||
assert score == pytest.approx(0.50)
|
||||
|
||||
|
||||
def test_recompute_category_uses_pooled_peer_rate(tmp_path):
|
||||
"""Peer rate is sample-weighted Σ(n·rate)/Σn, not unweighted mean of rates.
|
||||
|
||||
Pooled: (1×1.0 + 95×0.81) / (1+95) = 77.95/96 ≈ 0.812
|
||||
Unweighted: (1.0 + 0.81) / 2 = 0.905
|
||||
|
||||
Model A: 1 sample at 1.0. Model B: 95 samples at 0.81.
|
||||
Model C: no outcomes, no benchmark — so prior = peer_rate directly.
|
||||
Model C's blended_score must be ~0.812 (pooled), not ~0.905 (unweighted).
|
||||
"""
|
||||
from config import load_config
|
||||
from proficiency_store import add_outcome, set_leaderboard
|
||||
|
||||
cfg = load_config("config/config.yaml")
|
||||
conn = _models_db(tmp_path, [
|
||||
("model-a", "a", "standard", "default", "full"),
|
||||
("model-b", "b", "standard", "default", "full"),
|
||||
("model-c", "c", "standard", "default", "full"),
|
||||
])
|
||||
|
||||
# Model C gets a row with no benchmark so its prior = peer_rate directly.
|
||||
set_leaderboard(conn, cfg, "model-c", "nw", "coding_general", None)
|
||||
|
||||
# Model A: 1 sample at 1.0
|
||||
add_outcome(conn, cfg, "model-a", "nw", "coding_general", [1.0])
|
||||
# Model B: 95 samples at 0.81
|
||||
add_outcome(conn, cfg, "model-b", "nw", "coding_general", [0.81] * 95)
|
||||
|
||||
# After the second add_outcome, recompute_category runs over all three:
|
||||
# peer_rate is now pooled from A (1×1.0) + B (95×0.81).
|
||||
c_row = _row(conn, "model-c")
|
||||
expected_pooled = (1.0 + 95 * 0.81) / (1 + 95) # ≈ 0.812
|
||||
assert c_row["source"] == "outcome_prior"
|
||||
assert c_row["blended_score"] == pytest.approx(expected_pooled, rel=1e-3)
|
||||
# Confirm it's NOT the unweighted result
|
||||
unweighted = (1.0 + 0.81) / 2 # ≈ 0.905
|
||||
assert c_row["blended_score"] < unweighted - 0.09
|
||||
|
||||
Reference in New Issue
Block a user