fix(proficiency): exposure-bias fix + backlog fold — empirical-Bayes scoring, pooled peer_rate, exploration live #62

Merged
alee merged 1 commits from exploration-bias-fix into main 2026-09-09 02:00:44 +00:00
2 changed files with 56 additions and 10 deletions

View File

@@ -72,9 +72,10 @@ def recompute_category(
For each row: For each row:
1. Re-compute benchmark via ``blend()`` on stored leaderboard / self_eval 1. Re-compute benchmark via ``blend()`` on stored leaderboard / self_eval
components. If neither component exists the row skips the conversion. components. If neither component exists the row skips the conversion.
2. Collect *all* rows with ``outcome_samples > 0`` and compute the 2. Collect *all* rows with ``outcome_samples > 0`` and compute the
category-level *peer_rate* (mean outcome) and *peer_benchmark* (mean category-level *peer_rate* (sample-weighted pooled rate:
benchmark_score). Σoutcome_samples × outcome_score / Σoutcome_samples) and
*peer_benchmark* (mean benchmark_score over the same set).
3. Call ``expected_success_rate`` per row with the benchmark, outcome, and 3. Call ``expected_success_rate`` per row with the benchmark, outcome, and
peer statistics. Write the final ``blended_score`` and ``source``. peer statistics. Write the final ``blended_score`` and ``source``.
""" """
@@ -127,14 +128,20 @@ def recompute_category(
] ]
if outcome_rows: if outcome_rows:
# Pooled peer_rate: Σ(outcome_samples × outcome_score) / Σ(outcome_samples)
# over all trafficked models in the category. The unweighted mean of
# per-model rates would let a 1-sample model at 100 % dominate the
# category signal — exactly the small-sample noise the pooled formula
# exists to suppress. Exclude a row from BOTH sides if its
# outcome_score is None (invariant from add_outcome, but be safe).
rates = [d for d in outcome_rows if d["outcome_score"] is not None]
if rates:
peer_rate = ( peer_rate = (
sum( sum(d["outcome_samples"] * d["outcome_score"] for d in rates)
d["outcome_score"] / sum(d["outcome_samples"] for d in rates)
for d in outcome_rows
if d["outcome_score"] is not None
)
/ len(outcome_rows)
) )
else:
peer_rate = None
benchmark_scores = [ benchmark_scores = [
d["benchmark_score"] d["benchmark_score"]
for d in outcome_rows for d in outcome_rows

View File

@@ -867,3 +867,42 @@ def test_expected_success_rate_prior_falls_back_to_peer_rate_when_peer_benchmark
) )
assert source == "outcome_prior" assert source == "outcome_prior"
assert score == pytest.approx(0.50) assert score == pytest.approx(0.50)
def test_recompute_category_uses_pooled_peer_rate(tmp_path):
"""Peer rate is sample-weighted Σ(n·rate)/Σn, not unweighted mean of rates.
Pooled: (1×1.0 + 95×0.81) / (1+95) = 77.95/96 ≈ 0.812
Unweighted: (1.0 + 0.81) / 2 = 0.905
Model A: 1 sample at 1.0. Model B: 95 samples at 0.81.
Model C: no outcomes, no benchmark — so prior = peer_rate directly.
Model C's blended_score must be ~0.812 (pooled), not ~0.905 (unweighted).
"""
from config import load_config
from proficiency_store import add_outcome, set_leaderboard
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("model-a", "a", "standard", "default", "full"),
("model-b", "b", "standard", "default", "full"),
("model-c", "c", "standard", "default", "full"),
])
# Model C gets a row with no benchmark so its prior = peer_rate directly.
set_leaderboard(conn, cfg, "model-c", "nw", "coding_general", None)
# Model A: 1 sample at 1.0
add_outcome(conn, cfg, "model-a", "nw", "coding_general", [1.0])
# Model B: 95 samples at 0.81
add_outcome(conn, cfg, "model-b", "nw", "coding_general", [0.81] * 95)
# After the second add_outcome, recompute_category runs over all three:
# peer_rate is now pooled from A (1×1.0) + B (95×0.81).
c_row = _row(conn, "model-c")
expected_pooled = (1.0 + 95 * 0.81) / (1 + 95) # ≈ 0.812
assert c_row["source"] == "outcome_prior"
assert c_row["blended_score"] == pytest.approx(expected_pooled, rel=1e-3)
# Confirm it's NOT the unweighted result
unweighted = (1.0 + 0.81) / 2 # ≈ 0.905
assert c_row["blended_score"] < unweighted - 0.09