fix(proficiency): exposure-bias fix + backlog fold — empirical-Bayes scoring, pooled peer_rate, exploration live #62
@@ -72,9 +72,10 @@ def recompute_category(
|
|||||||
For each row:
|
For each row:
|
||||||
1. Re-compute benchmark via ``blend()`` on stored leaderboard / self_eval
|
1. Re-compute benchmark via ``blend()`` on stored leaderboard / self_eval
|
||||||
components. If neither component exists the row skips the conversion.
|
components. If neither component exists the row skips the conversion.
|
||||||
2. Collect *all* rows with ``outcome_samples > 0`` and compute the
|
2. Collect *all* rows with ``outcome_samples > 0`` and compute the
|
||||||
category-level *peer_rate* (mean outcome) and *peer_benchmark* (mean
|
category-level *peer_rate* (sample-weighted pooled rate:
|
||||||
benchmark_score).
|
Σoutcome_samples × outcome_score / Σoutcome_samples) and
|
||||||
|
*peer_benchmark* (mean benchmark_score over the same set).
|
||||||
3. Call ``expected_success_rate`` per row with the benchmark, outcome, and
|
3. Call ``expected_success_rate`` per row with the benchmark, outcome, and
|
||||||
peer statistics. Write the final ``blended_score`` and ``source``.
|
peer statistics. Write the final ``blended_score`` and ``source``.
|
||||||
"""
|
"""
|
||||||
@@ -127,14 +128,20 @@ def recompute_category(
|
|||||||
]
|
]
|
||||||
|
|
||||||
if outcome_rows:
|
if outcome_rows:
|
||||||
peer_rate = (
|
# Pooled peer_rate: Σ(outcome_samples × outcome_score) / Σ(outcome_samples)
|
||||||
sum(
|
# over all trafficked models in the category. The unweighted mean of
|
||||||
d["outcome_score"]
|
# per-model rates would let a 1-sample model at 100 % dominate the
|
||||||
for d in outcome_rows
|
# category signal — exactly the small-sample noise the pooled formula
|
||||||
if d["outcome_score"] is not None
|
# exists to suppress. Exclude a row from BOTH sides if its
|
||||||
|
# outcome_score is None (invariant from add_outcome, but be safe).
|
||||||
|
rates = [d for d in outcome_rows if d["outcome_score"] is not None]
|
||||||
|
if rates:
|
||||||
|
peer_rate = (
|
||||||
|
sum(d["outcome_samples"] * d["outcome_score"] for d in rates)
|
||||||
|
/ sum(d["outcome_samples"] for d in rates)
|
||||||
)
|
)
|
||||||
/ len(outcome_rows)
|
else:
|
||||||
)
|
peer_rate = None
|
||||||
benchmark_scores = [
|
benchmark_scores = [
|
||||||
d["benchmark_score"]
|
d["benchmark_score"]
|
||||||
for d in outcome_rows
|
for d in outcome_rows
|
||||||
|
|||||||
@@ -867,3 +867,42 @@ def test_expected_success_rate_prior_falls_back_to_peer_rate_when_peer_benchmark
|
|||||||
)
|
)
|
||||||
assert source == "outcome_prior"
|
assert source == "outcome_prior"
|
||||||
assert score == pytest.approx(0.50)
|
assert score == pytest.approx(0.50)
|
||||||
|
|
||||||
|
|
||||||
|
def test_recompute_category_uses_pooled_peer_rate(tmp_path):
|
||||||
|
"""Peer rate is sample-weighted Σ(n·rate)/Σn, not unweighted mean of rates.
|
||||||
|
|
||||||
|
Pooled: (1×1.0 + 95×0.81) / (1+95) = 77.95/96 ≈ 0.812
|
||||||
|
Unweighted: (1.0 + 0.81) / 2 = 0.905
|
||||||
|
|
||||||
|
Model A: 1 sample at 1.0. Model B: 95 samples at 0.81.
|
||||||
|
Model C: no outcomes, no benchmark — so prior = peer_rate directly.
|
||||||
|
Model C's blended_score must be ~0.812 (pooled), not ~0.905 (unweighted).
|
||||||
|
"""
|
||||||
|
from config import load_config
|
||||||
|
from proficiency_store import add_outcome, set_leaderboard
|
||||||
|
|
||||||
|
cfg = load_config("config/config.yaml")
|
||||||
|
conn = _models_db(tmp_path, [
|
||||||
|
("model-a", "a", "standard", "default", "full"),
|
||||||
|
("model-b", "b", "standard", "default", "full"),
|
||||||
|
("model-c", "c", "standard", "default", "full"),
|
||||||
|
])
|
||||||
|
|
||||||
|
# Model C gets a row with no benchmark so its prior = peer_rate directly.
|
||||||
|
set_leaderboard(conn, cfg, "model-c", "nw", "coding_general", None)
|
||||||
|
|
||||||
|
# Model A: 1 sample at 1.0
|
||||||
|
add_outcome(conn, cfg, "model-a", "nw", "coding_general", [1.0])
|
||||||
|
# Model B: 95 samples at 0.81
|
||||||
|
add_outcome(conn, cfg, "model-b", "nw", "coding_general", [0.81] * 95)
|
||||||
|
|
||||||
|
# After the second add_outcome, recompute_category runs over all three:
|
||||||
|
# peer_rate is now pooled from A (1×1.0) + B (95×0.81).
|
||||||
|
c_row = _row(conn, "model-c")
|
||||||
|
expected_pooled = (1.0 + 95 * 0.81) / (1 + 95) # ≈ 0.812
|
||||||
|
assert c_row["source"] == "outcome_prior"
|
||||||
|
assert c_row["blended_score"] == pytest.approx(expected_pooled, rel=1e-3)
|
||||||
|
# Confirm it's NOT the unweighted result
|
||||||
|
unweighted = (1.0 + 0.81) / 2 # ≈ 0.905
|
||||||
|
assert c_row["blended_score"] < unweighted - 0.09
|
||||||
|
|||||||
Reference in New Issue
Block a user