recompute_category took the unweighted mean of per-model outcome rates, letting a 1-sample 100% model dominate the category prior as much as a 95-sample 81% one — the exact small-sample noise the empirical-Bayes fix exists to suppress. Now Σ(n·rate)/Σn over the trafficked set, as plans/proficiency-exposure-bias-and-exploration.md §2 specifies. Measured on the live replay: −2.2% est. cost vs the unweighted form, and the debugging prior un-distorts by 0.18. Lands before the backlog fold so the whole table re-derives contract-correct on the fold's recompute. Retroactive to the 38 rows already folded 2026-09-07 (recompute_category re-derives from stored components; no migration needed).
909 lines
34 KiB
Python
909 lines
34 KiB
Python
"""Tests for proficiency.py — blending leaderboard priors with self-eval.
|
||
|
||
The interesting cases are the ones the design doc's one-line rule does not
|
||
cover: what happens when a model has no leaderboard prior (common, because
|
||
NeuralWatt ships models faster than benchmarks cover them) and when a run
|
||
produces no usable scores at all.
|
||
"""
|
||
|
||
import pytest
|
||
|
||
from config import load_config
|
||
from proficiency import accumulate, blend, expected_success_rate
|
||
from proficiency_outcome import add_outcome, recompute_category
|
||
|
||
BLEND_KW = {"leaderboard_weight": 0.3, "self_eval_weight": 0.7, "min_samples": 10}
|
||
|
||
|
||
def _blend(lb, se, n, **over):
|
||
return blend(lb, se, n, **{**BLEND_KW, **over})
|
||
|
||
|
||
# --- the documented rule --------------------------------------------------
|
||
|
||
def test_blends_both_sources_once_samples_suffice():
|
||
score, source = _blend(0.4, 0.9, 10)
|
||
assert score == pytest.approx(0.3 * 0.4 + 0.7 * 0.9)
|
||
assert source == "blended"
|
||
|
||
|
||
def test_leaderboard_alone_below_the_sample_threshold():
|
||
# Thin self-eval must not dominate a published benchmark early
|
||
score, source = _blend(0.4, 0.9, 9)
|
||
assert (score, source) == (0.4, "leaderboard")
|
||
|
||
|
||
def test_threshold_is_inclusive():
|
||
assert _blend(0.4, 0.9, 10)[1] == "blended"
|
||
assert _blend(0.4, 0.9, 9)[1] == "leaderboard"
|
||
|
||
|
||
# --- no leaderboard prior, which is the common case -----------------------
|
||
|
||
def test_self_eval_alone_when_no_prior_exists():
|
||
# Given: a model NeuralWatt added that no public benchmark covers yet.
|
||
# Read literally, the design doc's rule would fall back to a leaderboard
|
||
# score that does not exist and yield nothing — discarding real evidence.
|
||
score, source = _blend(None, 0.8, 10)
|
||
assert (score, source) == (0.8, "self_eval")
|
||
|
||
|
||
def test_thin_self_eval_still_beats_nothing():
|
||
# Nine real samples are worse than twelve, but far better than the
|
||
# neutral 0.5 a model gets for being unmeasured. Flagged so a caller can
|
||
# tell it apart from a score that cleared the threshold.
|
||
score, source = _blend(None, 0.8, 9)
|
||
assert (score, source) == (0.8, "self_eval_thin")
|
||
|
||
|
||
def test_unmeasured_model_returns_none_not_zero():
|
||
# None leaves the candidate on the neutral 0.5 downstream. Zero would
|
||
# rank it below every measured model for the crime of being new.
|
||
assert _blend(None, None, 0) == (None, None)
|
||
|
||
|
||
def test_zero_samples_is_not_evidence():
|
||
# A score with no samples behind it is a leftover, not a measurement
|
||
assert _blend(None, 0.9, 0) == (None, None)
|
||
assert _blend(0.4, 0.9, 0) == (0.4, "leaderboard")
|
||
|
||
|
||
def test_a_genuine_zero_score_is_kept():
|
||
# 0.0 means "measured, and it failed everything" — distinct from unmeasured
|
||
score, source = _blend(None, 0.0, 12)
|
||
assert (score, source) == (0.0, "self_eval")
|
||
|
||
|
||
# --- accumulation ---------------------------------------------------------
|
||
|
||
def test_first_run_sets_the_mean():
|
||
assert accumulate(None, 0, [1.0, 0.5, 0.0]) == (0.5, 3)
|
||
|
||
|
||
def test_later_runs_tighten_rather_than_replace():
|
||
# Given: 0.8 over 10 samples, then a run of 2 perfect scores
|
||
score, samples = accumulate(0.8, 10, [1.0, 1.0])
|
||
assert samples == 12
|
||
assert score == pytest.approx((0.8 * 10 + 2.0) / 12)
|
||
|
||
|
||
def test_an_all_errors_run_changes_nothing():
|
||
# Given: every task errored, so there are no scores. Resetting a model's
|
||
# history to zero on a bad run would silently erase months of evidence.
|
||
assert accumulate(0.8, 10, []) == (0.8, 10)
|
||
|
||
|
||
def test_no_history_and_no_scores_stays_empty():
|
||
assert accumulate(None, 0, []) == (None, 0)
|
||
|
||
|
||
def test_stale_score_with_zero_samples_is_overwritten():
|
||
# A score with no samples behind it carries no weight in the average
|
||
assert accumulate(0.9, 0, [0.1, 0.3]) == (pytest.approx(0.2), 2)
|
||
|
||
|
||
# --- variant inheritance --------------------------------------------------
|
||
|
||
def _models_db(tmp_path, rows, provider="nw"):
|
||
import sqlite3
|
||
from pathlib import Path
|
||
schema = (Path(__file__).resolve().parent.parent / "config" / "schema.sql").read_text()
|
||
conn = sqlite3.connect(tmp_path / "t.db")
|
||
conn.row_factory = sqlite3.Row
|
||
conn.executescript(schema)
|
||
for model_id, base, latency, reasoning, ctx in rows:
|
||
conn.execute(
|
||
"""
|
||
INSERT INTO models (model_id, provider, base_model_id, latency_class,
|
||
reasoning_mode, context_variant, availability, last_updated)
|
||
VALUES (?, ?, ?, ?, ?, ?, 'active', '2026-08-17T00:00:00+00:00')
|
||
""",
|
||
(model_id, provider, base, latency, reasoning, ctx),
|
||
)
|
||
conn.commit()
|
||
return conn
|
||
|
||
|
||
def test_flex_inherits_from_its_standard_equivalent(tmp_path):
|
||
from pathlib import Path
|
||
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval, propagate_to_variants
|
||
|
||
cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0])
|
||
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
|
||
got = conn.execute(
|
||
"SELECT self_eval_score FROM proficiency WHERE model_id='kimi-k3-flex'"
|
||
).fetchone()
|
||
assert got["self_eval_score"] == 1.0
|
||
|
||
|
||
def test_fast_does_not_inherit_reasoning_on_quality(tmp_path):
|
||
# A '-fast' row runs with reasoning off or capped, so it is NOT the same
|
||
# model for quality purposes. Inheriting across that would credit it with
|
||
# its reasoning-enabled sibling's answers.
|
||
from pathlib import Path
|
||
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval, propagate_to_variants
|
||
|
||
cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-fast", "kimi-k3", "standard", "reduced", "full"),
|
||
])
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "reasoning_math", [1.0])
|
||
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0
|
||
assert conn.execute(
|
||
"SELECT COUNT(*) c FROM proficiency WHERE model_id='kimi-k3-fast'"
|
||
).fetchone()["c"] == 0
|
||
|
||
|
||
def test_a_directly_measured_variant_is_never_overwritten(tmp_path):
|
||
from pathlib import Path
|
||
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval, propagate_to_variants
|
||
|
||
cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
|
||
add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.2])
|
||
propagate_to_variants(conn, cfg, "kimi-k3", "nw")
|
||
got = conn.execute(
|
||
"SELECT self_eval_score FROM proficiency WHERE model_id='kimi-k3-flex'"
|
||
).fetchone()
|
||
assert got["self_eval_score"] == 0.2
|
||
|
||
|
||
def test_a_score_above_one_never_reaches_the_table(tmp_path):
|
||
"""The single write path is where a bad scorer gets stopped.
|
||
|
||
A blended_score above 1.0 does not just misreport one row: it raises
|
||
`best` in rank_candidates and shifts every other candidate's quality band.
|
||
"""
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.5, 1.0])
|
||
|
||
row = conn.execute(
|
||
"SELECT self_eval_score, blended_score FROM proficiency WHERE model_id = 'kimi-k3'"
|
||
).fetchone()
|
||
assert row["self_eval_score"] == 1.0
|
||
assert row["blended_score"] <= 1.0
|
||
|
||
|
||
def test_a_negative_score_is_floored_at_zero(tmp_path):
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [-0.5, 0.5])
|
||
|
||
assert conn.execute(
|
||
"SELECT self_eval_score FROM proficiency WHERE model_id = 'kimi-k3'"
|
||
).fetchone()["self_eval_score"] == 0.25
|
||
|
||
|
||
def test_an_inherited_variant_refreshes_when_its_family_is_remeasured(tmp_path):
|
||
"""Inheritance was a one-shot: a variant froze at its first copy, forever.
|
||
|
||
The old guard skipped any row with self_eval_samples > 0, and an inherited
|
||
row has samples > 0 because inheritance copies them -- so it could never
|
||
tell "measured here" from "copied here" and never refreshed. Observed live:
|
||
kimi-k3 reached 0.957 while kimi-k3-flex sat at 0.85 with a week-old
|
||
timestamp, and the eval reported "propagated 0 inherited rows". Flex rows
|
||
serve auto:batch, so those requests ranked on stale scores.
|
||
"""
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval, propagate_to_variants
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [0.85, 0.85])
|
||
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
|
||
inherited = conn.execute(
|
||
"SELECT blended_score, inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'"
|
||
).fetchone()
|
||
assert inherited["blended_score"] == pytest.approx(0.85)
|
||
assert inherited["inherited_from"] == "kimi-k3"
|
||
|
||
# The family learns more; the variant must follow it.
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0, 1.0, 1.0, 1.0])
|
||
|
||
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
|
||
refreshed = conn.execute(
|
||
"SELECT blended_score, self_eval_samples FROM proficiency WHERE model_id='kimi-k3-flex'"
|
||
).fetchone()
|
||
assert refreshed["blended_score"] == pytest.approx(0.95)
|
||
assert refreshed["self_eval_samples"] == 6
|
||
|
||
|
||
def test_a_measured_variant_still_outranks_its_family(tmp_path):
|
||
"""The protection the old guard was reaching for, kept intact."""
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval, propagate_to_variants
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0])
|
||
add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "docs_writing", [0.2])
|
||
|
||
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0
|
||
row = conn.execute(
|
||
"SELECT blended_score, inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'"
|
||
).fetchone()
|
||
assert row["blended_score"] == pytest.approx(0.2)
|
||
assert row["inherited_from"] is None
|
||
|
||
|
||
def test_a_database_without_the_column_is_migrated(tmp_path):
|
||
"""schema.sql only defines a NEW database; existing ones need the ALTER."""
|
||
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval, ensure_columns
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
|
||
conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from")
|
||
assert "inherited_from" not in {
|
||
r[1] for r in conn.execute("PRAGMA table_info(proficiency)")
|
||
}
|
||
|
||
ensure_columns(conn)
|
||
ensure_columns(conn) # idempotent
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0])
|
||
assert conn.execute(
|
||
"SELECT inherited_from FROM proficiency WHERE model_id='kimi-k3'"
|
||
).fetchone()["inherited_from"] is None
|
||
|
||
|
||
def test_migration_unfreezes_variants_that_predate_the_column(tmp_path):
|
||
"""The migration must ship the repair, not just the fix.
|
||
|
||
ADD COLUMN gives every existing row NULL, and NULL means "measured here" --
|
||
so without the backfill, the rows this whole change exists for would stay
|
||
frozen and the live flex rows would never catch up.
|
||
"""
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval, ensure_columns, propagate_to_variants
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
# An old database: both rows carry samples, neither records provenance.
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [0.957])
|
||
add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "docs_writing", [0.85])
|
||
conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from")
|
||
|
||
ensure_columns(conn)
|
||
|
||
assert conn.execute(
|
||
"SELECT inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'"
|
||
).fetchone()["inherited_from"] == "kimi-k3"
|
||
# ...so the next run actually moves it.
|
||
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
|
||
assert conn.execute(
|
||
"SELECT blended_score FROM proficiency WHERE model_id='kimi-k3-flex'"
|
||
).fetchone()["blended_score"] == pytest.approx(0.957)
|
||
|
||
|
||
def test_the_backfill_never_touches_a_standard_row(tmp_path):
|
||
"""Only flex rows with a standard equivalent are provably inherited."""
|
||
from config import load_config
|
||
from proficiency_store import add_self_eval, ensure_columns
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-fast", "kimi-k3", "standard", "reduced", "full"),
|
||
])
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0])
|
||
add_self_eval(conn, cfg, "kimi-k3-fast", "nw", "docs_writing", [0.888])
|
||
conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from")
|
||
|
||
ensure_columns(conn)
|
||
|
||
for model_id in ("kimi-k3", "kimi-k3-fast"):
|
||
assert conn.execute(
|
||
"SELECT inherited_from FROM proficiency WHERE model_id=?", (model_id,)
|
||
).fetchone()["inherited_from"] is None
|
||
|
||
|
||
def test_apply_priors_recomputes_each_touched_category(tmp_path):
|
||
"""A leaderboard import must run category-wide recompute for every touched category."""
|
||
from unittest import mock
|
||
|
||
from config import load_config
|
||
from leaderboard import apply_priors
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("qwen-3", "qwen-3", "standard", "default", "full"),
|
||
], provider="neuralwatt")
|
||
|
||
families = {"kimi-k3": ["kimi-k3"], "qwen-3": ["qwen-3"]}
|
||
priors = {
|
||
"kimi-k3": {"coding_general": 0.9, "docs_writing": 0.8},
|
||
"qwen-3": {"coding_general": 0.7},
|
||
}
|
||
|
||
with mock.patch("leaderboard.recompute_category") as mock_recompute:
|
||
written = apply_priors(conn, cfg, families, priors, "neuralwatt")
|
||
|
||
assert written == 3
|
||
touched = {call.args[2] for call in mock_recompute.call_args_list}
|
||
assert touched == {"coding_general", "docs_writing"}
|
||
|
||
|
||
def test_apply_priors_blends_leaderboard_scores(tmp_path):
|
||
"""A leaderboard import writes the prior as the blended score/source."""
|
||
from config import load_config
|
||
from leaderboard import apply_priors
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
], provider="neuralwatt")
|
||
|
||
families = {"kimi-k3": ["kimi-k3"]}
|
||
priors = {"kimi-k3": {"coding_general": 0.85}}
|
||
|
||
assert apply_priors(conn, cfg, families, priors, "neuralwatt") == 1
|
||
|
||
row = conn.execute(
|
||
"SELECT blended_score, source FROM proficiency "
|
||
"WHERE model_id='kimi-k3' AND category='coding_general'"
|
||
).fetchone()
|
||
assert row["blended_score"] == pytest.approx(0.85)
|
||
assert row["source"] == "leaderboard"
|
||
|
||
|
||
def test_apply_priors_recompute_handles_empty_category_gracefully(tmp_path):
|
||
"""The final category recompute must not crash when a touched category has no rows."""
|
||
from config import load_config
|
||
from leaderboard import apply_priors
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [], provider="neuralwatt")
|
||
|
||
families: dict[str, list[str]] = {"ghost-family": ["ghost-model"]}
|
||
priors: dict[str, dict[str, float]] = {"ghost-family": {"coding_general": 0.9}}
|
||
|
||
assert apply_priors(conn, cfg, families, priors, "neuralwatt") == 0
|
||
|
||
|
||
def _row(conn, model_id):
|
||
return conn.execute(
|
||
"""
|
||
SELECT leaderboard_score, self_eval_score, self_eval_samples,
|
||
outcome_score, outcome_samples, blended_score, source,
|
||
inherited_from
|
||
FROM proficiency
|
||
WHERE model_id = ? AND provider = 'nw' AND category = 'coding_general'
|
||
""",
|
||
(model_id,),
|
||
).fetchone()
|
||
|
||
|
||
def test_set_leaderboard_preserves_outcome_evidence(tmp_path):
|
||
from config import load_config
|
||
from proficiency_store import add_outcome, set_leaderboard
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
|
||
|
||
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.7)
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0])
|
||
before = _row(conn, "kimi-k3")
|
||
|
||
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8)
|
||
after = _row(conn, "kimi-k3")
|
||
|
||
assert after["leaderboard_score"] == pytest.approx(0.8)
|
||
assert after["outcome_score"] == before["outcome_score"]
|
||
assert after["outcome_samples"] == before["outcome_samples"]
|
||
|
||
|
||
def test_add_self_eval_preserves_outcome_evidence(tmp_path):
|
||
from config import load_config
|
||
from proficiency_store import add_outcome, add_self_eval
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.6, 0.6])
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.8])
|
||
|
||
row = _row(conn, "kimi-k3")
|
||
assert row["self_eval_score"] == pytest.approx(0.6667, abs=1e-4)
|
||
assert row["outcome_score"] == 1.0
|
||
assert row["outcome_samples"] == 1
|
||
|
||
|
||
def test_add_outcome_accumulates_and_recomputes_category(tmp_path):
|
||
from config import load_config
|
||
from proficiency_store import add_outcome, set_leaderboard
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("qwen-3", "qwen-3", "standard", "default", "full"),
|
||
])
|
||
|
||
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.9)
|
||
set_leaderboard(conn, cfg, "qwen-3", "nw", "coding_general", 0.9)
|
||
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0] * 5)
|
||
add_outcome(conn, cfg, "qwen-3", "nw", "coding_general", [0.0] * 5)
|
||
|
||
strong = _row(conn, "kimi-k3")
|
||
weak = _row(conn, "qwen-3")
|
||
|
||
assert strong["outcome_score"] == pytest.approx(1.0)
|
||
assert weak["outcome_score"] == pytest.approx(0.0)
|
||
assert strong["blended_score"] > weak["blended_score"]
|
||
assert strong["source"] == "outcome_blended"
|
||
assert weak["source"] == "outcome_blended"
|
||
|
||
|
||
def test_recompute_category_is_idempotent(tmp_path):
|
||
from config import load_config
|
||
from proficiency_store import add_outcome, recompute_category, set_leaderboard
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
|
||
|
||
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8)
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 0.5])
|
||
|
||
first = _row(conn, "kimi-k3")
|
||
recompute_category(conn, cfg, "coding_general")
|
||
second = _row(conn, "kimi-k3")
|
||
|
||
assert first["blended_score"] == pytest.approx(second["blended_score"])
|
||
assert first["source"] == second["source"]
|
||
|
||
|
||
def test_propagate_to_variants_skips_variant_with_own_outcomes(tmp_path):
|
||
from config import load_config
|
||
from proficiency_store import add_outcome, add_self_eval, propagate_to_variants
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
|
||
add_outcome(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.0])
|
||
|
||
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0
|
||
|
||
variant = _row(conn, "kimi-k3-flex")
|
||
assert variant["outcome_score"] == pytest.approx(0.0)
|
||
assert variant["outcome_samples"] == 1
|
||
|
||
|
||
def test_propagate_to_variants_copies_outcomes_to_blank_variant(tmp_path):
|
||
from config import load_config
|
||
from proficiency_store import add_outcome, add_self_eval, propagate_to_variants
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
|
||
|
||
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
|
||
|
||
variant = _row(conn, "kimi-k3-flex")
|
||
assert variant["outcome_score"] == pytest.approx(1.0)
|
||
assert variant["outcome_samples"] == 1
|
||
assert variant["inherited_from"] == "kimi-k3"
|
||
|
||
|
||
def test_add_outcome_commits_category_recompute_atomically(tmp_path):
|
||
import sqlite3
|
||
import threading
|
||
from unittest import mock
|
||
|
||
from config import load_config
|
||
from proficiency_outcome import _write as real_write
|
||
from proficiency_store import add_outcome, set_leaderboard
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
setup = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("qwen-3", "qwen-3", "standard", "default", "full"),
|
||
])
|
||
db_path = tmp_path / "t.db"
|
||
|
||
set_leaderboard(setup, cfg, "kimi-k3", "nw", "coding_general", 0.9)
|
||
set_leaderboard(setup, cfg, "qwen-3", "nw", "coding_general", 0.9)
|
||
|
||
pre = setup.execute(
|
||
"""
|
||
SELECT model_id, source, blended_score FROM proficiency
|
||
WHERE category = 'coding_general'
|
||
ORDER BY model_id
|
||
"""
|
||
).fetchall()
|
||
setup.close()
|
||
|
||
ready = threading.Event()
|
||
resume = threading.Event()
|
||
recompute_writes = [0]
|
||
|
||
def _pausing_write(*args, **kwargs):
|
||
if kwargs.get("_blended_override"):
|
||
recompute_writes[0] += 1
|
||
if recompute_writes[0] == 1:
|
||
ready.set()
|
||
resume.wait(timeout=5.0)
|
||
return real_write(*args, **kwargs)
|
||
|
||
def _target():
|
||
conn = sqlite3.connect(db_path)
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0] * 5)
|
||
conn.close()
|
||
|
||
with mock.patch("proficiency_outcome._write", side_effect=_pausing_write):
|
||
runner = threading.Thread(target=_target)
|
||
runner.start()
|
||
ready.wait(timeout=5.0)
|
||
|
||
reader = sqlite3.connect(db_path)
|
||
reader.row_factory = sqlite3.Row
|
||
during = reader.execute(
|
||
"""
|
||
SELECT model_id, source, blended_score FROM proficiency
|
||
WHERE category = 'coding_general'
|
||
ORDER BY model_id
|
||
"""
|
||
).fetchall()
|
||
reader.close()
|
||
|
||
assert [(r["model_id"], r["source"], r["blended_score"]) for r in during] == [
|
||
(r["model_id"], r["source"], r["blended_score"]) for r in pre
|
||
]
|
||
|
||
resume.set()
|
||
runner.join(timeout=5.0)
|
||
|
||
final = sqlite3.connect(db_path)
|
||
final.row_factory = sqlite3.Row
|
||
rows = final.execute(
|
||
"""
|
||
SELECT model_id, source, blended_score FROM proficiency
|
||
WHERE category = 'coding_general'
|
||
ORDER BY model_id
|
||
"""
|
||
).fetchall()
|
||
final.close()
|
||
assert {r["source"] for r in rows} == {"outcome_blended", "outcome_prior"}
|
||
assert all(r["blended_score"] == pytest.approx(1.0) for r in rows)
|
||
|
||
|
||
def test_cold_benchmark_model_does_not_outrank_proven_model_in_same_category(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("proven-mod", "proven", "standard", "default", "full"),
|
||
("cold-mod", "cold", "standard", "default", "full"),
|
||
])
|
||
from proficiency_store import set_leaderboard
|
||
|
||
set_leaderboard(conn, cfg, "proven-mod", "nw", "coding_general", 0.85)
|
||
add_outcome(conn, cfg, "proven-mod", "nw", "coding_general", [0.9, 0.9])
|
||
set_leaderboard(conn, cfg, "cold-mod", "nw", "coding_general", 0.7)
|
||
|
||
cold = _row(conn, "cold-mod")
|
||
proven = _row(conn, "proven-mod")
|
||
|
||
assert proven["source"] == "outcome_blended"
|
||
assert cold["source"] == "outcome_prior"
|
||
assert proven["blended_score"] > cold["blended_score"]
|
||
|
||
|
||
def test_cold_category_preserves_benchmark_score_and_source_verbatim(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("thin-mod", "thin", "standard", "default", "full")])
|
||
from proficiency_store import add_self_eval
|
||
|
||
add_self_eval(conn, cfg, "thin-mod", "nw", "coding_general", [0.75, 0.75, 0.75])
|
||
|
||
before = _row(conn, "thin-mod")
|
||
assert before["source"] == "self_eval_thin"
|
||
saved_score = before["blended_score"]
|
||
|
||
recompute_category(conn, cfg, "coding_general")
|
||
|
||
after = _row(conn, "thin-mod")
|
||
assert after["blended_score"] == pytest.approx(saved_score)
|
||
assert after["source"] == "self_eval_thin"
|
||
|
||
|
||
def test_taxonomy_cold_category_returns_benchmark_source_verbatim(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("cold-cat", "cold", "standard", "default", "full")])
|
||
from proficiency_store import add_self_eval
|
||
|
||
add_self_eval(conn, cfg, "cold-cat", "nw", "coding_general", [0.6])
|
||
|
||
row = _row(conn, "cold-cat")
|
||
assert row["source"] == "self_eval_thin"
|
||
|
||
|
||
def test_taxonomy_trafficked_no_per_model_returns_outcome_prior(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("target", "tgt", "standard", "default", "full"),
|
||
("peer", "peer", "standard", "default", "full"),
|
||
])
|
||
from proficiency_store import add_self_eval, set_leaderboard
|
||
|
||
add_self_eval(conn, cfg, "peer", "nw", "coding_general", [0.8])
|
||
add_outcome(conn, cfg, "peer", "nw", "coding_general", [1.0, 1.0])
|
||
set_leaderboard(conn, cfg, "target", "nw", "coding_general", 0.7)
|
||
recompute_category(conn, cfg, "coding_general")
|
||
|
||
target = _row(conn, "target")
|
||
assert target["source"] == "outcome_prior"
|
||
|
||
|
||
def test_taxonomy_trafficked_with_per_model_returns_outcome_blended(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [("own", "own", "standard", "default", "full")])
|
||
from proficiency_store import set_leaderboard
|
||
|
||
set_leaderboard(conn, cfg, "own", "nw", "coding_general", 0.7)
|
||
add_outcome(conn, cfg, "own", "nw", "coding_general", [0.8, 0.9])
|
||
|
||
row = _row(conn, "own")
|
||
assert row["source"] == "outcome_blended"
|
||
|
||
|
||
def test_unmeasured_model_in_trafficked_category_borrows_peer_rate(tmp_path):
|
||
"""A model with no benchmark/outcome still gets calibrated when peers exist."""
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("ghost", "ghost", "standard", "default", "full"),
|
||
("peer", "peer", "standard", "default", "full"),
|
||
])
|
||
from proficiency_store import add_outcome, add_self_eval, set_leaderboard
|
||
|
||
set_leaderboard(conn, cfg, "ghost", "nw", "coding_general", None)
|
||
add_self_eval(conn, cfg, "peer", "nw", "coding_general", [0.9])
|
||
add_outcome(conn, cfg, "peer", "nw", "coding_general", [1.0])
|
||
recompute_category(conn, cfg, "coding_general")
|
||
|
||
row = _row(conn, "ghost")
|
||
assert row["source"] == "outcome_prior"
|
||
assert row["blended_score"] == pytest.approx(1.0)
|
||
|
||
|
||
def test_recompute_category_is_idempotent_two_models(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("alpha", "alpha", "standard", "default", "full"),
|
||
("beta", "beta", "standard", "default", "full"),
|
||
])
|
||
from proficiency_store import add_outcome, set_leaderboard
|
||
|
||
set_leaderboard(conn, cfg, "alpha", "nw", "coding_general", 0.9)
|
||
set_leaderboard(conn, cfg, "beta", "nw", "coding_general", 0.9)
|
||
add_outcome(conn, cfg, "alpha", "nw", "coding_general", [1.0, 1.0])
|
||
|
||
first_alpha = _row(conn, "alpha")
|
||
first_beta = _row(conn, "beta")
|
||
|
||
recompute_category(conn, cfg, "coding_general")
|
||
second_alpha = _row(conn, "alpha")
|
||
second_beta = _row(conn, "beta")
|
||
|
||
assert second_alpha["blended_score"] == pytest.approx(first_alpha["blended_score"])
|
||
assert second_alpha["source"] == first_alpha["source"]
|
||
assert second_beta["blended_score"] == pytest.approx(first_beta["blended_score"])
|
||
assert second_beta["source"] == first_beta["source"]
|
||
|
||
|
||
def test_recompute_category_leaves_other_categories_untouched(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("a-model", "a", "standard", "default", "full"),
|
||
("b-model", "b", "standard", "default", "full"),
|
||
])
|
||
from proficiency_store import add_self_eval
|
||
|
||
add_self_eval(conn, cfg, "a-model", "nw", "cat_a", [0.8])
|
||
add_self_eval(conn, cfg, "b-model", "nw", "cat_b", [0.6])
|
||
|
||
row_b_before = conn.execute(
|
||
"SELECT blended_score, source FROM proficiency WHERE model_id='b-model'"
|
||
).fetchone()
|
||
|
||
recompute_category(conn, cfg, "cat_a")
|
||
|
||
row_b_after = conn.execute(
|
||
"SELECT blended_score, source FROM proficiency WHERE model_id='b-model'"
|
||
).fetchone()
|
||
assert row_b_after["blended_score"] == pytest.approx(row_b_before["blended_score"])
|
||
assert row_b_after["source"] == row_b_before["source"]
|
||
|
||
|
||
def test_propagate_to_variants_preserves_outcome_columns(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
from proficiency_store import (
|
||
add_outcome,
|
||
add_self_eval,
|
||
propagate_to_variants,
|
||
set_leaderboard,
|
||
)
|
||
|
||
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8)
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 0.5])
|
||
|
||
before = _row(conn, "kimi-k3")
|
||
assert before["outcome_score"] == pytest.approx(0.75)
|
||
assert before["outcome_samples"] == 2
|
||
|
||
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.888])
|
||
propagate_to_variants(conn, cfg, "kimi-k3", "nw")
|
||
|
||
after = _row(conn, "kimi-k3")
|
||
assert after["outcome_score"] == pytest.approx(0.75)
|
||
assert after["outcome_samples"] == 2
|
||
|
||
|
||
def test_directly_measured_variant_refuses_outcome_propagation(tmp_path):
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("kimi-k3", "kimi-k3", "standard", "default", "full"),
|
||
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
|
||
])
|
||
from proficiency_store import (
|
||
add_outcome,
|
||
add_self_eval,
|
||
propagate_to_variants,
|
||
set_leaderboard,
|
||
)
|
||
|
||
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.9)
|
||
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0])
|
||
add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.5])
|
||
add_outcome(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.3])
|
||
|
||
variant_before = _row(conn, "kimi-k3-flex")
|
||
assert variant_before["outcome_score"] == pytest.approx(0.3)
|
||
assert variant_before["outcome_samples"] == 1
|
||
|
||
count = propagate_to_variants(conn, cfg, "kimi-k3", "nw")
|
||
assert count == 0
|
||
|
||
variant_after = _row(conn, "kimi-k3-flex")
|
||
assert variant_after["outcome_score"] == pytest.approx(0.3)
|
||
assert variant_after["outcome_samples"] == 1
|
||
|
||
|
||
def test_expected_success_rate_prior_pulls_towards_peer_rate():
|
||
score, source = expected_success_rate(
|
||
benchmark_score=0.90,
|
||
benchmark_source="leaderboard",
|
||
outcome_score=None,
|
||
outcome_samples=0,
|
||
peer_rate=0.40,
|
||
peer_benchmark=0.70,
|
||
prior_strength=20,
|
||
)
|
||
assert source == "outcome_prior"
|
||
assert score is not None
|
||
assert score < 0.90
|
||
|
||
|
||
def test_expected_success_rate_prior_falls_back_to_peer_rate_when_peer_benchmark_missing():
|
||
score, source = expected_success_rate(
|
||
benchmark_score=0.60,
|
||
benchmark_source="leaderboard",
|
||
outcome_score=None,
|
||
outcome_samples=0,
|
||
peer_rate=0.50,
|
||
peer_benchmark=None,
|
||
prior_strength=20,
|
||
)
|
||
assert source == "outcome_prior"
|
||
assert score == pytest.approx(0.50)
|
||
|
||
|
||
def test_recompute_category_uses_pooled_peer_rate(tmp_path):
|
||
"""Peer rate is sample-weighted Σ(n·rate)/Σn, not unweighted mean of rates.
|
||
|
||
Pooled: (1×1.0 + 95×0.81) / (1+95) = 77.95/96 ≈ 0.812
|
||
Unweighted: (1.0 + 0.81) / 2 = 0.905
|
||
|
||
Model A: 1 sample at 1.0. Model B: 95 samples at 0.81.
|
||
Model C: no outcomes, no benchmark — so prior = peer_rate directly.
|
||
Model C's blended_score must be ~0.812 (pooled), not ~0.905 (unweighted).
|
||
"""
|
||
from config import load_config
|
||
from proficiency_store import add_outcome, set_leaderboard
|
||
|
||
cfg = load_config("config/config.yaml")
|
||
conn = _models_db(tmp_path, [
|
||
("model-a", "a", "standard", "default", "full"),
|
||
("model-b", "b", "standard", "default", "full"),
|
||
("model-c", "c", "standard", "default", "full"),
|
||
])
|
||
|
||
# Model C gets a row with no benchmark so its prior = peer_rate directly.
|
||
set_leaderboard(conn, cfg, "model-c", "nw", "coding_general", None)
|
||
|
||
# Model A: 1 sample at 1.0
|
||
add_outcome(conn, cfg, "model-a", "nw", "coding_general", [1.0])
|
||
# Model B: 95 samples at 0.81
|
||
add_outcome(conn, cfg, "model-b", "nw", "coding_general", [0.81] * 95)
|
||
|
||
# After the second add_outcome, recompute_category runs over all three:
|
||
# peer_rate is now pooled from A (1×1.0) + B (95×0.81).
|
||
c_row = _row(conn, "model-c")
|
||
expected_pooled = (1.0 + 95 * 0.81) / (1 + 95) # ≈ 0.812
|
||
assert c_row["source"] == "outcome_prior"
|
||
assert c_row["blended_score"] == pytest.approx(expected_pooled, rel=1e-3)
|
||
# Confirm it's NOT the unweighted result
|
||
unweighted = (1.0 + 0.81) / 2 # ≈ 0.905
|
||
assert c_row["blended_score"] < unweighted - 0.09
|