Files
6krrt/tests/test_proficiency.py
adlee-was-taken 9c7119ad4f fix(proficiency): peer_rate is the sample-weighted pooled rate per plan §2
recompute_category took the unweighted mean of per-model outcome rates,
letting a 1-sample 100% model dominate the category prior as much as a
95-sample 81% one — the exact small-sample noise the empirical-Bayes fix
exists to suppress. Now Σ(n·rate)/Σn over the trafficked set, as
plans/proficiency-exposure-bias-and-exploration.md §2 specifies. Measured
on the live replay: −2.2% est. cost vs the unweighted form, and the
debugging prior un-distorts by 0.18.

Lands before the backlog fold so the whole table re-derives contract-correct
on the fold's recompute. Retroactive to the 38 rows already folded 2026-09-07
(recompute_category re-derives from stored components; no migration needed).
2026-09-08 21:24:58 -04:00

909 lines
34 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Tests for proficiency.py — blending leaderboard priors with self-eval.
The interesting cases are the ones the design doc's one-line rule does not
cover: what happens when a model has no leaderboard prior (common, because
NeuralWatt ships models faster than benchmarks cover them) and when a run
produces no usable scores at all.
"""
import pytest
from config import load_config
from proficiency import accumulate, blend, expected_success_rate
from proficiency_outcome import add_outcome, recompute_category
BLEND_KW = {"leaderboard_weight": 0.3, "self_eval_weight": 0.7, "min_samples": 10}
def _blend(lb, se, n, **over):
return blend(lb, se, n, **{**BLEND_KW, **over})
# --- the documented rule --------------------------------------------------
def test_blends_both_sources_once_samples_suffice():
score, source = _blend(0.4, 0.9, 10)
assert score == pytest.approx(0.3 * 0.4 + 0.7 * 0.9)
assert source == "blended"
def test_leaderboard_alone_below_the_sample_threshold():
# Thin self-eval must not dominate a published benchmark early
score, source = _blend(0.4, 0.9, 9)
assert (score, source) == (0.4, "leaderboard")
def test_threshold_is_inclusive():
assert _blend(0.4, 0.9, 10)[1] == "blended"
assert _blend(0.4, 0.9, 9)[1] == "leaderboard"
# --- no leaderboard prior, which is the common case -----------------------
def test_self_eval_alone_when_no_prior_exists():
# Given: a model NeuralWatt added that no public benchmark covers yet.
# Read literally, the design doc's rule would fall back to a leaderboard
# score that does not exist and yield nothing — discarding real evidence.
score, source = _blend(None, 0.8, 10)
assert (score, source) == (0.8, "self_eval")
def test_thin_self_eval_still_beats_nothing():
# Nine real samples are worse than twelve, but far better than the
# neutral 0.5 a model gets for being unmeasured. Flagged so a caller can
# tell it apart from a score that cleared the threshold.
score, source = _blend(None, 0.8, 9)
assert (score, source) == (0.8, "self_eval_thin")
def test_unmeasured_model_returns_none_not_zero():
# None leaves the candidate on the neutral 0.5 downstream. Zero would
# rank it below every measured model for the crime of being new.
assert _blend(None, None, 0) == (None, None)
def test_zero_samples_is_not_evidence():
# A score with no samples behind it is a leftover, not a measurement
assert _blend(None, 0.9, 0) == (None, None)
assert _blend(0.4, 0.9, 0) == (0.4, "leaderboard")
def test_a_genuine_zero_score_is_kept():
# 0.0 means "measured, and it failed everything" — distinct from unmeasured
score, source = _blend(None, 0.0, 12)
assert (score, source) == (0.0, "self_eval")
# --- accumulation ---------------------------------------------------------
def test_first_run_sets_the_mean():
assert accumulate(None, 0, [1.0, 0.5, 0.0]) == (0.5, 3)
def test_later_runs_tighten_rather_than_replace():
# Given: 0.8 over 10 samples, then a run of 2 perfect scores
score, samples = accumulate(0.8, 10, [1.0, 1.0])
assert samples == 12
assert score == pytest.approx((0.8 * 10 + 2.0) / 12)
def test_an_all_errors_run_changes_nothing():
# Given: every task errored, so there are no scores. Resetting a model's
# history to zero on a bad run would silently erase months of evidence.
assert accumulate(0.8, 10, []) == (0.8, 10)
def test_no_history_and_no_scores_stays_empty():
assert accumulate(None, 0, []) == (None, 0)
def test_stale_score_with_zero_samples_is_overwritten():
# A score with no samples behind it carries no weight in the average
assert accumulate(0.9, 0, [0.1, 0.3]) == (pytest.approx(0.2), 2)
# --- variant inheritance --------------------------------------------------
def _models_db(tmp_path, rows, provider="nw"):
import sqlite3
from pathlib import Path
schema = (Path(__file__).resolve().parent.parent / "config" / "schema.sql").read_text()
conn = sqlite3.connect(tmp_path / "t.db")
conn.row_factory = sqlite3.Row
conn.executescript(schema)
for model_id, base, latency, reasoning, ctx in rows:
conn.execute(
"""
INSERT INTO models (model_id, provider, base_model_id, latency_class,
reasoning_mode, context_variant, availability, last_updated)
VALUES (?, ?, ?, ?, ?, ?, 'active', '2026-08-17T00:00:00+00:00')
""",
(model_id, provider, base, latency, reasoning, ctx),
)
conn.commit()
return conn
def test_flex_inherits_from_its_standard_equivalent(tmp_path):
from pathlib import Path
from config import load_config
from proficiency_store import add_self_eval, propagate_to_variants
cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0])
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
got = conn.execute(
"SELECT self_eval_score FROM proficiency WHERE model_id='kimi-k3-flex'"
).fetchone()
assert got["self_eval_score"] == 1.0
def test_fast_does_not_inherit_reasoning_on_quality(tmp_path):
# A '-fast' row runs with reasoning off or capped, so it is NOT the same
# model for quality purposes. Inheriting across that would credit it with
# its reasoning-enabled sibling's answers.
from pathlib import Path
from config import load_config
from proficiency_store import add_self_eval, propagate_to_variants
cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-fast", "kimi-k3", "standard", "reduced", "full"),
])
add_self_eval(conn, cfg, "kimi-k3", "nw", "reasoning_math", [1.0])
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0
assert conn.execute(
"SELECT COUNT(*) c FROM proficiency WHERE model_id='kimi-k3-fast'"
).fetchone()["c"] == 0
def test_a_directly_measured_variant_is_never_overwritten(tmp_path):
from pathlib import Path
from config import load_config
from proficiency_store import add_self_eval, propagate_to_variants
cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.2])
propagate_to_variants(conn, cfg, "kimi-k3", "nw")
got = conn.execute(
"SELECT self_eval_score FROM proficiency WHERE model_id='kimi-k3-flex'"
).fetchone()
assert got["self_eval_score"] == 0.2
def test_a_score_above_one_never_reaches_the_table(tmp_path):
"""The single write path is where a bad scorer gets stopped.
A blended_score above 1.0 does not just misreport one row: it raises
`best` in rank_candidates and shifts every other candidate's quality band.
"""
from config import load_config
from proficiency_store import add_self_eval
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.5, 1.0])
row = conn.execute(
"SELECT self_eval_score, blended_score FROM proficiency WHERE model_id = 'kimi-k3'"
).fetchone()
assert row["self_eval_score"] == 1.0
assert row["blended_score"] <= 1.0
def test_a_negative_score_is_floored_at_zero(tmp_path):
from config import load_config
from proficiency_store import add_self_eval
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [-0.5, 0.5])
assert conn.execute(
"SELECT self_eval_score FROM proficiency WHERE model_id = 'kimi-k3'"
).fetchone()["self_eval_score"] == 0.25
def test_an_inherited_variant_refreshes_when_its_family_is_remeasured(tmp_path):
"""Inheritance was a one-shot: a variant froze at its first copy, forever.
The old guard skipped any row with self_eval_samples > 0, and an inherited
row has samples > 0 because inheritance copies them -- so it could never
tell "measured here" from "copied here" and never refreshed. Observed live:
kimi-k3 reached 0.957 while kimi-k3-flex sat at 0.85 with a week-old
timestamp, and the eval reported "propagated 0 inherited rows". Flex rows
serve auto:batch, so those requests ranked on stale scores.
"""
from config import load_config
from proficiency_store import add_self_eval, propagate_to_variants
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [0.85, 0.85])
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
inherited = conn.execute(
"SELECT blended_score, inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'"
).fetchone()
assert inherited["blended_score"] == pytest.approx(0.85)
assert inherited["inherited_from"] == "kimi-k3"
# The family learns more; the variant must follow it.
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0, 1.0, 1.0, 1.0])
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
refreshed = conn.execute(
"SELECT blended_score, self_eval_samples FROM proficiency WHERE model_id='kimi-k3-flex'"
).fetchone()
assert refreshed["blended_score"] == pytest.approx(0.95)
assert refreshed["self_eval_samples"] == 6
def test_a_measured_variant_still_outranks_its_family(tmp_path):
"""The protection the old guard was reaching for, kept intact."""
from config import load_config
from proficiency_store import add_self_eval, propagate_to_variants
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0])
add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "docs_writing", [0.2])
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0
row = conn.execute(
"SELECT blended_score, inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'"
).fetchone()
assert row["blended_score"] == pytest.approx(0.2)
assert row["inherited_from"] is None
def test_a_database_without_the_column_is_migrated(tmp_path):
"""schema.sql only defines a NEW database; existing ones need the ALTER."""
from config import load_config
from proficiency_store import add_self_eval, ensure_columns
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from")
assert "inherited_from" not in {
r[1] for r in conn.execute("PRAGMA table_info(proficiency)")
}
ensure_columns(conn)
ensure_columns(conn) # idempotent
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0])
assert conn.execute(
"SELECT inherited_from FROM proficiency WHERE model_id='kimi-k3'"
).fetchone()["inherited_from"] is None
def test_migration_unfreezes_variants_that_predate_the_column(tmp_path):
"""The migration must ship the repair, not just the fix.
ADD COLUMN gives every existing row NULL, and NULL means "measured here" --
so without the backfill, the rows this whole change exists for would stay
frozen and the live flex rows would never catch up.
"""
from config import load_config
from proficiency_store import add_self_eval, ensure_columns, propagate_to_variants
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
# An old database: both rows carry samples, neither records provenance.
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [0.957])
add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "docs_writing", [0.85])
conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from")
ensure_columns(conn)
assert conn.execute(
"SELECT inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'"
).fetchone()["inherited_from"] == "kimi-k3"
# ...so the next run actually moves it.
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
assert conn.execute(
"SELECT blended_score FROM proficiency WHERE model_id='kimi-k3-flex'"
).fetchone()["blended_score"] == pytest.approx(0.957)
def test_the_backfill_never_touches_a_standard_row(tmp_path):
"""Only flex rows with a standard equivalent are provably inherited."""
from config import load_config
from proficiency_store import add_self_eval, ensure_columns
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-fast", "kimi-k3", "standard", "reduced", "full"),
])
add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0])
add_self_eval(conn, cfg, "kimi-k3-fast", "nw", "docs_writing", [0.888])
conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from")
ensure_columns(conn)
for model_id in ("kimi-k3", "kimi-k3-fast"):
assert conn.execute(
"SELECT inherited_from FROM proficiency WHERE model_id=?", (model_id,)
).fetchone()["inherited_from"] is None
def test_apply_priors_recomputes_each_touched_category(tmp_path):
"""A leaderboard import must run category-wide recompute for every touched category."""
from unittest import mock
from config import load_config
from leaderboard import apply_priors
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("qwen-3", "qwen-3", "standard", "default", "full"),
], provider="neuralwatt")
families = {"kimi-k3": ["kimi-k3"], "qwen-3": ["qwen-3"]}
priors = {
"kimi-k3": {"coding_general": 0.9, "docs_writing": 0.8},
"qwen-3": {"coding_general": 0.7},
}
with mock.patch("leaderboard.recompute_category") as mock_recompute:
written = apply_priors(conn, cfg, families, priors, "neuralwatt")
assert written == 3
touched = {call.args[2] for call in mock_recompute.call_args_list}
assert touched == {"coding_general", "docs_writing"}
def test_apply_priors_blends_leaderboard_scores(tmp_path):
"""A leaderboard import writes the prior as the blended score/source."""
from config import load_config
from leaderboard import apply_priors
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
], provider="neuralwatt")
families = {"kimi-k3": ["kimi-k3"]}
priors = {"kimi-k3": {"coding_general": 0.85}}
assert apply_priors(conn, cfg, families, priors, "neuralwatt") == 1
row = conn.execute(
"SELECT blended_score, source FROM proficiency "
"WHERE model_id='kimi-k3' AND category='coding_general'"
).fetchone()
assert row["blended_score"] == pytest.approx(0.85)
assert row["source"] == "leaderboard"
def test_apply_priors_recompute_handles_empty_category_gracefully(tmp_path):
"""The final category recompute must not crash when a touched category has no rows."""
from config import load_config
from leaderboard import apply_priors
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [], provider="neuralwatt")
families: dict[str, list[str]] = {"ghost-family": ["ghost-model"]}
priors: dict[str, dict[str, float]] = {"ghost-family": {"coding_general": 0.9}}
assert apply_priors(conn, cfg, families, priors, "neuralwatt") == 0
def _row(conn, model_id):
return conn.execute(
"""
SELECT leaderboard_score, self_eval_score, self_eval_samples,
outcome_score, outcome_samples, blended_score, source,
inherited_from
FROM proficiency
WHERE model_id = ? AND provider = 'nw' AND category = 'coding_general'
""",
(model_id,),
).fetchone()
def test_set_leaderboard_preserves_outcome_evidence(tmp_path):
from config import load_config
from proficiency_store import add_outcome, set_leaderboard
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.7)
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0])
before = _row(conn, "kimi-k3")
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8)
after = _row(conn, "kimi-k3")
assert after["leaderboard_score"] == pytest.approx(0.8)
assert after["outcome_score"] == before["outcome_score"]
assert after["outcome_samples"] == before["outcome_samples"]
def test_add_self_eval_preserves_outcome_evidence(tmp_path):
from config import load_config
from proficiency_store import add_outcome, add_self_eval
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.6, 0.6])
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.8])
row = _row(conn, "kimi-k3")
assert row["self_eval_score"] == pytest.approx(0.6667, abs=1e-4)
assert row["outcome_score"] == 1.0
assert row["outcome_samples"] == 1
def test_add_outcome_accumulates_and_recomputes_category(tmp_path):
from config import load_config
from proficiency_store import add_outcome, set_leaderboard
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("qwen-3", "qwen-3", "standard", "default", "full"),
])
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.9)
set_leaderboard(conn, cfg, "qwen-3", "nw", "coding_general", 0.9)
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0] * 5)
add_outcome(conn, cfg, "qwen-3", "nw", "coding_general", [0.0] * 5)
strong = _row(conn, "kimi-k3")
weak = _row(conn, "qwen-3")
assert strong["outcome_score"] == pytest.approx(1.0)
assert weak["outcome_score"] == pytest.approx(0.0)
assert strong["blended_score"] > weak["blended_score"]
assert strong["source"] == "outcome_blended"
assert weak["source"] == "outcome_blended"
def test_recompute_category_is_idempotent(tmp_path):
from config import load_config
from proficiency_store import add_outcome, recompute_category, set_leaderboard
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")])
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8)
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 0.5])
first = _row(conn, "kimi-k3")
recompute_category(conn, cfg, "coding_general")
second = _row(conn, "kimi-k3")
assert first["blended_score"] == pytest.approx(second["blended_score"])
assert first["source"] == second["source"]
def test_propagate_to_variants_skips_variant_with_own_outcomes(tmp_path):
from config import load_config
from proficiency_store import add_outcome, add_self_eval, propagate_to_variants
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
add_outcome(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.0])
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0
variant = _row(conn, "kimi-k3-flex")
assert variant["outcome_score"] == pytest.approx(0.0)
assert variant["outcome_samples"] == 1
def test_propagate_to_variants_copies_outcomes_to_blank_variant(tmp_path):
from config import load_config
from proficiency_store import add_outcome, add_self_eval, propagate_to_variants
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0])
assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1
variant = _row(conn, "kimi-k3-flex")
assert variant["outcome_score"] == pytest.approx(1.0)
assert variant["outcome_samples"] == 1
assert variant["inherited_from"] == "kimi-k3"
def test_add_outcome_commits_category_recompute_atomically(tmp_path):
import sqlite3
import threading
from unittest import mock
from config import load_config
from proficiency_outcome import _write as real_write
from proficiency_store import add_outcome, set_leaderboard
cfg = load_config("config/config.yaml")
setup = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("qwen-3", "qwen-3", "standard", "default", "full"),
])
db_path = tmp_path / "t.db"
set_leaderboard(setup, cfg, "kimi-k3", "nw", "coding_general", 0.9)
set_leaderboard(setup, cfg, "qwen-3", "nw", "coding_general", 0.9)
pre = setup.execute(
"""
SELECT model_id, source, blended_score FROM proficiency
WHERE category = 'coding_general'
ORDER BY model_id
"""
).fetchall()
setup.close()
ready = threading.Event()
resume = threading.Event()
recompute_writes = [0]
def _pausing_write(*args, **kwargs):
if kwargs.get("_blended_override"):
recompute_writes[0] += 1
if recompute_writes[0] == 1:
ready.set()
resume.wait(timeout=5.0)
return real_write(*args, **kwargs)
def _target():
conn = sqlite3.connect(db_path)
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0] * 5)
conn.close()
with mock.patch("proficiency_outcome._write", side_effect=_pausing_write):
runner = threading.Thread(target=_target)
runner.start()
ready.wait(timeout=5.0)
reader = sqlite3.connect(db_path)
reader.row_factory = sqlite3.Row
during = reader.execute(
"""
SELECT model_id, source, blended_score FROM proficiency
WHERE category = 'coding_general'
ORDER BY model_id
"""
).fetchall()
reader.close()
assert [(r["model_id"], r["source"], r["blended_score"]) for r in during] == [
(r["model_id"], r["source"], r["blended_score"]) for r in pre
]
resume.set()
runner.join(timeout=5.0)
final = sqlite3.connect(db_path)
final.row_factory = sqlite3.Row
rows = final.execute(
"""
SELECT model_id, source, blended_score FROM proficiency
WHERE category = 'coding_general'
ORDER BY model_id
"""
).fetchall()
final.close()
assert {r["source"] for r in rows} == {"outcome_blended", "outcome_prior"}
assert all(r["blended_score"] == pytest.approx(1.0) for r in rows)
def test_cold_benchmark_model_does_not_outrank_proven_model_in_same_category(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("proven-mod", "proven", "standard", "default", "full"),
("cold-mod", "cold", "standard", "default", "full"),
])
from proficiency_store import set_leaderboard
set_leaderboard(conn, cfg, "proven-mod", "nw", "coding_general", 0.85)
add_outcome(conn, cfg, "proven-mod", "nw", "coding_general", [0.9, 0.9])
set_leaderboard(conn, cfg, "cold-mod", "nw", "coding_general", 0.7)
cold = _row(conn, "cold-mod")
proven = _row(conn, "proven-mod")
assert proven["source"] == "outcome_blended"
assert cold["source"] == "outcome_prior"
assert proven["blended_score"] > cold["blended_score"]
def test_cold_category_preserves_benchmark_score_and_source_verbatim(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("thin-mod", "thin", "standard", "default", "full")])
from proficiency_store import add_self_eval
add_self_eval(conn, cfg, "thin-mod", "nw", "coding_general", [0.75, 0.75, 0.75])
before = _row(conn, "thin-mod")
assert before["source"] == "self_eval_thin"
saved_score = before["blended_score"]
recompute_category(conn, cfg, "coding_general")
after = _row(conn, "thin-mod")
assert after["blended_score"] == pytest.approx(saved_score)
assert after["source"] == "self_eval_thin"
def test_taxonomy_cold_category_returns_benchmark_source_verbatim(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("cold-cat", "cold", "standard", "default", "full")])
from proficiency_store import add_self_eval
add_self_eval(conn, cfg, "cold-cat", "nw", "coding_general", [0.6])
row = _row(conn, "cold-cat")
assert row["source"] == "self_eval_thin"
def test_taxonomy_trafficked_no_per_model_returns_outcome_prior(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("target", "tgt", "standard", "default", "full"),
("peer", "peer", "standard", "default", "full"),
])
from proficiency_store import add_self_eval, set_leaderboard
add_self_eval(conn, cfg, "peer", "nw", "coding_general", [0.8])
add_outcome(conn, cfg, "peer", "nw", "coding_general", [1.0, 1.0])
set_leaderboard(conn, cfg, "target", "nw", "coding_general", 0.7)
recompute_category(conn, cfg, "coding_general")
target = _row(conn, "target")
assert target["source"] == "outcome_prior"
def test_taxonomy_trafficked_with_per_model_returns_outcome_blended(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [("own", "own", "standard", "default", "full")])
from proficiency_store import set_leaderboard
set_leaderboard(conn, cfg, "own", "nw", "coding_general", 0.7)
add_outcome(conn, cfg, "own", "nw", "coding_general", [0.8, 0.9])
row = _row(conn, "own")
assert row["source"] == "outcome_blended"
def test_unmeasured_model_in_trafficked_category_borrows_peer_rate(tmp_path):
"""A model with no benchmark/outcome still gets calibrated when peers exist."""
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("ghost", "ghost", "standard", "default", "full"),
("peer", "peer", "standard", "default", "full"),
])
from proficiency_store import add_outcome, add_self_eval, set_leaderboard
set_leaderboard(conn, cfg, "ghost", "nw", "coding_general", None)
add_self_eval(conn, cfg, "peer", "nw", "coding_general", [0.9])
add_outcome(conn, cfg, "peer", "nw", "coding_general", [1.0])
recompute_category(conn, cfg, "coding_general")
row = _row(conn, "ghost")
assert row["source"] == "outcome_prior"
assert row["blended_score"] == pytest.approx(1.0)
def test_recompute_category_is_idempotent_two_models(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("alpha", "alpha", "standard", "default", "full"),
("beta", "beta", "standard", "default", "full"),
])
from proficiency_store import add_outcome, set_leaderboard
set_leaderboard(conn, cfg, "alpha", "nw", "coding_general", 0.9)
set_leaderboard(conn, cfg, "beta", "nw", "coding_general", 0.9)
add_outcome(conn, cfg, "alpha", "nw", "coding_general", [1.0, 1.0])
first_alpha = _row(conn, "alpha")
first_beta = _row(conn, "beta")
recompute_category(conn, cfg, "coding_general")
second_alpha = _row(conn, "alpha")
second_beta = _row(conn, "beta")
assert second_alpha["blended_score"] == pytest.approx(first_alpha["blended_score"])
assert second_alpha["source"] == first_alpha["source"]
assert second_beta["blended_score"] == pytest.approx(first_beta["blended_score"])
assert second_beta["source"] == first_beta["source"]
def test_recompute_category_leaves_other_categories_untouched(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("a-model", "a", "standard", "default", "full"),
("b-model", "b", "standard", "default", "full"),
])
from proficiency_store import add_self_eval
add_self_eval(conn, cfg, "a-model", "nw", "cat_a", [0.8])
add_self_eval(conn, cfg, "b-model", "nw", "cat_b", [0.6])
row_b_before = conn.execute(
"SELECT blended_score, source FROM proficiency WHERE model_id='b-model'"
).fetchone()
recompute_category(conn, cfg, "cat_a")
row_b_after = conn.execute(
"SELECT blended_score, source FROM proficiency WHERE model_id='b-model'"
).fetchone()
assert row_b_after["blended_score"] == pytest.approx(row_b_before["blended_score"])
assert row_b_after["source"] == row_b_before["source"]
def test_propagate_to_variants_preserves_outcome_columns(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
from proficiency_store import (
add_outcome,
add_self_eval,
propagate_to_variants,
set_leaderboard,
)
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8)
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 0.5])
before = _row(conn, "kimi-k3")
assert before["outcome_score"] == pytest.approx(0.75)
assert before["outcome_samples"] == 2
add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.888])
propagate_to_variants(conn, cfg, "kimi-k3", "nw")
after = _row(conn, "kimi-k3")
assert after["outcome_score"] == pytest.approx(0.75)
assert after["outcome_samples"] == 2
def test_directly_measured_variant_refuses_outcome_propagation(tmp_path):
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("kimi-k3", "kimi-k3", "standard", "default", "full"),
("kimi-k3-flex", "kimi-k3", "flex", "default", "full"),
])
from proficiency_store import (
add_outcome,
add_self_eval,
propagate_to_variants,
set_leaderboard,
)
set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.9)
add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0])
add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.5])
add_outcome(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.3])
variant_before = _row(conn, "kimi-k3-flex")
assert variant_before["outcome_score"] == pytest.approx(0.3)
assert variant_before["outcome_samples"] == 1
count = propagate_to_variants(conn, cfg, "kimi-k3", "nw")
assert count == 0
variant_after = _row(conn, "kimi-k3-flex")
assert variant_after["outcome_score"] == pytest.approx(0.3)
assert variant_after["outcome_samples"] == 1
def test_expected_success_rate_prior_pulls_towards_peer_rate():
score, source = expected_success_rate(
benchmark_score=0.90,
benchmark_source="leaderboard",
outcome_score=None,
outcome_samples=0,
peer_rate=0.40,
peer_benchmark=0.70,
prior_strength=20,
)
assert source == "outcome_prior"
assert score is not None
assert score < 0.90
def test_expected_success_rate_prior_falls_back_to_peer_rate_when_peer_benchmark_missing():
score, source = expected_success_rate(
benchmark_score=0.60,
benchmark_source="leaderboard",
outcome_score=None,
outcome_samples=0,
peer_rate=0.50,
peer_benchmark=None,
prior_strength=20,
)
assert source == "outcome_prior"
assert score == pytest.approx(0.50)
def test_recompute_category_uses_pooled_peer_rate(tmp_path):
"""Peer rate is sample-weighted Σ(n·rate)/Σn, not unweighted mean of rates.
Pooled: (1×1.0 + 95×0.81) / (1+95) = 77.95/96 ≈ 0.812
Unweighted: (1.0 + 0.81) / 2 = 0.905
Model A: 1 sample at 1.0. Model B: 95 samples at 0.81.
Model C: no outcomes, no benchmark — so prior = peer_rate directly.
Model C's blended_score must be ~0.812 (pooled), not ~0.905 (unweighted).
"""
from config import load_config
from proficiency_store import add_outcome, set_leaderboard
cfg = load_config("config/config.yaml")
conn = _models_db(tmp_path, [
("model-a", "a", "standard", "default", "full"),
("model-b", "b", "standard", "default", "full"),
("model-c", "c", "standard", "default", "full"),
])
# Model C gets a row with no benchmark so its prior = peer_rate directly.
set_leaderboard(conn, cfg, "model-c", "nw", "coding_general", None)
# Model A: 1 sample at 1.0
add_outcome(conn, cfg, "model-a", "nw", "coding_general", [1.0])
# Model B: 95 samples at 0.81
add_outcome(conn, cfg, "model-b", "nw", "coding_general", [0.81] * 95)
# After the second add_outcome, recompute_category runs over all three:
# peer_rate is now pooled from A (1×1.0) + B (95×0.81).
c_row = _row(conn, "model-c")
expected_pooled = (1.0 + 95 * 0.81) / (1 + 95) # ≈ 0.812
assert c_row["source"] == "outcome_prior"
assert c_row["blended_score"] == pytest.approx(expected_pooled, rel=1e-3)
# Confirm it's NOT the unweighted result
unweighted = (1.0 + 0.81) / 2 # ≈ 0.905
assert c_row["blended_score"] < unweighted - 0.09