"""Tests for proficiency.py — blending leaderboard priors with self-eval. The interesting cases are the ones the design doc's one-line rule does not cover: what happens when a model has no leaderboard prior (common, because NeuralWatt ships models faster than benchmarks cover them) and when a run produces no usable scores at all. """ import pytest from config import load_config from proficiency import accumulate, blend, expected_success_rate from proficiency_outcome import add_outcome, recompute_category BLEND_KW = {"leaderboard_weight": 0.3, "self_eval_weight": 0.7, "min_samples": 10} def _blend(lb, se, n, **over): return blend(lb, se, n, **{**BLEND_KW, **over}) # --- the documented rule -------------------------------------------------- def test_blends_both_sources_once_samples_suffice(): score, source = _blend(0.4, 0.9, 10) assert score == pytest.approx(0.3 * 0.4 + 0.7 * 0.9) assert source == "blended" def test_leaderboard_alone_below_the_sample_threshold(): # Thin self-eval must not dominate a published benchmark early score, source = _blend(0.4, 0.9, 9) assert (score, source) == (0.4, "leaderboard") def test_threshold_is_inclusive(): assert _blend(0.4, 0.9, 10)[1] == "blended" assert _blend(0.4, 0.9, 9)[1] == "leaderboard" # --- no leaderboard prior, which is the common case ----------------------- def test_self_eval_alone_when_no_prior_exists(): # Given: a model NeuralWatt added that no public benchmark covers yet. # Read literally, the design doc's rule would fall back to a leaderboard # score that does not exist and yield nothing — discarding real evidence. score, source = _blend(None, 0.8, 10) assert (score, source) == (0.8, "self_eval") def test_thin_self_eval_still_beats_nothing(): # Nine real samples are worse than twelve, but far better than the # neutral 0.5 a model gets for being unmeasured. Flagged so a caller can # tell it apart from a score that cleared the threshold. score, source = _blend(None, 0.8, 9) assert (score, source) == (0.8, "self_eval_thin") def test_unmeasured_model_returns_none_not_zero(): # None leaves the candidate on the neutral 0.5 downstream. Zero would # rank it below every measured model for the crime of being new. assert _blend(None, None, 0) == (None, None) def test_zero_samples_is_not_evidence(): # A score with no samples behind it is a leftover, not a measurement assert _blend(None, 0.9, 0) == (None, None) assert _blend(0.4, 0.9, 0) == (0.4, "leaderboard") def test_a_genuine_zero_score_is_kept(): # 0.0 means "measured, and it failed everything" — distinct from unmeasured score, source = _blend(None, 0.0, 12) assert (score, source) == (0.0, "self_eval") # --- accumulation --------------------------------------------------------- def test_first_run_sets_the_mean(): assert accumulate(None, 0, [1.0, 0.5, 0.0]) == (0.5, 3) def test_later_runs_tighten_rather_than_replace(): # Given: 0.8 over 10 samples, then a run of 2 perfect scores score, samples = accumulate(0.8, 10, [1.0, 1.0]) assert samples == 12 assert score == pytest.approx((0.8 * 10 + 2.0) / 12) def test_an_all_errors_run_changes_nothing(): # Given: every task errored, so there are no scores. Resetting a model's # history to zero on a bad run would silently erase months of evidence. assert accumulate(0.8, 10, []) == (0.8, 10) def test_no_history_and_no_scores_stays_empty(): assert accumulate(None, 0, []) == (None, 0) def test_stale_score_with_zero_samples_is_overwritten(): # A score with no samples behind it carries no weight in the average assert accumulate(0.9, 0, [0.1, 0.3]) == (pytest.approx(0.2), 2) # --- variant inheritance -------------------------------------------------- def _models_db(tmp_path, rows, provider="nw"): import sqlite3 from pathlib import Path schema = (Path(__file__).resolve().parent.parent / "config" / "schema.sql").read_text() conn = sqlite3.connect(tmp_path / "t.db") conn.row_factory = sqlite3.Row conn.executescript(schema) for model_id, base, latency, reasoning, ctx in rows: conn.execute( """ INSERT INTO models (model_id, provider, base_model_id, latency_class, reasoning_mode, context_variant, availability, last_updated) VALUES (?, ?, ?, ?, ?, ?, 'active', '2026-08-17T00:00:00+00:00') """, (model_id, provider, base, latency, reasoning, ctx), ) conn.commit() return conn def test_flex_inherits_from_its_standard_equivalent(tmp_path): from pathlib import Path from config import load_config from proficiency_store import add_self_eval, propagate_to_variants cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0]) assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1 got = conn.execute( "SELECT self_eval_score FROM proficiency WHERE model_id='kimi-k3-flex'" ).fetchone() assert got["self_eval_score"] == 1.0 def test_fast_does_not_inherit_reasoning_on_quality(tmp_path): # A '-fast' row runs with reasoning off or capped, so it is NOT the same # model for quality purposes. Inheriting across that would credit it with # its reasoning-enabled sibling's answers. from pathlib import Path from config import load_config from proficiency_store import add_self_eval, propagate_to_variants cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-fast", "kimi-k3", "standard", "reduced", "full"), ]) add_self_eval(conn, cfg, "kimi-k3", "nw", "reasoning_math", [1.0]) assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0 assert conn.execute( "SELECT COUNT(*) c FROM proficiency WHERE model_id='kimi-k3-fast'" ).fetchone()["c"] == 0 def test_a_directly_measured_variant_is_never_overwritten(tmp_path): from pathlib import Path from config import load_config from proficiency_store import add_self_eval, propagate_to_variants cfg = load_config(Path(__file__).resolve().parent.parent / "config" / "config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0]) add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.2]) propagate_to_variants(conn, cfg, "kimi-k3", "nw") got = conn.execute( "SELECT self_eval_score FROM proficiency WHERE model_id='kimi-k3-flex'" ).fetchone() assert got["self_eval_score"] == 0.2 def test_a_score_above_one_never_reaches_the_table(tmp_path): """The single write path is where a bad scorer gets stopped. A blended_score above 1.0 does not just misreport one row: it raises `best` in rank_candidates and shifts every other candidate's quality band. """ from config import load_config from proficiency_store import add_self_eval cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")]) add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.5, 1.0]) row = conn.execute( "SELECT self_eval_score, blended_score FROM proficiency WHERE model_id = 'kimi-k3'" ).fetchone() assert row["self_eval_score"] == 1.0 assert row["blended_score"] <= 1.0 def test_a_negative_score_is_floored_at_zero(tmp_path): from config import load_config from proficiency_store import add_self_eval cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")]) add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [-0.5, 0.5]) assert conn.execute( "SELECT self_eval_score FROM proficiency WHERE model_id = 'kimi-k3'" ).fetchone()["self_eval_score"] == 0.25 def test_an_inherited_variant_refreshes_when_its_family_is_remeasured(tmp_path): """Inheritance was a one-shot: a variant froze at its first copy, forever. The old guard skipped any row with self_eval_samples > 0, and an inherited row has samples > 0 because inheritance copies them -- so it could never tell "measured here" from "copied here" and never refreshed. Observed live: kimi-k3 reached 0.957 while kimi-k3-flex sat at 0.85 with a week-old timestamp, and the eval reported "propagated 0 inherited rows". Flex rows serve auto:batch, so those requests ranked on stale scores. """ from config import load_config from proficiency_store import add_self_eval, propagate_to_variants cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [0.85, 0.85]) assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1 inherited = conn.execute( "SELECT blended_score, inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'" ).fetchone() assert inherited["blended_score"] == pytest.approx(0.85) assert inherited["inherited_from"] == "kimi-k3" # The family learns more; the variant must follow it. add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0, 1.0, 1.0, 1.0]) assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1 refreshed = conn.execute( "SELECT blended_score, self_eval_samples FROM proficiency WHERE model_id='kimi-k3-flex'" ).fetchone() assert refreshed["blended_score"] == pytest.approx(0.95) assert refreshed["self_eval_samples"] == 6 def test_a_measured_variant_still_outranks_its_family(tmp_path): """The protection the old guard was reaching for, kept intact.""" from config import load_config from proficiency_store import add_self_eval, propagate_to_variants cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0]) add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "docs_writing", [0.2]) assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0 row = conn.execute( "SELECT blended_score, inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'" ).fetchone() assert row["blended_score"] == pytest.approx(0.2) assert row["inherited_from"] is None def test_a_database_without_the_column_is_migrated(tmp_path): """schema.sql only defines a NEW database; existing ones need the ALTER.""" from config import load_config from proficiency_store import add_self_eval, ensure_columns cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")]) conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from") assert "inherited_from" not in { r[1] for r in conn.execute("PRAGMA table_info(proficiency)") } ensure_columns(conn) ensure_columns(conn) # idempotent add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0]) assert conn.execute( "SELECT inherited_from FROM proficiency WHERE model_id='kimi-k3'" ).fetchone()["inherited_from"] is None def test_migration_unfreezes_variants_that_predate_the_column(tmp_path): """The migration must ship the repair, not just the fix. ADD COLUMN gives every existing row NULL, and NULL means "measured here" -- so without the backfill, the rows this whole change exists for would stay frozen and the live flex rows would never catch up. """ from config import load_config from proficiency_store import add_self_eval, ensure_columns, propagate_to_variants cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) # An old database: both rows carry samples, neither records provenance. add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [0.957]) add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "docs_writing", [0.85]) conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from") ensure_columns(conn) assert conn.execute( "SELECT inherited_from FROM proficiency WHERE model_id='kimi-k3-flex'" ).fetchone()["inherited_from"] == "kimi-k3" # ...so the next run actually moves it. assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1 assert conn.execute( "SELECT blended_score FROM proficiency WHERE model_id='kimi-k3-flex'" ).fetchone()["blended_score"] == pytest.approx(0.957) def test_the_backfill_never_touches_a_standard_row(tmp_path): """Only flex rows with a standard equivalent are provably inherited.""" from config import load_config from proficiency_store import add_self_eval, ensure_columns cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-fast", "kimi-k3", "standard", "reduced", "full"), ]) add_self_eval(conn, cfg, "kimi-k3", "nw", "docs_writing", [1.0]) add_self_eval(conn, cfg, "kimi-k3-fast", "nw", "docs_writing", [0.888]) conn.execute("ALTER TABLE proficiency DROP COLUMN inherited_from") ensure_columns(conn) for model_id in ("kimi-k3", "kimi-k3-fast"): assert conn.execute( "SELECT inherited_from FROM proficiency WHERE model_id=?", (model_id,) ).fetchone()["inherited_from"] is None def test_apply_priors_recomputes_each_touched_category(tmp_path): """A leaderboard import must run category-wide recompute for every touched category.""" from unittest import mock from config import load_config from leaderboard import apply_priors cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("qwen-3", "qwen-3", "standard", "default", "full"), ], provider="neuralwatt") families = {"kimi-k3": ["kimi-k3"], "qwen-3": ["qwen-3"]} priors = { "kimi-k3": {"coding_general": 0.9, "docs_writing": 0.8}, "qwen-3": {"coding_general": 0.7}, } with mock.patch("leaderboard.recompute_category") as mock_recompute: written = apply_priors(conn, cfg, families, priors, "neuralwatt") assert written == 3 touched = {call.args[2] for call in mock_recompute.call_args_list} assert touched == {"coding_general", "docs_writing"} def test_apply_priors_blends_leaderboard_scores(tmp_path): """A leaderboard import writes the prior as the blended score/source.""" from config import load_config from leaderboard import apply_priors cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ], provider="neuralwatt") families = {"kimi-k3": ["kimi-k3"]} priors = {"kimi-k3": {"coding_general": 0.85}} assert apply_priors(conn, cfg, families, priors, "neuralwatt") == 1 row = conn.execute( "SELECT blended_score, source FROM proficiency " "WHERE model_id='kimi-k3' AND category='coding_general'" ).fetchone() assert row["blended_score"] == pytest.approx(0.85) assert row["source"] == "leaderboard" def test_apply_priors_recompute_handles_empty_category_gracefully(tmp_path): """The final category recompute must not crash when a touched category has no rows.""" from config import load_config from leaderboard import apply_priors cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [], provider="neuralwatt") families: dict[str, list[str]] = {"ghost-family": ["ghost-model"]} priors: dict[str, dict[str, float]] = {"ghost-family": {"coding_general": 0.9}} assert apply_priors(conn, cfg, families, priors, "neuralwatt") == 0 def _row(conn, model_id): return conn.execute( """ SELECT leaderboard_score, self_eval_score, self_eval_samples, outcome_score, outcome_samples, blended_score, source, inherited_from FROM proficiency WHERE model_id = ? AND provider = 'nw' AND category = 'coding_general' """, (model_id,), ).fetchone() def test_set_leaderboard_preserves_outcome_evidence(tmp_path): from config import load_config from proficiency_store import add_outcome, set_leaderboard cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")]) set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.7) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0]) before = _row(conn, "kimi-k3") set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8) after = _row(conn, "kimi-k3") assert after["leaderboard_score"] == pytest.approx(0.8) assert after["outcome_score"] == before["outcome_score"] assert after["outcome_samples"] == before["outcome_samples"] def test_add_self_eval_preserves_outcome_evidence(tmp_path): from config import load_config from proficiency_store import add_outcome, add_self_eval cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")]) add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.6, 0.6]) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0]) add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.8]) row = _row(conn, "kimi-k3") assert row["self_eval_score"] == pytest.approx(0.6667, abs=1e-4) assert row["outcome_score"] == 1.0 assert row["outcome_samples"] == 1 def test_add_outcome_accumulates_and_recomputes_category(tmp_path): from config import load_config from proficiency_store import add_outcome, set_leaderboard cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("qwen-3", "qwen-3", "standard", "default", "full"), ]) set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.9) set_leaderboard(conn, cfg, "qwen-3", "nw", "coding_general", 0.9) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0] * 5) add_outcome(conn, cfg, "qwen-3", "nw", "coding_general", [0.0] * 5) strong = _row(conn, "kimi-k3") weak = _row(conn, "qwen-3") assert strong["outcome_score"] == pytest.approx(1.0) assert weak["outcome_score"] == pytest.approx(0.0) assert strong["blended_score"] > weak["blended_score"] assert strong["source"] == "outcome_blended" assert weak["source"] == "outcome_blended" def test_recompute_category_is_idempotent(tmp_path): from config import load_config from proficiency_store import add_outcome, recompute_category, set_leaderboard cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("kimi-k3", "kimi-k3", "standard", "default", "full")]) set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 0.5]) first = _row(conn, "kimi-k3") recompute_category(conn, cfg, "coding_general") second = _row(conn, "kimi-k3") assert first["blended_score"] == pytest.approx(second["blended_score"]) assert first["source"] == second["source"] def test_propagate_to_variants_skips_variant_with_own_outcomes(tmp_path): from config import load_config from proficiency_store import add_outcome, add_self_eval, propagate_to_variants cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0]) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0]) add_outcome(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.0]) assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 0 variant = _row(conn, "kimi-k3-flex") assert variant["outcome_score"] == pytest.approx(0.0) assert variant["outcome_samples"] == 1 def test_propagate_to_variants_copies_outcomes_to_blank_variant(tmp_path): from config import load_config from proficiency_store import add_outcome, add_self_eval, propagate_to_variants cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0]) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0]) assert propagate_to_variants(conn, cfg, "kimi-k3", "nw") == 1 variant = _row(conn, "kimi-k3-flex") assert variant["outcome_score"] == pytest.approx(1.0) assert variant["outcome_samples"] == 1 assert variant["inherited_from"] == "kimi-k3" def test_add_outcome_commits_category_recompute_atomically(tmp_path): import sqlite3 import threading from unittest import mock from config import load_config from proficiency_outcome import _write as real_write from proficiency_store import add_outcome, set_leaderboard cfg = load_config("config/config.yaml") setup = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("qwen-3", "qwen-3", "standard", "default", "full"), ]) db_path = tmp_path / "t.db" set_leaderboard(setup, cfg, "kimi-k3", "nw", "coding_general", 0.9) set_leaderboard(setup, cfg, "qwen-3", "nw", "coding_general", 0.9) pre = setup.execute( """ SELECT model_id, source, blended_score FROM proficiency WHERE category = 'coding_general' ORDER BY model_id """ ).fetchall() setup.close() ready = threading.Event() resume = threading.Event() recompute_writes = [0] def _pausing_write(*args, **kwargs): if kwargs.get("_blended_override"): recompute_writes[0] += 1 if recompute_writes[0] == 1: ready.set() resume.wait(timeout=5.0) return real_write(*args, **kwargs) def _target(): conn = sqlite3.connect(db_path) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0] * 5) conn.close() with mock.patch("proficiency_outcome._write", side_effect=_pausing_write): runner = threading.Thread(target=_target) runner.start() ready.wait(timeout=5.0) reader = sqlite3.connect(db_path) reader.row_factory = sqlite3.Row during = reader.execute( """ SELECT model_id, source, blended_score FROM proficiency WHERE category = 'coding_general' ORDER BY model_id """ ).fetchall() reader.close() assert [(r["model_id"], r["source"], r["blended_score"]) for r in during] == [ (r["model_id"], r["source"], r["blended_score"]) for r in pre ] resume.set() runner.join(timeout=5.0) final = sqlite3.connect(db_path) final.row_factory = sqlite3.Row rows = final.execute( """ SELECT model_id, source, blended_score FROM proficiency WHERE category = 'coding_general' ORDER BY model_id """ ).fetchall() final.close() assert {r["source"] for r in rows} == {"outcome_blended", "outcome_prior"} assert all(r["blended_score"] == pytest.approx(1.0) for r in rows) def test_cold_benchmark_model_does_not_outrank_proven_model_in_same_category(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("proven-mod", "proven", "standard", "default", "full"), ("cold-mod", "cold", "standard", "default", "full"), ]) from proficiency_store import set_leaderboard set_leaderboard(conn, cfg, "proven-mod", "nw", "coding_general", 0.85) add_outcome(conn, cfg, "proven-mod", "nw", "coding_general", [0.9, 0.9]) set_leaderboard(conn, cfg, "cold-mod", "nw", "coding_general", 0.7) cold = _row(conn, "cold-mod") proven = _row(conn, "proven-mod") assert proven["source"] == "outcome_blended" assert cold["source"] == "outcome_prior" assert proven["blended_score"] > cold["blended_score"] def test_cold_category_preserves_benchmark_score_and_source_verbatim(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("thin-mod", "thin", "standard", "default", "full")]) from proficiency_store import add_self_eval add_self_eval(conn, cfg, "thin-mod", "nw", "coding_general", [0.75, 0.75, 0.75]) before = _row(conn, "thin-mod") assert before["source"] == "self_eval_thin" saved_score = before["blended_score"] recompute_category(conn, cfg, "coding_general") after = _row(conn, "thin-mod") assert after["blended_score"] == pytest.approx(saved_score) assert after["source"] == "self_eval_thin" def test_taxonomy_cold_category_returns_benchmark_source_verbatim(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("cold-cat", "cold", "standard", "default", "full")]) from proficiency_store import add_self_eval add_self_eval(conn, cfg, "cold-cat", "nw", "coding_general", [0.6]) row = _row(conn, "cold-cat") assert row["source"] == "self_eval_thin" def test_taxonomy_trafficked_no_per_model_returns_outcome_prior(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("target", "tgt", "standard", "default", "full"), ("peer", "peer", "standard", "default", "full"), ]) from proficiency_store import add_self_eval, set_leaderboard add_self_eval(conn, cfg, "peer", "nw", "coding_general", [0.8]) add_outcome(conn, cfg, "peer", "nw", "coding_general", [1.0, 1.0]) set_leaderboard(conn, cfg, "target", "nw", "coding_general", 0.7) recompute_category(conn, cfg, "coding_general") target = _row(conn, "target") assert target["source"] == "outcome_prior" def test_taxonomy_trafficked_with_per_model_returns_outcome_blended(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [("own", "own", "standard", "default", "full")]) from proficiency_store import set_leaderboard set_leaderboard(conn, cfg, "own", "nw", "coding_general", 0.7) add_outcome(conn, cfg, "own", "nw", "coding_general", [0.8, 0.9]) row = _row(conn, "own") assert row["source"] == "outcome_blended" def test_unmeasured_model_in_trafficked_category_borrows_peer_rate(tmp_path): """A model with no benchmark/outcome still gets calibrated when peers exist.""" cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("ghost", "ghost", "standard", "default", "full"), ("peer", "peer", "standard", "default", "full"), ]) from proficiency_store import add_outcome, add_self_eval, set_leaderboard set_leaderboard(conn, cfg, "ghost", "nw", "coding_general", None) add_self_eval(conn, cfg, "peer", "nw", "coding_general", [0.9]) add_outcome(conn, cfg, "peer", "nw", "coding_general", [1.0]) recompute_category(conn, cfg, "coding_general") row = _row(conn, "ghost") assert row["source"] == "outcome_prior" assert row["blended_score"] == pytest.approx(1.0) def test_recompute_category_is_idempotent_two_models(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("alpha", "alpha", "standard", "default", "full"), ("beta", "beta", "standard", "default", "full"), ]) from proficiency_store import add_outcome, set_leaderboard set_leaderboard(conn, cfg, "alpha", "nw", "coding_general", 0.9) set_leaderboard(conn, cfg, "beta", "nw", "coding_general", 0.9) add_outcome(conn, cfg, "alpha", "nw", "coding_general", [1.0, 1.0]) first_alpha = _row(conn, "alpha") first_beta = _row(conn, "beta") recompute_category(conn, cfg, "coding_general") second_alpha = _row(conn, "alpha") second_beta = _row(conn, "beta") assert second_alpha["blended_score"] == pytest.approx(first_alpha["blended_score"]) assert second_alpha["source"] == first_alpha["source"] assert second_beta["blended_score"] == pytest.approx(first_beta["blended_score"]) assert second_beta["source"] == first_beta["source"] def test_recompute_category_leaves_other_categories_untouched(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("a-model", "a", "standard", "default", "full"), ("b-model", "b", "standard", "default", "full"), ]) from proficiency_store import add_self_eval add_self_eval(conn, cfg, "a-model", "nw", "cat_a", [0.8]) add_self_eval(conn, cfg, "b-model", "nw", "cat_b", [0.6]) row_b_before = conn.execute( "SELECT blended_score, source FROM proficiency WHERE model_id='b-model'" ).fetchone() recompute_category(conn, cfg, "cat_a") row_b_after = conn.execute( "SELECT blended_score, source FROM proficiency WHERE model_id='b-model'" ).fetchone() assert row_b_after["blended_score"] == pytest.approx(row_b_before["blended_score"]) assert row_b_after["source"] == row_b_before["source"] def test_propagate_to_variants_preserves_outcome_columns(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) from proficiency_store import ( add_outcome, add_self_eval, propagate_to_variants, set_leaderboard, ) set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.8) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 0.5]) before = _row(conn, "kimi-k3") assert before["outcome_score"] == pytest.approx(0.75) assert before["outcome_samples"] == 2 add_self_eval(conn, cfg, "kimi-k3", "nw", "coding_general", [0.888]) propagate_to_variants(conn, cfg, "kimi-k3", "nw") after = _row(conn, "kimi-k3") assert after["outcome_score"] == pytest.approx(0.75) assert after["outcome_samples"] == 2 def test_directly_measured_variant_refuses_outcome_propagation(tmp_path): cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("kimi-k3", "kimi-k3", "standard", "default", "full"), ("kimi-k3-flex", "kimi-k3", "flex", "default", "full"), ]) from proficiency_store import ( add_outcome, add_self_eval, propagate_to_variants, set_leaderboard, ) set_leaderboard(conn, cfg, "kimi-k3", "nw", "coding_general", 0.9) add_outcome(conn, cfg, "kimi-k3", "nw", "coding_general", [1.0, 1.0]) add_self_eval(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.5]) add_outcome(conn, cfg, "kimi-k3-flex", "nw", "coding_general", [0.3]) variant_before = _row(conn, "kimi-k3-flex") assert variant_before["outcome_score"] == pytest.approx(0.3) assert variant_before["outcome_samples"] == 1 count = propagate_to_variants(conn, cfg, "kimi-k3", "nw") assert count == 0 variant_after = _row(conn, "kimi-k3-flex") assert variant_after["outcome_score"] == pytest.approx(0.3) assert variant_after["outcome_samples"] == 1 def test_expected_success_rate_prior_pulls_towards_peer_rate(): score, source = expected_success_rate( benchmark_score=0.90, benchmark_source="leaderboard", outcome_score=None, outcome_samples=0, peer_rate=0.40, peer_benchmark=0.70, prior_strength=20, ) assert source == "outcome_prior" assert score is not None assert score < 0.90 def test_expected_success_rate_prior_falls_back_to_peer_rate_when_peer_benchmark_missing(): score, source = expected_success_rate( benchmark_score=0.60, benchmark_source="leaderboard", outcome_score=None, outcome_samples=0, peer_rate=0.50, peer_benchmark=None, prior_strength=20, ) assert source == "outcome_prior" assert score == pytest.approx(0.50) def test_recompute_category_uses_pooled_peer_rate(tmp_path): """Peer rate is sample-weighted Σ(n·rate)/Σn, not unweighted mean of rates. Pooled: (1×1.0 + 95×0.81) / (1+95) = 77.95/96 ≈ 0.812 Unweighted: (1.0 + 0.81) / 2 = 0.905 Model A: 1 sample at 1.0. Model B: 95 samples at 0.81. Model C: no outcomes, no benchmark — so prior = peer_rate directly. Model C's blended_score must be ~0.812 (pooled), not ~0.905 (unweighted). """ from config import load_config from proficiency_store import add_outcome, set_leaderboard cfg = load_config("config/config.yaml") conn = _models_db(tmp_path, [ ("model-a", "a", "standard", "default", "full"), ("model-b", "b", "standard", "default", "full"), ("model-c", "c", "standard", "default", "full"), ]) # Model C gets a row with no benchmark so its prior = peer_rate directly. set_leaderboard(conn, cfg, "model-c", "nw", "coding_general", None) # Model A: 1 sample at 1.0 add_outcome(conn, cfg, "model-a", "nw", "coding_general", [1.0]) # Model B: 95 samples at 0.81 add_outcome(conn, cfg, "model-b", "nw", "coding_general", [0.81] * 95) # After the second add_outcome, recompute_category runs over all three: # peer_rate is now pooled from A (1×1.0) + B (95×0.81). c_row = _row(conn, "model-c") expected_pooled = (1.0 + 95 * 0.81) / (1 + 95) # ≈ 0.812 assert c_row["source"] == "outcome_prior" assert c_row["blended_score"] == pytest.approx(expected_pooled, rel=1e-3) # Confirm it's NOT the unweighted result unweighted = (1.0 + 0.81) / 2 # ≈ 0.905 assert c_row["blended_score"] < unweighted - 0.09