204 lines
8.0 KiB
Python
204 lines
8.0 KiB
Python
"""Pure proficiency blending for the local LLM model router.
|
||
|
||
Like ``scoring.py``, ``tiering.py`` and ``routing.py``, this module is free of
|
||
I/O: scores and thresholds come in as arguments. ``eval_proficiency.py`` owns
|
||
the DB writes and the model calls.
|
||
|
||
Two independent sources feed one number per (model, category):
|
||
|
||
- **leaderboard** — a curated prior from published benchmarks. Its job is
|
||
cold start: NeuralWatt adds models, and a newly listed one has no self-eval
|
||
history at all. Without a prior it scores the neutral 0.5 and is
|
||
indistinguishable from a model that was measured and found average.
|
||
- **self-eval** — this router's own task set, run against the real endpoint.
|
||
More predictive of actual routing quality, but it accumulates slowly.
|
||
|
||
Design doc §3.3 gives the blend as ``0.3 x leaderboard + 0.7 x self_eval``
|
||
once self-eval crosses ``self_eval_min_samples``, falling back to the
|
||
leaderboard alone before that so thin, noisy self-eval data cannot dominate
|
||
early.
|
||
|
||
That rule assumes a leaderboard entry exists. It often will not — the curated
|
||
file is hand-maintained and NeuralWatt ships models faster than public
|
||
benchmarks cover them. Taken literally, a model with no prior and 9 samples
|
||
would score nothing at all, which is strictly worse than the 9 samples it
|
||
actually has. So the fallback ladder is:
|
||
|
||
leaderboard + enough self-eval -> weighted blend 'blended'
|
||
enough self-eval, no prior -> self-eval alone 'self_eval'
|
||
prior, not enough self-eval -> leaderboard alone 'leaderboard'
|
||
thin self-eval, no prior -> self-eval alone 'self_eval_thin'
|
||
neither -> None (neutral 0.5 downstream)
|
||
|
||
``self_eval_thin`` is deliberately distinguishable: it is real measurement,
|
||
but from too few samples to trust as much as the label 'self_eval' implies,
|
||
and a caller wanting to exclude it can.
|
||
|
||
---
|
||
|
||
### Empirical-Bayes outcome conversion (§$137.76 gate)
|
||
|
||
The ``expected_success_rate`` function converts any proficiency source score
|
||
into an **expected pass rate on real traffic** — a calibrated, peer-relative
|
||
signal that respects the original source's provenance:
|
||
|
||
benchmark alone, no category traffic -> verbatim benchmark 'blended'|'self_eval'|'leaderboard'|'self_eval_thin'
|
||
category traffic, no per-model data -> benchmark shrunk 'outcome_prior'
|
||
category + per-model traffic -> Bayesian blend 'outcome_blended'
|
||
|
||
Each result carries a ``Source`` tag (LOCK the branch) so downstream routing
|
||
can distinguish how much evidence came from external benchmarks vs real
|
||
verification outcomes. This is the keystone of the $137.76 gate: a model
|
||
that looks good only in benchmarks (no outcome traffic) gets flagged with
|
||
``outcome_prior``, while a model verified by clients gets ``outcome_blended``.
|
||
When neither exists, ``None`` leaves the candidate neutral.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from typing import Literal, Optional
|
||
|
||
Source = Literal[
|
||
"blended",
|
||
"self_eval",
|
||
"leaderboard",
|
||
"self_eval_thin",
|
||
"outcome_prior",
|
||
"outcome_blended",
|
||
]
|
||
|
||
|
||
def blend(
|
||
leaderboard_score: Optional[float],
|
||
self_eval_score: Optional[float],
|
||
self_eval_samples: int,
|
||
*,
|
||
leaderboard_weight: float,
|
||
self_eval_weight: float,
|
||
min_samples: int,
|
||
) -> tuple[Optional[float], Optional[Source]]:
|
||
"""Combine the two sources into one score, with the source that produced it.
|
||
|
||
Returns ``(None, None)`` when neither source has anything, which leaves
|
||
the candidate on the neutral 0.5 in ``scoring.proficiency_score`` rather
|
||
than penalizing it for being unmeasured.
|
||
"""
|
||
has_leaderboard = leaderboard_score is not None
|
||
has_self_eval = self_eval_score is not None and self_eval_samples > 0
|
||
enough_samples = has_self_eval and self_eval_samples >= min_samples
|
||
|
||
if has_leaderboard and enough_samples:
|
||
return (
|
||
leaderboard_weight * leaderboard_score
|
||
+ self_eval_weight * self_eval_score,
|
||
"blended",
|
||
)
|
||
if enough_samples:
|
||
return self_eval_score, "self_eval"
|
||
if has_leaderboard:
|
||
return leaderboard_score, "leaderboard"
|
||
if has_self_eval:
|
||
return self_eval_score, "self_eval_thin"
|
||
return None, None
|
||
|
||
|
||
def accumulate(
|
||
previous_score: Optional[float],
|
||
previous_samples: int,
|
||
new_scores: list[float],
|
||
) -> tuple[Optional[float], int]:
|
||
"""Fold a fresh eval run into a running mean.
|
||
|
||
Keeps a running average rather than replacing, so ``self_eval_samples``
|
||
means what the blending rule assumes it means: how much evidence stands
|
||
behind the score. Re-running the harness therefore tightens an estimate
|
||
instead of discarding everything learned before it.
|
||
|
||
Returns ``(previous_score, previous_samples)`` unchanged when handed no
|
||
new scores, so a run where every task errored cannot quietly reset a
|
||
model's history to zero.
|
||
"""
|
||
if not new_scores:
|
||
return previous_score, previous_samples
|
||
|
||
total_samples = previous_samples + len(new_scores)
|
||
if previous_score is None or previous_samples <= 0:
|
||
return sum(new_scores) / len(new_scores), len(new_scores)
|
||
|
||
weighted = previous_score * previous_samples + sum(new_scores)
|
||
return weighted / total_samples, total_samples
|
||
|
||
|
||
def expected_success_rate(
|
||
benchmark_score: Optional[float],
|
||
benchmark_source: Optional[Source],
|
||
outcome_score: Optional[float],
|
||
outcome_samples: int,
|
||
*,
|
||
peer_rate: Optional[float], # None when the category has no traffic yet
|
||
peer_benchmark: Optional[float],
|
||
prior_strength: int,
|
||
) -> tuple[Optional[float], Optional[Source]]:
|
||
"""Return (expected_success_rate, Source) using a three-way source taxonomy.
|
||
|
||
This empirical-Bayes scoring function converts any proficiency score into an
|
||
**expected pass rate on real traffic**, calibrated against category-level
|
||
outcomes while preserving the original source's provenance.
|
||
|
||
Three branches (LOCK the logic):
|
||
|
||
1. **Neutral** — no category traffic and no benchmark → ``(None, None)``.
|
||
2. **No category traffic** — ``peer_rate is None`` but a benchmark exists →
|
||
``benchmark_score`` and ``benchmark_source`` returned verbatim.
|
||
3. **Category has traffic** — ``peer_rate is not None``:
|
||
|
||
* **No per-model outcome** (*outcome_samples == 0*):
|
||
|
||
Prior = peer_rate × (benchmark / peer_benchmark), guarding
|
||
peer_benchmark at 0/empty (falls back to peer_rate alone).
|
||
Returns ``(prior, "outcome_prior")``.
|
||
|
||
* **Per-model outcome available** (*outcome_samples > 0*):
|
||
|
||
Bayesian-weighted average: ``score = (n*rate + k*prior) / (n+k)``
|
||
where ``n = outcome_samples``, ``k = prior_strength``,
|
||
``rate = outcome_score`` (the per-sample mean).
|
||
Returns ``(score, "outcome_blended")``.
|
||
|
||
A category with exactly one trafficked model follows the general formula —
|
||
NO special case.
|
||
"""
|
||
if peer_rate is None and benchmark_score is None and outcome_score is None:
|
||
return None, None
|
||
|
||
if peer_rate is None:
|
||
return benchmark_score, benchmark_source
|
||
|
||
prior = _compute_prior(peer_rate, benchmark_score, peer_benchmark)
|
||
if outcome_samples <= 0:
|
||
return float(prior), "outcome_prior"
|
||
|
||
rate = 0.0 if outcome_score is None else outcome_score
|
||
score = (outcome_samples * rate + prior_strength * prior) / (
|
||
outcome_samples + prior_strength
|
||
)
|
||
return float(min(1.0, max(0.0, score))), "outcome_blended"
|
||
|
||
|
||
def _compute_prior(
|
||
peer_rate: float,
|
||
benchmark_score: Optional[float],
|
||
peer_benchmark: Optional[float],
|
||
) -> float:
|
||
"""Shrinkage prior: peer_rate × (benchmark / peer_benchmark).
|
||
|
||
When benchmark or peer_benchmark is missing, falls back to peer_rate alone.
|
||
When peer_benchmark is 0, also falls back to peer_rate.
|
||
Result is capped at [0.0, 1.0].
|
||
"""
|
||
if peer_benchmark and benchmark_score is not None and peer_benchmark > 0:
|
||
prior = peer_rate * (benchmark_score / peer_benchmark)
|
||
else:
|
||
prior = peer_rate
|
||
return max(0.0, min(1.0, prior))
|