Wave 5.3 — premise-expiry checks for quality_tolerance and cost-as-tiebreak #97
@@ -10,14 +10,24 @@ objective:
|
||||
# This replaced a weighted blend (cost 0.4 / eco 0.2 / proficiency 0.4).
|
||||
# Measurement killed it: turning the cost weight from 0.4 to ZERO changed
|
||||
# the winner in only 2 of 6 categories, so the blend was never steering on
|
||||
# quality — while 60% of every decision adjudicated fractions of a cent
|
||||
# (all real traffic to date totals $0.07).
|
||||
# quality — cost decided almost nothing while consuming 40% of every decision.
|
||||
#
|
||||
# The old comment cited $0.07 total spend as evidence it was "fractions of a
|
||||
# cent" — that figure expired long ago. /metrics now surfaces current all-time
|
||||
# spend via cumulative_spend_series() and warns via cumulative_spend_warnings()
|
||||
# once it has grown past the tiebreak's "harmless fractions of a cent" frame,
|
||||
# so the justification is checked live rather than embedded as a stale number.
|
||||
|
||||
# Proficiency differences smaller than this are treated as equal and the
|
||||
# cheaper model wins. This is measurement noise, not preference: scores
|
||||
# currently rest on 2-3 samples per category, so a 0.05 gap is
|
||||
# indistinguishable from sampling variation and paying for it buys noise.
|
||||
# Narrow it as samples accumulate.
|
||||
# cheaper model wins. This is measurement noise, not preference: scores once
|
||||
# rested on 2-3 samples per category, and a 0.05 gap is indistinguishable
|
||||
# from sampling variation when the evidence is that thin.
|
||||
#
|
||||
# /metrics now enforces the "narrow it" promise: once average sample depth
|
||||
# (outcome_samples + self_eval_samples) crosses the threshold in
|
||||
# proficiency_sample_depth_warnings(), the premise has expired and the warning
|
||||
# fires — so the loose tolerance is called out rather than silently prolonged.
|
||||
# That threshold defaults to 20; see objective.proficiency_depth_warn_min_samples.
|
||||
# Above this pace ratio, the dashboard raises an alarm.
|
||||
# 1.25 = burning 25% faster than the plan allows.
|
||||
plan_pace_warn_ratio: 1.25
|
||||
@@ -139,6 +149,23 @@ objective:
|
||||
# would let a single request's luck read as a measurement.
|
||||
cache_rate_warn_min_observations: 25
|
||||
|
||||
# Premise-expiry check for quality_tolerance's "2-3 per category" claim.
|
||||
# /metrics warns when average proficiency sample depth exceeds this — the
|
||||
# point where the old thin-data justification is clearly obsolete and the
|
||||
# loose tolerance should be narrowed.
|
||||
proficiency_depth_warn_min_samples: 20
|
||||
# Minimum proficiency rows before the depth check fires. Prevents a warning
|
||||
# from a near-empty table (novelty-or-rate convention).
|
||||
proficiency_depth_warn_min_rows: 10
|
||||
|
||||
# Premise-expiry check for the cost-as-tiebreak justification.
|
||||
# /metrics warns when all-time SUM(cost_usd) exceeds this — the point where
|
||||
# the "fractions of a cent / $0.07" claim is no longer the honest frame.
|
||||
cumulative_spend_warn_usd: 50.0
|
||||
# Minimum priced rows before the spend check fires. Prevents a warning from
|
||||
# a DB with no priced rows.
|
||||
cumulative_spend_warn_min_rows: 10
|
||||
|
||||
# --- Report-only measurement series. Neither is read by routing. ---------
|
||||
#
|
||||
# Cost-estimator calibration: routing.estimated_cost's prediction against
|
||||
@@ -435,10 +462,11 @@ pinch:
|
||||
# round-trip of 1.4-2.0 s, plus ~10 bytes per decision row.
|
||||
prefix_probe: true
|
||||
relevance:
|
||||
# Off by default, matching every other new-and-unproven knob in this
|
||||
# project — and specifically requires pinch.enabled too, since this has no
|
||||
# effect otherwise. Ship it, watch route_decisions / pinch stats on real
|
||||
# traffic, then decide the default.
|
||||
# ON by default. The embed is now budget-gated (it only fires when the
|
||||
# conversation exceeds pinch.budget_tokens and pruning would actually
|
||||
# happen), so the relevance path is safe to leave on. Still requires
|
||||
# pinch.enabled (no effect otherwise) and still needs nomic-embed-text on
|
||||
# the configured Ollama (pull: ollama pull nomic-embed-text).
|
||||
enabled: true
|
||||
# An EMBEDDING model, not a chat model — this must not point at
|
||||
# classifier.model or verification.model. Pull one on the same Ollama:
|
||||
|
||||
@@ -125,6 +125,15 @@ class Objective(StrictModel):
|
||||
cache_rate_warn_margin: Optional[float] = None
|
||||
cache_rate_warn_min_observations: Optional[int] = None
|
||||
|
||||
# Premise-expiry checks for the quality_tolerance and cost-as-tiebreak
|
||||
# justifications in config.yaml. These are /metrics warning thresholds,
|
||||
# not dispatch controls — they fire once the comment's premise is clearly
|
||||
# expired, so the operator knows to re-read that justification.
|
||||
proficiency_depth_warn_min_samples: Optional[int] = None
|
||||
proficiency_depth_warn_min_rows: Optional[int] = None
|
||||
cumulative_spend_warn_usd: Optional[float] = None
|
||||
cumulative_spend_warn_min_rows: Optional[int] = None
|
||||
|
||||
# Report-only measurement windows. Neither series is read by routing;
|
||||
# see metrics.cost_estimate_calibration / metrics.latency_series.
|
||||
cost_calibration_window_hours: Optional[int] = None
|
||||
@@ -172,6 +181,43 @@ class Objective(StrictModel):
|
||||
"objective.billing_reset_day must be in 1..28, or null to disable"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("proficiency_depth_warn_min_samples")
|
||||
@classmethod
|
||||
def depth_min_samples_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||
if v is not None and v <= 0:
|
||||
raise ValueError(
|
||||
"objective.proficiency_depth_warn_min_samples must be > 0"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("proficiency_depth_warn_min_rows")
|
||||
@classmethod
|
||||
def depth_min_rows_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||
if v is not None and v <= 0:
|
||||
raise ValueError(
|
||||
"objective.proficiency_depth_warn_min_rows must be > 0"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("cumulative_spend_warn_usd")
|
||||
@classmethod
|
||||
def spend_warn_usd_positive(cls, v: Optional[float]) -> Optional[float]:
|
||||
if v is not None and v <= 0:
|
||||
raise ValueError(
|
||||
"objective.cumulative_spend_warn_usd must be > 0"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("cumulative_spend_warn_min_rows")
|
||||
@classmethod
|
||||
def spend_min_rows_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||
if v is not None and v <= 0:
|
||||
raise ValueError(
|
||||
"objective.cumulative_spend_warn_min_rows must be > 0"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("max_energy_per_request")
|
||||
@classmethod
|
||||
def ceiling_positive(cls, v: Optional[float]) -> Optional[float]:
|
||||
|
||||
274
src/metrics.py
274
src/metrics.py
@@ -1348,6 +1348,14 @@ def scoring_coverage(
|
||||
cache = cache_rate_series(conn, cfg)
|
||||
warnings.extend(cache_rate_warnings(conn, cfg, cache))
|
||||
|
||||
# Premise-expiry checks: computed once and passed in, same pattern as the
|
||||
# cache-rate warning so /metrics shows one consistent reading.
|
||||
depth = proficiency_sample_depth_series(conn, cfg)
|
||||
warnings.extend(proficiency_sample_depth_warnings(conn, cfg, depth))
|
||||
|
||||
spend = cumulative_spend_series(conn, cfg)
|
||||
warnings.extend(cumulative_spend_warnings(conn, cfg, spend))
|
||||
|
||||
selection = selection_coverage(conn, cfg)
|
||||
warnings.extend(selection_coverage_warnings(selection))
|
||||
|
||||
@@ -1363,6 +1371,11 @@ def scoring_coverage(
|
||||
# its payload in dispatcher.py, and this is a telemetry-coverage
|
||||
# question anyway: which routes can even be measured, and at what rate.
|
||||
"cache": cache,
|
||||
# Premise-expiry series: same pattern, another telemetry-coverage
|
||||
# reading — the operator needs the number that triggered the warning,
|
||||
# not only the warning text.
|
||||
"proficiency_sample_depth": depth,
|
||||
"cumulative_spend": spend,
|
||||
}
|
||||
|
||||
|
||||
@@ -2151,6 +2164,14 @@ _CACHE_RATE_WINDOW_HOURS: Final = 168
|
||||
_CACHE_RATE_WARN_MARGIN: Final = 0.10
|
||||
_CACHE_RATE_MIN_OBSERVATIONS: Final = 25
|
||||
|
||||
# Defaults for the premise-expiry checks, matching the pattern above. These
|
||||
# live in config.yaml under objective:; these Final values exist so a
|
||||
# SimpleNamespace or an older config still produces a sane series.
|
||||
_PRF_DEPTH_WARN_MIN_SAMPLES: Final = 20
|
||||
_PRF_DEPTH_WARN_MIN_ROWS: Final = 10
|
||||
_CUMULATIVE_SPEND_WARN_USD: Final = 50.0
|
||||
_CUMULATIVE_SPEND_WARN_MIN_ROWS: Final = 10
|
||||
|
||||
|
||||
def _has_column(conn: sqlite3.Connection, table: str, column: str) -> bool:
|
||||
"""True when *table* carries *column*, for additively-migrated schemas.
|
||||
@@ -2352,6 +2373,259 @@ def cache_rate_warnings(
|
||||
return warnings
|
||||
|
||||
|
||||
def proficiency_sample_depth_series(
|
||||
conn: sqlite3.Connection, cfg: Any
|
||||
) -> dict:
|
||||
"""Average sample depth (outcome_samples + self_eval_samples) per category.
|
||||
|
||||
Draws from the full proficiency table (cumulative, no trailing window). The
|
||||
depth per row follows ``proficiency_matrix`` at the top of this file:
|
||||
``samples = (self_eval_samples or 0) + (outcome_samples or 0)``.
|
||||
|
||||
Returns overall and per-category averages, plus thin row counts.
|
||||
"""
|
||||
min_samples = getattr(
|
||||
cfg.objective, "proficiency_depth_warn_min_samples", None
|
||||
)
|
||||
if min_samples is None:
|
||||
min_samples = _PRF_DEPTH_WARN_MIN_SAMPLES
|
||||
|
||||
series: dict = {
|
||||
"overall_avg_depth": None,
|
||||
"total_rows": 0,
|
||||
"thin_rows": 0,
|
||||
"by_category": [],
|
||||
}
|
||||
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT category,
|
||||
COUNT(*) AS cnt,
|
||||
AVG(COALESCE(self_eval_samples, 0)
|
||||
+ COALESCE(outcome_samples, 0)) AS avg_depth,
|
||||
SUM(CASE WHEN COALESCE(self_eval_samples, 0)
|
||||
+ COALESCE(outcome_samples, 0) < ? THEN 1 ELSE 0 END)
|
||||
AS thin
|
||||
FROM proficiency
|
||||
GROUP BY category
|
||||
""",
|
||||
(min_samples,),
|
||||
).fetchall()
|
||||
|
||||
total_rows = 0
|
||||
total_sampled_rows = 0
|
||||
depth_sum = 0.0
|
||||
thin_total = 0
|
||||
by_cat: list[dict] = []
|
||||
for row in rows:
|
||||
cnt = row["cnt"] or 0
|
||||
avg = row["avg_depth"]
|
||||
thin = row["thin"] or 0
|
||||
total_rows += cnt
|
||||
thin_total += thin
|
||||
by_cat.append(
|
||||
{
|
||||
"category": row["category"],
|
||||
"rows": cnt,
|
||||
"avg_depth": avg,
|
||||
"thin": thin,
|
||||
}
|
||||
)
|
||||
if avg is not None:
|
||||
total_sampled_rows += cnt
|
||||
depth_sum += avg * cnt
|
||||
|
||||
by_cat.sort(key=lambda d: d["category"])
|
||||
series["total_rows"] = total_rows
|
||||
series["thin_rows"] = thin_total
|
||||
series["by_category"] = by_cat
|
||||
series["overall_avg_depth"] = (
|
||||
(depth_sum / total_sampled_rows) if total_sampled_rows else None
|
||||
)
|
||||
return series
|
||||
|
||||
|
||||
def proficiency_sample_depth_warnings(
|
||||
conn: sqlite3.Connection,
|
||||
cfg: Any,
|
||||
series: Optional[dict] = None,
|
||||
) -> List[str]:
|
||||
"""Expiry check on ``objective.quality_tolerance``'s thin-data justification.
|
||||
|
||||
The config.yaml comment on ``quality_tolerance`` says "scores rest on 2-3
|
||||
samples per category... Narrow it as samples accumulate." Once average
|
||||
sample depth has grown past ``objective.proficiency_depth_warn_min_samples``
|
||||
(default 20), that justification is clearly obsolete and /metrics warns that
|
||||
the premise has expired.
|
||||
|
||||
Respects a minimum-observation floor
|
||||
(``objective.proficiency_depth_warn_min_rows``) so a near-empty table does
|
||||
not fire.
|
||||
"""
|
||||
if series is None:
|
||||
series = proficiency_sample_depth_series(conn, cfg)
|
||||
|
||||
min_rows = getattr(cfg.objective, "proficiency_depth_warn_min_rows", None)
|
||||
if min_rows is None:
|
||||
min_rows = _PRF_DEPTH_WARN_MIN_ROWS
|
||||
min_samples = getattr(
|
||||
cfg.objective, "proficiency_depth_warn_min_samples", None
|
||||
)
|
||||
if min_samples is None:
|
||||
min_samples = _PRF_DEPTH_WARN_MIN_SAMPLES
|
||||
|
||||
warnings: List[str] = []
|
||||
|
||||
total_rows = series.get("total_rows") or 0
|
||||
avg_depth = series.get("overall_avg_depth")
|
||||
if avg_depth is None or total_rows < min_rows:
|
||||
return warnings
|
||||
|
||||
if avg_depth > min_samples:
|
||||
warnings.append(
|
||||
f"quality_tolerance premise expired: average proficiency sample "
|
||||
f"depth {avg_depth:.1f} across {total_rows} rows has passed the "
|
||||
f"warning threshold ({min_samples}), so the '2-3 samples per "
|
||||
f"category' justification for objective.quality_tolerance={getattr(cfg.objective, 'quality_tolerance', '?')} "
|
||||
f"is obsolete — the loose tolerance can be narrowed."
|
||||
)
|
||||
|
||||
for entry in series.get("by_category") or []:
|
||||
cat_avg = entry.get("avg_depth")
|
||||
cat_rows = entry.get("rows") or 0
|
||||
if cat_avg is None or cat_rows < min_rows:
|
||||
continue
|
||||
if cat_avg > min_samples:
|
||||
warnings.append(
|
||||
f"quality_tolerance premise expired: category "
|
||||
f"{entry['category']} average depth {cat_avg:.1f} over "
|
||||
f"{cat_rows} rows has passed the warning threshold "
|
||||
f"({min_samples}) — objective.quality_tolerance premise "
|
||||
f"is obsolete for this category."
|
||||
)
|
||||
|
||||
return warnings
|
||||
|
||||
|
||||
def cumulative_spend_series(conn: sqlite3.Connection, cfg: Any) -> dict:
|
||||
"""All-time SUM(cost_usd) from energy_observations, excluding seed_reference.
|
||||
|
||||
Mirrors the cache-rate series' exclusion pattern: ``task_category !=
|
||||
'seed_reference'`` keeps reference sweeps from inflating "real traffic".
|
||||
Also surfaces per-provider breakdown and row counts. NULL cost_usd is
|
||||
treated as 0.
|
||||
"""
|
||||
series: dict = {
|
||||
"total_spend_usd": 0.0,
|
||||
"priced_rows": 0,
|
||||
"total_rows": 0,
|
||||
"by_provider": [],
|
||||
}
|
||||
|
||||
total_row = conn.execute(
|
||||
"""
|
||||
SELECT COUNT(*) AS total_rows,
|
||||
COUNT(cost_usd) AS priced_rows,
|
||||
COALESCE(SUM(cost_usd), 0) AS total_spend
|
||||
FROM energy_observations
|
||||
WHERE task_category IS NULL OR task_category != ?
|
||||
""",
|
||||
(SEED_CATEGORY,),
|
||||
).fetchone()
|
||||
total_rows = total_row["total_rows"] or 0
|
||||
priced_rows = total_row["priced_rows"] or 0
|
||||
total_spend = float(total_row["total_spend"] or 0.0)
|
||||
|
||||
provider_rows = conn.execute(
|
||||
"""
|
||||
SELECT provider,
|
||||
COUNT(*) AS total_rows,
|
||||
COUNT(cost_usd) AS priced_rows,
|
||||
COALESCE(SUM(cost_usd), 0) AS provider_spend
|
||||
FROM energy_observations
|
||||
WHERE task_category IS NULL OR task_category != ?
|
||||
GROUP BY provider
|
||||
ORDER BY provider
|
||||
""",
|
||||
(SEED_CATEGORY,),
|
||||
).fetchall()
|
||||
|
||||
series["total_spend_usd"] = round(total_spend, 4)
|
||||
series["priced_rows"] = priced_rows
|
||||
series["total_rows"] = total_rows
|
||||
|
||||
by_provider: list[dict] = []
|
||||
for row in provider_rows:
|
||||
by_provider.append(
|
||||
{
|
||||
"provider": row["provider"],
|
||||
"total_rows": row["total_rows"] or 0,
|
||||
"priced_rows": row["priced_rows"] or 0,
|
||||
"provider_spend_usd": round(
|
||||
float(row["provider_spend"] or 0.0), 4
|
||||
),
|
||||
}
|
||||
)
|
||||
series["by_provider"] = by_provider
|
||||
return series
|
||||
|
||||
|
||||
def cumulative_spend_warnings(
|
||||
conn: sqlite3.Connection,
|
||||
cfg: Any,
|
||||
series: Optional[dict] = None,
|
||||
) -> List[str]:
|
||||
"""Expiry check on the cost-as-tiebreak comment's $0.07 claim.
|
||||
|
||||
The config.yaml comment on the blend section says "60% of every decision
|
||||
adjudicated fractions of a cent (all real traffic to date totals $0.07)".
|
||||
Once all-time spend exceeds ``objective.cumulative_spend_warn_usd``
|
||||
(default $50.00), that justification no longer matches reality. Respects a
|
||||
minimum-observation floor so a DB with no priced rows does not fire.
|
||||
"""
|
||||
if series is None:
|
||||
series = cumulative_spend_series(conn, cfg)
|
||||
|
||||
warn_usd = getattr(cfg.objective, "cumulative_spend_warn_usd", None)
|
||||
if warn_usd is None:
|
||||
warn_usd = _CUMULATIVE_SPEND_WARN_USD
|
||||
min_rows = getattr(cfg.objective, "cumulative_spend_warn_min_rows", None)
|
||||
if min_rows is None:
|
||||
min_rows = _CUMULATIVE_SPEND_WARN_MIN_ROWS
|
||||
|
||||
warnings: List[str] = []
|
||||
|
||||
priced_rows = series.get("priced_rows") or 0
|
||||
total_spend = series.get("total_spend_usd") or 0.0
|
||||
|
||||
if priced_rows < min_rows:
|
||||
return warnings
|
||||
|
||||
if total_spend > warn_usd:
|
||||
warnings.append(
|
||||
f"cost-as-tiebreak premise expired: cumulative spend "
|
||||
f"${total_spend:.2f} across {priced_rows} priced rows has "
|
||||
f"passed the warning threshold (${warn_usd:.2f}), so the "
|
||||
f"blend comment's '$0.07 / fractions of a cent' justification "
|
||||
f"in config.yaml no longer matches reality — it is off by "
|
||||
f"{total_spend / 0.07:.0f}x."
|
||||
)
|
||||
|
||||
for entry in series.get("by_provider") or []:
|
||||
provider_spend = entry.get("provider_spend_usd") or 0.0
|
||||
provider_priced = entry.get("priced_rows") or 0
|
||||
if provider_spend > warn_usd and provider_priced >= min_rows:
|
||||
warnings.append(
|
||||
f"cost-as-tiebreak premise expired: provider "
|
||||
f"{entry['provider']} spend ${provider_spend:.2f} has "
|
||||
f"passed the warning threshold (${warn_usd:.2f}) — "
|
||||
f"the blend comment's 'fractions of a cent' frame is "
|
||||
f"stale for this provider."
|
||||
)
|
||||
|
||||
return warnings
|
||||
|
||||
|
||||
# Defaults for the two report-only series below, used when config omits the
|
||||
# keys. Same purpose as the cache-rate defaults above: a SimpleNamespace config
|
||||
# in a test, or a deployment on an older config file, still produces a series
|
||||
|
||||
@@ -280,6 +280,24 @@ DELIBERATELY_NOT_IN_ADMIN: dict[str, str] = {
|
||||
"objective.cache_rate_warn_min_observations": (
|
||||
"sample floor for the cache-rate warning; tunes a warning."
|
||||
),
|
||||
# --- premise-expiry check thresholds: they tune a /metrics warning
|
||||
# --- class, never dispatch. Same category as the cache-rate warn knobs.
|
||||
"objective.proficiency_depth_warn_min_samples": (
|
||||
"minimum average sample depth threshold for the proficiency-depth "
|
||||
"premise-expiry warning; tunes a /metrics warning, not dispatch."
|
||||
),
|
||||
"objective.proficiency_depth_warn_min_rows": (
|
||||
"observation floor for the proficiency-depth premise-expiry warning; "
|
||||
"tunes a /metrics warning, not dispatch."
|
||||
),
|
||||
"objective.cumulative_spend_warn_usd": (
|
||||
"dollar threshold for the cumulative-spend premise-expiry warning; "
|
||||
"tunes a /metrics warning, not dispatch."
|
||||
),
|
||||
"objective.cumulative_spend_warn_min_rows": (
|
||||
"priced-row floor for the cumulative-spend premise-expiry warning; "
|
||||
"tunes a /metrics warning, not dispatch."
|
||||
),
|
||||
# --- report-only series: they change what /metrics SHOWS, and not even
|
||||
# --- what it warns about. metrics.cost_estimate_calibration and
|
||||
# --- metrics.latency_series emit no warning class at all and are read by
|
||||
|
||||
@@ -36,10 +36,14 @@ from metrics import (
|
||||
capability_ceilings,
|
||||
capability_demand_warnings,
|
||||
context_ceilings,
|
||||
cumulative_spend_series,
|
||||
cumulative_spend_warnings,
|
||||
demand_ceiling_warnings,
|
||||
local_energy_summary,
|
||||
per_model,
|
||||
pinch_summary,
|
||||
proficiency_sample_depth_series,
|
||||
proficiency_sample_depth_warnings,
|
||||
quota_accounts,
|
||||
recent_decisions,
|
||||
rejection_warnings,
|
||||
@@ -2135,3 +2139,295 @@ def test_scoring_coverage_carries_the_cache_series_and_its_warnings(tmp_path):
|
||||
|
||||
assert coverage["cache"]["cache_rate"] == pytest.approx(0.4)
|
||||
assert any(w.startswith("cache rate: measured") for w in coverage["warnings"])
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# proficiency_sample_depth_series / _warnings (Wave 5.3)
|
||||
# =============================================================================
|
||||
|
||||
|
||||
def _seed_proficiency_depth_rows(
|
||||
conn: sqlite3.Connection,
|
||||
*,
|
||||
category: str = "coding_general",
|
||||
n: int,
|
||||
outcome_samples: int = 0,
|
||||
self_eval_samples: int = 0,
|
||||
model_id: str = "cheap",
|
||||
seed_models: bool = True,
|
||||
) -> None:
|
||||
"""Insert *n* proficiency rows with the given sample counts."""
|
||||
for i in range(n):
|
||||
suffix = str(i) if i else ""
|
||||
mid = model_id + suffix
|
||||
if seed_models:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT OR IGNORE INTO models (
|
||||
model_id, provider, base_model_id, tier, context_window,
|
||||
effective_context_window, max_output_tokens,
|
||||
cost_per_1m_prompt, cost_per_1m_completion,
|
||||
supports_vision, supports_json_mode,
|
||||
latency_class, reasoning_mode, context_variant,
|
||||
access_level, availability, last_updated
|
||||
) VALUES (?, 'neuralwatt', ?, 2, 262128, 192500, 16384, 0.30, 0.10,
|
||||
1, 1, 'standard', 'default', 'full', 'public', 'active',
|
||||
'2026-09-01T00:00:00+00:00')
|
||||
""",
|
||||
(mid, mid),
|
||||
)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO proficiency (
|
||||
model_id, provider, category, blended_score, source,
|
||||
self_eval_samples, outcome_samples, last_updated
|
||||
) VALUES (?, 'neuralwatt', ?, 0.9, 'self_eval_thin',
|
||||
?, ?, '2026-09-01T00:00:00+00:00')
|
||||
""",
|
||||
(mid, category, self_eval_samples, outcome_samples),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def test_proficiency_depth_series_computes_average_depth(tmp_path):
|
||||
"""The series aggregates sample depth per category and overall."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
|
||||
self_eval_samples=10, outcome_samples=5)
|
||||
_seed_proficiency_depth_rows(conn, n=2, category="debugging",
|
||||
self_eval_samples=20, outcome_samples=10)
|
||||
|
||||
series = proficiency_sample_depth_series(conn, CFG)
|
||||
|
||||
# coding_general: 3 rows, avg depth (15+15+15)/3 = 15
|
||||
# debugging: 2 rows, avg depth (30+30)/2 = 30
|
||||
# overall: (45+60)/5 = 21.0
|
||||
assert series["total_rows"] == 5
|
||||
assert series["overall_avg_depth"] == pytest.approx(21.0)
|
||||
by_cat = {e["category"]: e for e in series["by_category"]}
|
||||
assert by_cat["coding_general"]["avg_depth"] == pytest.approx(15.0)
|
||||
assert by_cat["coding_general"]["rows"] == 3
|
||||
assert by_cat["debugging"]["avg_depth"] == pytest.approx(30.0)
|
||||
assert by_cat["debugging"]["rows"] == 2
|
||||
|
||||
|
||||
def test_proficiency_depth_series_identifies_thin_rows(tmp_path):
|
||||
"""Rows below min_samples are counted as thin."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_proficiency_depth_rows(conn, n=2, category="coding_general",
|
||||
self_eval_samples=1, outcome_samples=1)
|
||||
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
|
||||
self_eval_samples=20, outcome_samples=10,
|
||||
model_id="dear")
|
||||
|
||||
series = proficiency_sample_depth_series(conn, CFG)
|
||||
|
||||
assert series["thin_rows"] == 2
|
||||
assert series["total_rows"] == 5
|
||||
|
||||
|
||||
def test_proficiency_depth_warning_silent_below_the_observation_floor(tmp_path):
|
||||
"""Never warns on a near-empty table (novelty-or-rate)."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
|
||||
self_eval_samples=5, outcome_samples=5)
|
||||
|
||||
warnings = proficiency_sample_depth_warnings(conn, CFG)
|
||||
|
||||
assert warnings == []
|
||||
|
||||
|
||||
def test_proficiency_depth_warning_silent_below_the_threshold(tmp_path):
|
||||
"""Average sample depth below min_samples does not fire."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_proficiency_depth_rows(conn, n=15, category="coding_general",
|
||||
self_eval_samples=5, outcome_samples=5)
|
||||
|
||||
cfg = _cfg_with(proficiency_depth_warn_min_samples=50)
|
||||
warnings = proficiency_sample_depth_warnings(conn, cfg)
|
||||
|
||||
assert warnings == []
|
||||
|
||||
|
||||
def test_proficiency_depth_warning_fires_when_depth_exceeds_threshold(tmp_path):
|
||||
"""Once average depth passes the configurable threshold, the premise is
|
||||
expired and the warning fires."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_proficiency_depth_rows(conn, n=20, category="coding_general",
|
||||
self_eval_samples=15, outcome_samples=10)
|
||||
|
||||
warnings = proficiency_sample_depth_warnings(conn, CFG)
|
||||
|
||||
premise = [w for w in warnings
|
||||
if w.startswith("quality_tolerance premise expired")]
|
||||
assert len(premise) >= 1
|
||||
assert "quality_tolerance" in premise[0]
|
||||
|
||||
|
||||
def test_proficiency_depth_warning_respects_the_observation_floor(tmp_path):
|
||||
"""Plenty of rows but all below the floor threshold — silent."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_proficiency_depth_rows(conn, n=5, category="coding_general",
|
||||
self_eval_samples=30, outcome_samples=30)
|
||||
|
||||
cfg = _cfg_with(proficiency_depth_warn_min_rows=10)
|
||||
warnings = proficiency_sample_depth_warnings(conn, cfg)
|
||||
|
||||
assert warnings == []
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# cumulative_spend_series / _warnings (Wave 5.3)
|
||||
# =============================================================================
|
||||
|
||||
|
||||
def _seed_spend_rows(
|
||||
conn: sqlite3.Connection,
|
||||
*,
|
||||
n: int,
|
||||
cost_usd: float = 0.0,
|
||||
model_id: str = "cheap",
|
||||
provider: str = "neuralwatt",
|
||||
task_category: str = "coding_general",
|
||||
) -> None:
|
||||
"""Insert *n* energy_observations rows with a cost_usd."""
|
||||
now = _now()
|
||||
for _ in range(n):
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO energy_observations (
|
||||
model_id, provider, task_category, prompt_tokens,
|
||||
completion_tokens, energy_kwh, cost_usd, observed_at
|
||||
) VALUES (?, ?, ?, 1000, 100, 0.001, ?, ?)
|
||||
""",
|
||||
(model_id, provider, task_category, cost_usd, now.isoformat()),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _seed_spend_seed_rows(conn: sqlite3.Connection, *, n: int = 5) -> None:
|
||||
"""Insert seed_reference energy rows (excluded from spend)."""
|
||||
now = _now()
|
||||
for _ in range(n):
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO energy_observations (
|
||||
model_id, provider, task_category, prompt_tokens,
|
||||
completion_tokens, energy_kwh, cost_usd, observed_at
|
||||
) VALUES ('cheap', 'neuralwatt', 'seed_reference', 1000, 100,
|
||||
0.001, 999.0, ?)
|
||||
""",
|
||||
(now.isoformat(),),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def test_cumulative_spend_series_aggregates_all_time_cost(tmp_path):
|
||||
"""The series sums cost_usd across all non-seed energy rows."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_spend_rows(conn, n=3, cost_usd=0.05)
|
||||
_seed_spend_rows(conn, n=2, cost_usd=0.10)
|
||||
|
||||
series = cumulative_spend_series(conn, CFG)
|
||||
|
||||
assert series["total_spend_usd"] == pytest.approx(0.35)
|
||||
assert series["priced_rows"] == 5
|
||||
assert series["total_rows"] == 5
|
||||
|
||||
|
||||
def test_cumulative_spend_series_excludes_seed_reference(tmp_path):
|
||||
"""seed_reference rows do not inflate 'real traffic' spend."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_spend_rows(conn, n=2, cost_usd=0.05)
|
||||
_seed_spend_seed_rows(conn, n=5)
|
||||
|
||||
series = cumulative_spend_series(conn, CFG)
|
||||
|
||||
assert series["total_spend_usd"] == pytest.approx(0.10)
|
||||
assert series["priced_rows"] == 2
|
||||
# total_rows is also seed-filtered: the series describes real traffic only,
|
||||
# so the excluded rows never appear in any of its counts.
|
||||
assert series["total_rows"] == 2
|
||||
|
||||
|
||||
def test_cumulative_spend_series_per_provider_breakdown(tmp_path):
|
||||
"""Results include per-provider sub-totals."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_spend_rows(conn, n=2, cost_usd=1.0, provider="neuralwatt")
|
||||
_seed_spend_rows(conn, n=3, cost_usd=2.0, provider="openrouter")
|
||||
|
||||
series = cumulative_spend_series(conn, CFG)
|
||||
|
||||
assert series["total_spend_usd"] == pytest.approx(8.0)
|
||||
by_provider = {e["provider"]: e for e in series["by_provider"]}
|
||||
assert by_provider["neuralwatt"]["provider_spend_usd"] == pytest.approx(2.0)
|
||||
assert by_provider["openrouter"]["provider_spend_usd"] == pytest.approx(6.0)
|
||||
|
||||
|
||||
def test_cumulative_spend_warning_silent_below_the_observation_floor(tmp_path):
|
||||
"""Never warns on too few priced rows (novelty-or-rate)."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_spend_rows(conn, n=3, cost_usd=100.0)
|
||||
|
||||
cfg = _cfg_with(cumulative_spend_warn_min_rows=10)
|
||||
warnings = cumulative_spend_warnings(conn, cfg)
|
||||
|
||||
assert warnings == []
|
||||
|
||||
|
||||
def test_cumulative_spend_warning_silent_below_the_threshold(tmp_path):
|
||||
"""Spend below the warning threshold does not fire."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_spend_rows(conn, n=15, cost_usd=0.01)
|
||||
|
||||
cfg = _cfg_with(cumulative_spend_warn_usd=50.0)
|
||||
warnings = cumulative_spend_warnings(conn, cfg)
|
||||
|
||||
assert warnings == []
|
||||
|
||||
|
||||
def test_cumulative_spend_warning_fires_when_premise_expires(tmp_path):
|
||||
"""Once all-time spend exceeds warn_usd, the warning fires naming the
|
||||
blend comment whose premise has expired."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_spend_rows(conn, n=15, cost_usd=50.0)
|
||||
|
||||
cfg = _cfg_with(cumulative_spend_warn_usd=10.0,
|
||||
cumulative_spend_warn_min_rows=5)
|
||||
warnings = cumulative_spend_warnings(conn, cfg)
|
||||
|
||||
premise = [w for w in warnings
|
||||
if w.startswith("cost-as-tiebreak premise expired")]
|
||||
assert len(premise) >= 1
|
||||
assert "cumulative spend" in premise[0]
|
||||
assert "$" in premise[0]
|
||||
|
||||
|
||||
|
||||
def test_scoring_coverage_carries_the_premise_expiry_series(tmp_path):
|
||||
"""scoring_coverage includes both new premise-expiry series and warnings."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
# Seed enough proficiency depth to trigger the warning
|
||||
_seed_proficiency_depth_rows(conn, n=20, category="coding_general",
|
||||
self_eval_samples=15, outcome_samples=10)
|
||||
_seed_spend_rows(conn, n=15, cost_usd=50.0)
|
||||
|
||||
cfg = _cfg_with(
|
||||
cumulative_spend_warn_usd=10.0,
|
||||
cumulative_spend_warn_min_rows=5,
|
||||
)
|
||||
coverage = scoring_coverage(conn, cfg)
|
||||
|
||||
assert "proficiency_sample_depth" in coverage
|
||||
assert coverage["proficiency_sample_depth"]["overall_avg_depth"] is not None
|
||||
assert "cumulative_spend" in coverage
|
||||
assert coverage["cumulative_spend"]["total_spend_usd"] == pytest.approx(750.0)
|
||||
assert any(
|
||||
w.startswith("quality_tolerance premise expired")
|
||||
for w in coverage["warnings"]
|
||||
)
|
||||
assert any(
|
||||
w.startswith("cost-as-tiebreak premise expired")
|
||||
for w in coverage["warnings"]
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user