Wave 5.3 — premise-expiry checks for quality_tolerance and cost-as-tiebreak #97

Merged
alee merged 1 commits from feat/wave53-premise-expiry into main 2026-09-19 15:57:02 +00:00
5 changed files with 672 additions and 10 deletions

View File

@@ -10,14 +10,24 @@ objective:
# This replaced a weighted blend (cost 0.4 / eco 0.2 / proficiency 0.4).
# Measurement killed it: turning the cost weight from 0.4 to ZERO changed
# the winner in only 2 of 6 categories, so the blend was never steering on
# quality — while 60% of every decision adjudicated fractions of a cent
# (all real traffic to date totals $0.07).
# quality — cost decided almost nothing while consuming 40% of every decision.
#
# The old comment cited $0.07 total spend as evidence it was "fractions of a
# cent" — that figure expired long ago. /metrics now surfaces current all-time
# spend via cumulative_spend_series() and warns via cumulative_spend_warnings()
# once it has grown past the tiebreak's "harmless fractions of a cent" frame,
# so the justification is checked live rather than embedded as a stale number.
# Proficiency differences smaller than this are treated as equal and the
# cheaper model wins. This is measurement noise, not preference: scores
# currently rest on 2-3 samples per category, so a 0.05 gap is
# indistinguishable from sampling variation and paying for it buys noise.
# Narrow it as samples accumulate.
# cheaper model wins. This is measurement noise, not preference: scores once
# rested on 2-3 samples per category, and a 0.05 gap is indistinguishable
# from sampling variation when the evidence is that thin.
#
# /metrics now enforces the "narrow it" promise: once average sample depth
# (outcome_samples + self_eval_samples) crosses the threshold in
# proficiency_sample_depth_warnings(), the premise has expired and the warning
# fires — so the loose tolerance is called out rather than silently prolonged.
# That threshold defaults to 20; see objective.proficiency_depth_warn_min_samples.
# Above this pace ratio, the dashboard raises an alarm.
# 1.25 = burning 25% faster than the plan allows.
plan_pace_warn_ratio: 1.25
@@ -139,6 +149,23 @@ objective:
# would let a single request's luck read as a measurement.
cache_rate_warn_min_observations: 25
# Premise-expiry check for quality_tolerance's "2-3 per category" claim.
# /metrics warns when average proficiency sample depth exceeds this — the
# point where the old thin-data justification is clearly obsolete and the
# loose tolerance should be narrowed.
proficiency_depth_warn_min_samples: 20
# Minimum proficiency rows before the depth check fires. Prevents a warning
# from a near-empty table (novelty-or-rate convention).
proficiency_depth_warn_min_rows: 10
# Premise-expiry check for the cost-as-tiebreak justification.
# /metrics warns when all-time SUM(cost_usd) exceeds this — the point where
# the "fractions of a cent / $0.07" claim is no longer the honest frame.
cumulative_spend_warn_usd: 50.0
# Minimum priced rows before the spend check fires. Prevents a warning from
# a DB with no priced rows.
cumulative_spend_warn_min_rows: 10
# --- Report-only measurement series. Neither is read by routing. ---------
#
# Cost-estimator calibration: routing.estimated_cost's prediction against
@@ -435,10 +462,11 @@ pinch:
# round-trip of 1.4-2.0 s, plus ~10 bytes per decision row.
prefix_probe: true
relevance:
# Off by default, matching every other new-and-unproven knob in this
# project — and specifically requires pinch.enabled too, since this has no
# effect otherwise. Ship it, watch route_decisions / pinch stats on real
# traffic, then decide the default.
# ON by default. The embed is now budget-gated (it only fires when the
# conversation exceeds pinch.budget_tokens and pruning would actually
# happen), so the relevance path is safe to leave on. Still requires
# pinch.enabled (no effect otherwise) and still needs nomic-embed-text on
# the configured Ollama (pull: ollama pull nomic-embed-text).
enabled: true
# An EMBEDDING model, not a chat model — this must not point at
# classifier.model or verification.model. Pull one on the same Ollama:

View File

@@ -125,6 +125,15 @@ class Objective(StrictModel):
cache_rate_warn_margin: Optional[float] = None
cache_rate_warn_min_observations: Optional[int] = None
# Premise-expiry checks for the quality_tolerance and cost-as-tiebreak
# justifications in config.yaml. These are /metrics warning thresholds,
# not dispatch controls — they fire once the comment's premise is clearly
# expired, so the operator knows to re-read that justification.
proficiency_depth_warn_min_samples: Optional[int] = None
proficiency_depth_warn_min_rows: Optional[int] = None
cumulative_spend_warn_usd: Optional[float] = None
cumulative_spend_warn_min_rows: Optional[int] = None
# Report-only measurement windows. Neither series is read by routing;
# see metrics.cost_estimate_calibration / metrics.latency_series.
cost_calibration_window_hours: Optional[int] = None
@@ -172,6 +181,43 @@ class Objective(StrictModel):
"objective.billing_reset_day must be in 1..28, or null to disable"
)
return v
@field_validator("proficiency_depth_warn_min_samples")
@classmethod
def depth_min_samples_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.proficiency_depth_warn_min_samples must be > 0"
)
return v
@field_validator("proficiency_depth_warn_min_rows")
@classmethod
def depth_min_rows_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.proficiency_depth_warn_min_rows must be > 0"
)
return v
@field_validator("cumulative_spend_warn_usd")
@classmethod
def spend_warn_usd_positive(cls, v: Optional[float]) -> Optional[float]:
if v is not None and v <= 0:
raise ValueError(
"objective.cumulative_spend_warn_usd must be > 0"
)
return v
@field_validator("cumulative_spend_warn_min_rows")
@classmethod
def spend_min_rows_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.cumulative_spend_warn_min_rows must be > 0"
)
return v
@field_validator("max_energy_per_request")
@classmethod
def ceiling_positive(cls, v: Optional[float]) -> Optional[float]:

View File

@@ -1348,6 +1348,14 @@ def scoring_coverage(
cache = cache_rate_series(conn, cfg)
warnings.extend(cache_rate_warnings(conn, cfg, cache))
# Premise-expiry checks: computed once and passed in, same pattern as the
# cache-rate warning so /metrics shows one consistent reading.
depth = proficiency_sample_depth_series(conn, cfg)
warnings.extend(proficiency_sample_depth_warnings(conn, cfg, depth))
spend = cumulative_spend_series(conn, cfg)
warnings.extend(cumulative_spend_warnings(conn, cfg, spend))
selection = selection_coverage(conn, cfg)
warnings.extend(selection_coverage_warnings(selection))
@@ -1363,6 +1371,11 @@ def scoring_coverage(
# its payload in dispatcher.py, and this is a telemetry-coverage
# question anyway: which routes can even be measured, and at what rate.
"cache": cache,
# Premise-expiry series: same pattern, another telemetry-coverage
# reading — the operator needs the number that triggered the warning,
# not only the warning text.
"proficiency_sample_depth": depth,
"cumulative_spend": spend,
}
@@ -2151,6 +2164,14 @@ _CACHE_RATE_WINDOW_HOURS: Final = 168
_CACHE_RATE_WARN_MARGIN: Final = 0.10
_CACHE_RATE_MIN_OBSERVATIONS: Final = 25
# Defaults for the premise-expiry checks, matching the pattern above. These
# live in config.yaml under objective:; these Final values exist so a
# SimpleNamespace or an older config still produces a sane series.
_PRF_DEPTH_WARN_MIN_SAMPLES: Final = 20
_PRF_DEPTH_WARN_MIN_ROWS: Final = 10
_CUMULATIVE_SPEND_WARN_USD: Final = 50.0
_CUMULATIVE_SPEND_WARN_MIN_ROWS: Final = 10
def _has_column(conn: sqlite3.Connection, table: str, column: str) -> bool:
"""True when *table* carries *column*, for additively-migrated schemas.
@@ -2352,6 +2373,259 @@ def cache_rate_warnings(
return warnings
def proficiency_sample_depth_series(
conn: sqlite3.Connection, cfg: Any
) -> dict:
"""Average sample depth (outcome_samples + self_eval_samples) per category.
Draws from the full proficiency table (cumulative, no trailing window). The
depth per row follows ``proficiency_matrix`` at the top of this file:
``samples = (self_eval_samples or 0) + (outcome_samples or 0)``.
Returns overall and per-category averages, plus thin row counts.
"""
min_samples = getattr(
cfg.objective, "proficiency_depth_warn_min_samples", None
)
if min_samples is None:
min_samples = _PRF_DEPTH_WARN_MIN_SAMPLES
series: dict = {
"overall_avg_depth": None,
"total_rows": 0,
"thin_rows": 0,
"by_category": [],
}
rows = conn.execute(
"""
SELECT category,
COUNT(*) AS cnt,
AVG(COALESCE(self_eval_samples, 0)
+ COALESCE(outcome_samples, 0)) AS avg_depth,
SUM(CASE WHEN COALESCE(self_eval_samples, 0)
+ COALESCE(outcome_samples, 0) < ? THEN 1 ELSE 0 END)
AS thin
FROM proficiency
GROUP BY category
""",
(min_samples,),
).fetchall()
total_rows = 0
total_sampled_rows = 0
depth_sum = 0.0
thin_total = 0
by_cat: list[dict] = []
for row in rows:
cnt = row["cnt"] or 0
avg = row["avg_depth"]
thin = row["thin"] or 0
total_rows += cnt
thin_total += thin
by_cat.append(
{
"category": row["category"],
"rows": cnt,
"avg_depth": avg,
"thin": thin,
}
)
if avg is not None:
total_sampled_rows += cnt
depth_sum += avg * cnt
by_cat.sort(key=lambda d: d["category"])
series["total_rows"] = total_rows
series["thin_rows"] = thin_total
series["by_category"] = by_cat
series["overall_avg_depth"] = (
(depth_sum / total_sampled_rows) if total_sampled_rows else None
)
return series
def proficiency_sample_depth_warnings(
conn: sqlite3.Connection,
cfg: Any,
series: Optional[dict] = None,
) -> List[str]:
"""Expiry check on ``objective.quality_tolerance``'s thin-data justification.
The config.yaml comment on ``quality_tolerance`` says "scores rest on 2-3
samples per category... Narrow it as samples accumulate." Once average
sample depth has grown past ``objective.proficiency_depth_warn_min_samples``
(default 20), that justification is clearly obsolete and /metrics warns that
the premise has expired.
Respects a minimum-observation floor
(``objective.proficiency_depth_warn_min_rows``) so a near-empty table does
not fire.
"""
if series is None:
series = proficiency_sample_depth_series(conn, cfg)
min_rows = getattr(cfg.objective, "proficiency_depth_warn_min_rows", None)
if min_rows is None:
min_rows = _PRF_DEPTH_WARN_MIN_ROWS
min_samples = getattr(
cfg.objective, "proficiency_depth_warn_min_samples", None
)
if min_samples is None:
min_samples = _PRF_DEPTH_WARN_MIN_SAMPLES
warnings: List[str] = []
total_rows = series.get("total_rows") or 0
avg_depth = series.get("overall_avg_depth")
if avg_depth is None or total_rows < min_rows:
return warnings
if avg_depth > min_samples:
warnings.append(
f"quality_tolerance premise expired: average proficiency sample "
f"depth {avg_depth:.1f} across {total_rows} rows has passed the "
f"warning threshold ({min_samples}), so the '2-3 samples per "
f"category' justification for objective.quality_tolerance={getattr(cfg.objective, 'quality_tolerance', '?')} "
f"is obsolete — the loose tolerance can be narrowed."
)
for entry in series.get("by_category") or []:
cat_avg = entry.get("avg_depth")
cat_rows = entry.get("rows") or 0
if cat_avg is None or cat_rows < min_rows:
continue
if cat_avg > min_samples:
warnings.append(
f"quality_tolerance premise expired: category "
f"{entry['category']} average depth {cat_avg:.1f} over "
f"{cat_rows} rows has passed the warning threshold "
f"({min_samples}) — objective.quality_tolerance premise "
f"is obsolete for this category."
)
return warnings
def cumulative_spend_series(conn: sqlite3.Connection, cfg: Any) -> dict:
"""All-time SUM(cost_usd) from energy_observations, excluding seed_reference.
Mirrors the cache-rate series' exclusion pattern: ``task_category !=
'seed_reference'`` keeps reference sweeps from inflating "real traffic".
Also surfaces per-provider breakdown and row counts. NULL cost_usd is
treated as 0.
"""
series: dict = {
"total_spend_usd": 0.0,
"priced_rows": 0,
"total_rows": 0,
"by_provider": [],
}
total_row = conn.execute(
"""
SELECT COUNT(*) AS total_rows,
COUNT(cost_usd) AS priced_rows,
COALESCE(SUM(cost_usd), 0) AS total_spend
FROM energy_observations
WHERE task_category IS NULL OR task_category != ?
""",
(SEED_CATEGORY,),
).fetchone()
total_rows = total_row["total_rows"] or 0
priced_rows = total_row["priced_rows"] or 0
total_spend = float(total_row["total_spend"] or 0.0)
provider_rows = conn.execute(
"""
SELECT provider,
COUNT(*) AS total_rows,
COUNT(cost_usd) AS priced_rows,
COALESCE(SUM(cost_usd), 0) AS provider_spend
FROM energy_observations
WHERE task_category IS NULL OR task_category != ?
GROUP BY provider
ORDER BY provider
""",
(SEED_CATEGORY,),
).fetchall()
series["total_spend_usd"] = round(total_spend, 4)
series["priced_rows"] = priced_rows
series["total_rows"] = total_rows
by_provider: list[dict] = []
for row in provider_rows:
by_provider.append(
{
"provider": row["provider"],
"total_rows": row["total_rows"] or 0,
"priced_rows": row["priced_rows"] or 0,
"provider_spend_usd": round(
float(row["provider_spend"] or 0.0), 4
),
}
)
series["by_provider"] = by_provider
return series
def cumulative_spend_warnings(
conn: sqlite3.Connection,
cfg: Any,
series: Optional[dict] = None,
) -> List[str]:
"""Expiry check on the cost-as-tiebreak comment's $0.07 claim.
The config.yaml comment on the blend section says "60% of every decision
adjudicated fractions of a cent (all real traffic to date totals $0.07)".
Once all-time spend exceeds ``objective.cumulative_spend_warn_usd``
(default $50.00), that justification no longer matches reality. Respects a
minimum-observation floor so a DB with no priced rows does not fire.
"""
if series is None:
series = cumulative_spend_series(conn, cfg)
warn_usd = getattr(cfg.objective, "cumulative_spend_warn_usd", None)
if warn_usd is None:
warn_usd = _CUMULATIVE_SPEND_WARN_USD
min_rows = getattr(cfg.objective, "cumulative_spend_warn_min_rows", None)
if min_rows is None:
min_rows = _CUMULATIVE_SPEND_WARN_MIN_ROWS
warnings: List[str] = []
priced_rows = series.get("priced_rows") or 0
total_spend = series.get("total_spend_usd") or 0.0
if priced_rows < min_rows:
return warnings
if total_spend > warn_usd:
warnings.append(
f"cost-as-tiebreak premise expired: cumulative spend "
f"${total_spend:.2f} across {priced_rows} priced rows has "
f"passed the warning threshold (${warn_usd:.2f}), so the "
f"blend comment's '$0.07 / fractions of a cent' justification "
f"in config.yaml no longer matches reality — it is off by "
f"{total_spend / 0.07:.0f}x."
)
for entry in series.get("by_provider") or []:
provider_spend = entry.get("provider_spend_usd") or 0.0
provider_priced = entry.get("priced_rows") or 0
if provider_spend > warn_usd and provider_priced >= min_rows:
warnings.append(
f"cost-as-tiebreak premise expired: provider "
f"{entry['provider']} spend ${provider_spend:.2f} has "
f"passed the warning threshold (${warn_usd:.2f}) — "
f"the blend comment's 'fractions of a cent' frame is "
f"stale for this provider."
)
return warnings
# Defaults for the two report-only series below, used when config omits the
# keys. Same purpose as the cache-rate defaults above: a SimpleNamespace config
# in a test, or a deployment on an older config file, still produces a series

View File

@@ -280,6 +280,24 @@ DELIBERATELY_NOT_IN_ADMIN: dict[str, str] = {
"objective.cache_rate_warn_min_observations": (
"sample floor for the cache-rate warning; tunes a warning."
),
# --- premise-expiry check thresholds: they tune a /metrics warning
# --- class, never dispatch. Same category as the cache-rate warn knobs.
"objective.proficiency_depth_warn_min_samples": (
"minimum average sample depth threshold for the proficiency-depth "
"premise-expiry warning; tunes a /metrics warning, not dispatch."
),
"objective.proficiency_depth_warn_min_rows": (
"observation floor for the proficiency-depth premise-expiry warning; "
"tunes a /metrics warning, not dispatch."
),
"objective.cumulative_spend_warn_usd": (
"dollar threshold for the cumulative-spend premise-expiry warning; "
"tunes a /metrics warning, not dispatch."
),
"objective.cumulative_spend_warn_min_rows": (
"priced-row floor for the cumulative-spend premise-expiry warning; "
"tunes a /metrics warning, not dispatch."
),
# --- report-only series: they change what /metrics SHOWS, and not even
# --- what it warns about. metrics.cost_estimate_calibration and
# --- metrics.latency_series emit no warning class at all and are read by

View File

@@ -36,10 +36,14 @@ from metrics import (
capability_ceilings,
capability_demand_warnings,
context_ceilings,
cumulative_spend_series,
cumulative_spend_warnings,
demand_ceiling_warnings,
local_energy_summary,
per_model,
pinch_summary,
proficiency_sample_depth_series,
proficiency_sample_depth_warnings,
quota_accounts,
recent_decisions,
rejection_warnings,
@@ -2135,3 +2139,295 @@ def test_scoring_coverage_carries_the_cache_series_and_its_warnings(tmp_path):
assert coverage["cache"]["cache_rate"] == pytest.approx(0.4)
assert any(w.startswith("cache rate: measured") for w in coverage["warnings"])
# =============================================================================
# proficiency_sample_depth_series / _warnings (Wave 5.3)
# =============================================================================
def _seed_proficiency_depth_rows(
conn: sqlite3.Connection,
*,
category: str = "coding_general",
n: int,
outcome_samples: int = 0,
self_eval_samples: int = 0,
model_id: str = "cheap",
seed_models: bool = True,
) -> None:
"""Insert *n* proficiency rows with the given sample counts."""
for i in range(n):
suffix = str(i) if i else ""
mid = model_id + suffix
if seed_models:
conn.execute(
"""
INSERT OR IGNORE INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, 2, 262128, 192500, 16384, 0.30, 0.10,
1, 1, 'standard', 'default', 'full', 'public', 'active',
'2026-09-01T00:00:00+00:00')
""",
(mid, mid),
)
conn.execute(
"""
INSERT INTO proficiency (
model_id, provider, category, blended_score, source,
self_eval_samples, outcome_samples, last_updated
) VALUES (?, 'neuralwatt', ?, 0.9, 'self_eval_thin',
?, ?, '2026-09-01T00:00:00+00:00')
""",
(mid, category, self_eval_samples, outcome_samples),
)
conn.commit()
def test_proficiency_depth_series_computes_average_depth(tmp_path):
"""The series aggregates sample depth per category and overall."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
self_eval_samples=10, outcome_samples=5)
_seed_proficiency_depth_rows(conn, n=2, category="debugging",
self_eval_samples=20, outcome_samples=10)
series = proficiency_sample_depth_series(conn, CFG)
# coding_general: 3 rows, avg depth (15+15+15)/3 = 15
# debugging: 2 rows, avg depth (30+30)/2 = 30
# overall: (45+60)/5 = 21.0
assert series["total_rows"] == 5
assert series["overall_avg_depth"] == pytest.approx(21.0)
by_cat = {e["category"]: e for e in series["by_category"]}
assert by_cat["coding_general"]["avg_depth"] == pytest.approx(15.0)
assert by_cat["coding_general"]["rows"] == 3
assert by_cat["debugging"]["avg_depth"] == pytest.approx(30.0)
assert by_cat["debugging"]["rows"] == 2
def test_proficiency_depth_series_identifies_thin_rows(tmp_path):
"""Rows below min_samples are counted as thin."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=2, category="coding_general",
self_eval_samples=1, outcome_samples=1)
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
self_eval_samples=20, outcome_samples=10,
model_id="dear")
series = proficiency_sample_depth_series(conn, CFG)
assert series["thin_rows"] == 2
assert series["total_rows"] == 5
def test_proficiency_depth_warning_silent_below_the_observation_floor(tmp_path):
"""Never warns on a near-empty table (novelty-or-rate)."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
self_eval_samples=5, outcome_samples=5)
warnings = proficiency_sample_depth_warnings(conn, CFG)
assert warnings == []
def test_proficiency_depth_warning_silent_below_the_threshold(tmp_path):
"""Average sample depth below min_samples does not fire."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=15, category="coding_general",
self_eval_samples=5, outcome_samples=5)
cfg = _cfg_with(proficiency_depth_warn_min_samples=50)
warnings = proficiency_sample_depth_warnings(conn, cfg)
assert warnings == []
def test_proficiency_depth_warning_fires_when_depth_exceeds_threshold(tmp_path):
"""Once average depth passes the configurable threshold, the premise is
expired and the warning fires."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=20, category="coding_general",
self_eval_samples=15, outcome_samples=10)
warnings = proficiency_sample_depth_warnings(conn, CFG)
premise = [w for w in warnings
if w.startswith("quality_tolerance premise expired")]
assert len(premise) >= 1
assert "quality_tolerance" in premise[0]
def test_proficiency_depth_warning_respects_the_observation_floor(tmp_path):
"""Plenty of rows but all below the floor threshold — silent."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=5, category="coding_general",
self_eval_samples=30, outcome_samples=30)
cfg = _cfg_with(proficiency_depth_warn_min_rows=10)
warnings = proficiency_sample_depth_warnings(conn, cfg)
assert warnings == []
# =============================================================================
# cumulative_spend_series / _warnings (Wave 5.3)
# =============================================================================
def _seed_spend_rows(
conn: sqlite3.Connection,
*,
n: int,
cost_usd: float = 0.0,
model_id: str = "cheap",
provider: str = "neuralwatt",
task_category: str = "coding_general",
) -> None:
"""Insert *n* energy_observations rows with a cost_usd."""
now = _now()
for _ in range(n):
conn.execute(
"""
INSERT INTO energy_observations (
model_id, provider, task_category, prompt_tokens,
completion_tokens, energy_kwh, cost_usd, observed_at
) VALUES (?, ?, ?, 1000, 100, 0.001, ?, ?)
""",
(model_id, provider, task_category, cost_usd, now.isoformat()),
)
conn.commit()
def _seed_spend_seed_rows(conn: sqlite3.Connection, *, n: int = 5) -> None:
"""Insert seed_reference energy rows (excluded from spend)."""
now = _now()
for _ in range(n):
conn.execute(
"""
INSERT INTO energy_observations (
model_id, provider, task_category, prompt_tokens,
completion_tokens, energy_kwh, cost_usd, observed_at
) VALUES ('cheap', 'neuralwatt', 'seed_reference', 1000, 100,
0.001, 999.0, ?)
""",
(now.isoformat(),),
)
conn.commit()
def test_cumulative_spend_series_aggregates_all_time_cost(tmp_path):
"""The series sums cost_usd across all non-seed energy rows."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=3, cost_usd=0.05)
_seed_spend_rows(conn, n=2, cost_usd=0.10)
series = cumulative_spend_series(conn, CFG)
assert series["total_spend_usd"] == pytest.approx(0.35)
assert series["priced_rows"] == 5
assert series["total_rows"] == 5
def test_cumulative_spend_series_excludes_seed_reference(tmp_path):
"""seed_reference rows do not inflate 'real traffic' spend."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=2, cost_usd=0.05)
_seed_spend_seed_rows(conn, n=5)
series = cumulative_spend_series(conn, CFG)
assert series["total_spend_usd"] == pytest.approx(0.10)
assert series["priced_rows"] == 2
# total_rows is also seed-filtered: the series describes real traffic only,
# so the excluded rows never appear in any of its counts.
assert series["total_rows"] == 2
def test_cumulative_spend_series_per_provider_breakdown(tmp_path):
"""Results include per-provider sub-totals."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=2, cost_usd=1.0, provider="neuralwatt")
_seed_spend_rows(conn, n=3, cost_usd=2.0, provider="openrouter")
series = cumulative_spend_series(conn, CFG)
assert series["total_spend_usd"] == pytest.approx(8.0)
by_provider = {e["provider"]: e for e in series["by_provider"]}
assert by_provider["neuralwatt"]["provider_spend_usd"] == pytest.approx(2.0)
assert by_provider["openrouter"]["provider_spend_usd"] == pytest.approx(6.0)
def test_cumulative_spend_warning_silent_below_the_observation_floor(tmp_path):
"""Never warns on too few priced rows (novelty-or-rate)."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=3, cost_usd=100.0)
cfg = _cfg_with(cumulative_spend_warn_min_rows=10)
warnings = cumulative_spend_warnings(conn, cfg)
assert warnings == []
def test_cumulative_spend_warning_silent_below_the_threshold(tmp_path):
"""Spend below the warning threshold does not fire."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=15, cost_usd=0.01)
cfg = _cfg_with(cumulative_spend_warn_usd=50.0)
warnings = cumulative_spend_warnings(conn, cfg)
assert warnings == []
def test_cumulative_spend_warning_fires_when_premise_expires(tmp_path):
"""Once all-time spend exceeds warn_usd, the warning fires naming the
blend comment whose premise has expired."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=15, cost_usd=50.0)
cfg = _cfg_with(cumulative_spend_warn_usd=10.0,
cumulative_spend_warn_min_rows=5)
warnings = cumulative_spend_warnings(conn, cfg)
premise = [w for w in warnings
if w.startswith("cost-as-tiebreak premise expired")]
assert len(premise) >= 1
assert "cumulative spend" in premise[0]
assert "$" in premise[0]
def test_scoring_coverage_carries_the_premise_expiry_series(tmp_path):
"""scoring_coverage includes both new premise-expiry series and warnings."""
conn = _make_db(tmp_path)
_seed_models(conn)
# Seed enough proficiency depth to trigger the warning
_seed_proficiency_depth_rows(conn, n=20, category="coding_general",
self_eval_samples=15, outcome_samples=10)
_seed_spend_rows(conn, n=15, cost_usd=50.0)
cfg = _cfg_with(
cumulative_spend_warn_usd=10.0,
cumulative_spend_warn_min_rows=5,
)
coverage = scoring_coverage(conn, cfg)
assert "proficiency_sample_depth" in coverage
assert coverage["proficiency_sample_depth"]["overall_avg_depth"] is not None
assert "cumulative_spend" in coverage
assert coverage["cumulative_spend"]["total_spend_usd"] == pytest.approx(750.0)
assert any(
w.startswith("quality_tolerance premise expired")
for w in coverage["warnings"]
)
assert any(
w.startswith("cost-as-tiebreak premise expired")
for w in coverage["warnings"]
)