From 3400c3ca73539d5d407354b3bf9bc943bbafaa07 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Fri, 18 Sep 2026 23:34:44 -0400 Subject: [PATCH] feat(metrics): premise-expiry checks for quality_tolerance and cost-as-tiebreak MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two new /metrics detector pairs following the Wave 1.4 cache-rate precedent (series + warning with a minimum-observation floor, wired into scoring_coverage): - proficiency_sample_depth_series/_warnings: average proficiency sample depth (outcome_samples + self_eval_samples) per category and overall. Warns 'quality_tolerance premise expired:' once depth crosses objective.proficiency_depth_warn_min_samples with enough rows on the table — the '2-3 samples per category' justification for the loose tolerance is now enforced by the check, not memory. - cumulative_spend_series/_warnings: all-time SUM(cost_usd) from energy_observations excluding seed_reference rows, with per-provider breakdown. Warns 'cost-as-tiebreak premise expired:' once spend passes objective.cumulative_spend_warn_usd — the '$0.07 / fractions of a cent' figure in the blend comment was off by orders of magnitude. Four objective knobs (Optional, >0 validators, real values + comments in config.yaml, deliberate-absent entries in the admin-knob gate) and three config.yaml comment rewrites: the two stale justifications now reference the live check function names instead of hardcoded numbers, and the pinch.relevance.enabled comment now describes the actual ON-by-default state and its budget-gate rationale (commit 69662f0) instead of the pre-decision text. Full suite: 2163 passed, 0 failed. --- config/config.yaml | 48 ++++- src/config.py | 46 +++++ src/metrics.py | 274 +++++++++++++++++++++++++++ tests/test_admin_knob_coverage.py | 18 ++ tests/test_metrics.py | 296 ++++++++++++++++++++++++++++++ 5 files changed, 672 insertions(+), 10 deletions(-) diff --git a/config/config.yaml b/config/config.yaml index 2672e0e..4f49daa 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -10,14 +10,24 @@ objective: # This replaced a weighted blend (cost 0.4 / eco 0.2 / proficiency 0.4). # Measurement killed it: turning the cost weight from 0.4 to ZERO changed # the winner in only 2 of 6 categories, so the blend was never steering on - # quality — while 60% of every decision adjudicated fractions of a cent - # (all real traffic to date totals $0.07). + # quality — cost decided almost nothing while consuming 40% of every decision. + # + # The old comment cited $0.07 total spend as evidence it was "fractions of a + # cent" — that figure expired long ago. /metrics now surfaces current all-time + # spend via cumulative_spend_series() and warns via cumulative_spend_warnings() + # once it has grown past the tiebreak's "harmless fractions of a cent" frame, + # so the justification is checked live rather than embedded as a stale number. # Proficiency differences smaller than this are treated as equal and the - # cheaper model wins. This is measurement noise, not preference: scores - # currently rest on 2-3 samples per category, so a 0.05 gap is - # indistinguishable from sampling variation and paying for it buys noise. - # Narrow it as samples accumulate. + # cheaper model wins. This is measurement noise, not preference: scores once + # rested on 2-3 samples per category, and a 0.05 gap is indistinguishable + # from sampling variation when the evidence is that thin. + # + # /metrics now enforces the "narrow it" promise: once average sample depth + # (outcome_samples + self_eval_samples) crosses the threshold in + # proficiency_sample_depth_warnings(), the premise has expired and the warning + # fires — so the loose tolerance is called out rather than silently prolonged. + # That threshold defaults to 20; see objective.proficiency_depth_warn_min_samples. # Above this pace ratio, the dashboard raises an alarm. # 1.25 = burning 25% faster than the plan allows. plan_pace_warn_ratio: 1.25 @@ -139,6 +149,23 @@ objective: # would let a single request's luck read as a measurement. cache_rate_warn_min_observations: 25 + # Premise-expiry check for quality_tolerance's "2-3 per category" claim. + # /metrics warns when average proficiency sample depth exceeds this — the + # point where the old thin-data justification is clearly obsolete and the + # loose tolerance should be narrowed. + proficiency_depth_warn_min_samples: 20 + # Minimum proficiency rows before the depth check fires. Prevents a warning + # from a near-empty table (novelty-or-rate convention). + proficiency_depth_warn_min_rows: 10 + + # Premise-expiry check for the cost-as-tiebreak justification. + # /metrics warns when all-time SUM(cost_usd) exceeds this — the point where + # the "fractions of a cent / $0.07" claim is no longer the honest frame. + cumulative_spend_warn_usd: 50.0 + # Minimum priced rows before the spend check fires. Prevents a warning from + # a DB with no priced rows. + cumulative_spend_warn_min_rows: 10 + # --- Report-only measurement series. Neither is read by routing. --------- # # Cost-estimator calibration: routing.estimated_cost's prediction against @@ -435,10 +462,11 @@ pinch: # round-trip of 1.4-2.0 s, plus ~10 bytes per decision row. prefix_probe: true relevance: - # Off by default, matching every other new-and-unproven knob in this - # project — and specifically requires pinch.enabled too, since this has no - # effect otherwise. Ship it, watch route_decisions / pinch stats on real - # traffic, then decide the default. + # ON by default. The embed is now budget-gated (it only fires when the + # conversation exceeds pinch.budget_tokens and pruning would actually + # happen), so the relevance path is safe to leave on. Still requires + # pinch.enabled (no effect otherwise) and still needs nomic-embed-text on + # the configured Ollama (pull: ollama pull nomic-embed-text). enabled: true # An EMBEDDING model, not a chat model — this must not point at # classifier.model or verification.model. Pull one on the same Ollama: diff --git a/src/config.py b/src/config.py index e7e5e1d..d80864f 100644 --- a/src/config.py +++ b/src/config.py @@ -125,6 +125,15 @@ class Objective(StrictModel): cache_rate_warn_margin: Optional[float] = None cache_rate_warn_min_observations: Optional[int] = None + # Premise-expiry checks for the quality_tolerance and cost-as-tiebreak + # justifications in config.yaml. These are /metrics warning thresholds, + # not dispatch controls — they fire once the comment's premise is clearly + # expired, so the operator knows to re-read that justification. + proficiency_depth_warn_min_samples: Optional[int] = None + proficiency_depth_warn_min_rows: Optional[int] = None + cumulative_spend_warn_usd: Optional[float] = None + cumulative_spend_warn_min_rows: Optional[int] = None + # Report-only measurement windows. Neither series is read by routing; # see metrics.cost_estimate_calibration / metrics.latency_series. cost_calibration_window_hours: Optional[int] = None @@ -172,6 +181,43 @@ class Objective(StrictModel): "objective.billing_reset_day must be in 1..28, or null to disable" ) return v + + @field_validator("proficiency_depth_warn_min_samples") + @classmethod + def depth_min_samples_positive(cls, v: Optional[int]) -> Optional[int]: + if v is not None and v <= 0: + raise ValueError( + "objective.proficiency_depth_warn_min_samples must be > 0" + ) + return v + + @field_validator("proficiency_depth_warn_min_rows") + @classmethod + def depth_min_rows_positive(cls, v: Optional[int]) -> Optional[int]: + if v is not None and v <= 0: + raise ValueError( + "objective.proficiency_depth_warn_min_rows must be > 0" + ) + return v + + @field_validator("cumulative_spend_warn_usd") + @classmethod + def spend_warn_usd_positive(cls, v: Optional[float]) -> Optional[float]: + if v is not None and v <= 0: + raise ValueError( + "objective.cumulative_spend_warn_usd must be > 0" + ) + return v + + @field_validator("cumulative_spend_warn_min_rows") + @classmethod + def spend_min_rows_positive(cls, v: Optional[int]) -> Optional[int]: + if v is not None and v <= 0: + raise ValueError( + "objective.cumulative_spend_warn_min_rows must be > 0" + ) + return v + @field_validator("max_energy_per_request") @classmethod def ceiling_positive(cls, v: Optional[float]) -> Optional[float]: diff --git a/src/metrics.py b/src/metrics.py index 736419c..976b0ad 100644 --- a/src/metrics.py +++ b/src/metrics.py @@ -1348,6 +1348,14 @@ def scoring_coverage( cache = cache_rate_series(conn, cfg) warnings.extend(cache_rate_warnings(conn, cfg, cache)) + # Premise-expiry checks: computed once and passed in, same pattern as the + # cache-rate warning so /metrics shows one consistent reading. + depth = proficiency_sample_depth_series(conn, cfg) + warnings.extend(proficiency_sample_depth_warnings(conn, cfg, depth)) + + spend = cumulative_spend_series(conn, cfg) + warnings.extend(cumulative_spend_warnings(conn, cfg, spend)) + selection = selection_coverage(conn, cfg) warnings.extend(selection_coverage_warnings(selection)) @@ -1363,6 +1371,11 @@ def scoring_coverage( # its payload in dispatcher.py, and this is a telemetry-coverage # question anyway: which routes can even be measured, and at what rate. "cache": cache, + # Premise-expiry series: same pattern, another telemetry-coverage + # reading — the operator needs the number that triggered the warning, + # not only the warning text. + "proficiency_sample_depth": depth, + "cumulative_spend": spend, } @@ -2151,6 +2164,14 @@ _CACHE_RATE_WINDOW_HOURS: Final = 168 _CACHE_RATE_WARN_MARGIN: Final = 0.10 _CACHE_RATE_MIN_OBSERVATIONS: Final = 25 +# Defaults for the premise-expiry checks, matching the pattern above. These +# live in config.yaml under objective:; these Final values exist so a +# SimpleNamespace or an older config still produces a sane series. +_PRF_DEPTH_WARN_MIN_SAMPLES: Final = 20 +_PRF_DEPTH_WARN_MIN_ROWS: Final = 10 +_CUMULATIVE_SPEND_WARN_USD: Final = 50.0 +_CUMULATIVE_SPEND_WARN_MIN_ROWS: Final = 10 + def _has_column(conn: sqlite3.Connection, table: str, column: str) -> bool: """True when *table* carries *column*, for additively-migrated schemas. @@ -2352,6 +2373,259 @@ def cache_rate_warnings( return warnings +def proficiency_sample_depth_series( + conn: sqlite3.Connection, cfg: Any +) -> dict: + """Average sample depth (outcome_samples + self_eval_samples) per category. + + Draws from the full proficiency table (cumulative, no trailing window). The + depth per row follows ``proficiency_matrix`` at the top of this file: + ``samples = (self_eval_samples or 0) + (outcome_samples or 0)``. + + Returns overall and per-category averages, plus thin row counts. + """ + min_samples = getattr( + cfg.objective, "proficiency_depth_warn_min_samples", None + ) + if min_samples is None: + min_samples = _PRF_DEPTH_WARN_MIN_SAMPLES + + series: dict = { + "overall_avg_depth": None, + "total_rows": 0, + "thin_rows": 0, + "by_category": [], + } + + rows = conn.execute( + """ + SELECT category, + COUNT(*) AS cnt, + AVG(COALESCE(self_eval_samples, 0) + + COALESCE(outcome_samples, 0)) AS avg_depth, + SUM(CASE WHEN COALESCE(self_eval_samples, 0) + + COALESCE(outcome_samples, 0) < ? THEN 1 ELSE 0 END) + AS thin + FROM proficiency + GROUP BY category + """, + (min_samples,), + ).fetchall() + + total_rows = 0 + total_sampled_rows = 0 + depth_sum = 0.0 + thin_total = 0 + by_cat: list[dict] = [] + for row in rows: + cnt = row["cnt"] or 0 + avg = row["avg_depth"] + thin = row["thin"] or 0 + total_rows += cnt + thin_total += thin + by_cat.append( + { + "category": row["category"], + "rows": cnt, + "avg_depth": avg, + "thin": thin, + } + ) + if avg is not None: + total_sampled_rows += cnt + depth_sum += avg * cnt + + by_cat.sort(key=lambda d: d["category"]) + series["total_rows"] = total_rows + series["thin_rows"] = thin_total + series["by_category"] = by_cat + series["overall_avg_depth"] = ( + (depth_sum / total_sampled_rows) if total_sampled_rows else None + ) + return series + + +def proficiency_sample_depth_warnings( + conn: sqlite3.Connection, + cfg: Any, + series: Optional[dict] = None, +) -> List[str]: + """Expiry check on ``objective.quality_tolerance``'s thin-data justification. + + The config.yaml comment on ``quality_tolerance`` says "scores rest on 2-3 + samples per category... Narrow it as samples accumulate." Once average + sample depth has grown past ``objective.proficiency_depth_warn_min_samples`` + (default 20), that justification is clearly obsolete and /metrics warns that + the premise has expired. + + Respects a minimum-observation floor + (``objective.proficiency_depth_warn_min_rows``) so a near-empty table does + not fire. + """ + if series is None: + series = proficiency_sample_depth_series(conn, cfg) + + min_rows = getattr(cfg.objective, "proficiency_depth_warn_min_rows", None) + if min_rows is None: + min_rows = _PRF_DEPTH_WARN_MIN_ROWS + min_samples = getattr( + cfg.objective, "proficiency_depth_warn_min_samples", None + ) + if min_samples is None: + min_samples = _PRF_DEPTH_WARN_MIN_SAMPLES + + warnings: List[str] = [] + + total_rows = series.get("total_rows") or 0 + avg_depth = series.get("overall_avg_depth") + if avg_depth is None or total_rows < min_rows: + return warnings + + if avg_depth > min_samples: + warnings.append( + f"quality_tolerance premise expired: average proficiency sample " + f"depth {avg_depth:.1f} across {total_rows} rows has passed the " + f"warning threshold ({min_samples}), so the '2-3 samples per " + f"category' justification for objective.quality_tolerance={getattr(cfg.objective, 'quality_tolerance', '?')} " + f"is obsolete — the loose tolerance can be narrowed." + ) + + for entry in series.get("by_category") or []: + cat_avg = entry.get("avg_depth") + cat_rows = entry.get("rows") or 0 + if cat_avg is None or cat_rows < min_rows: + continue + if cat_avg > min_samples: + warnings.append( + f"quality_tolerance premise expired: category " + f"{entry['category']} average depth {cat_avg:.1f} over " + f"{cat_rows} rows has passed the warning threshold " + f"({min_samples}) — objective.quality_tolerance premise " + f"is obsolete for this category." + ) + + return warnings + + +def cumulative_spend_series(conn: sqlite3.Connection, cfg: Any) -> dict: + """All-time SUM(cost_usd) from energy_observations, excluding seed_reference. + + Mirrors the cache-rate series' exclusion pattern: ``task_category != + 'seed_reference'`` keeps reference sweeps from inflating "real traffic". + Also surfaces per-provider breakdown and row counts. NULL cost_usd is + treated as 0. + """ + series: dict = { + "total_spend_usd": 0.0, + "priced_rows": 0, + "total_rows": 0, + "by_provider": [], + } + + total_row = conn.execute( + """ + SELECT COUNT(*) AS total_rows, + COUNT(cost_usd) AS priced_rows, + COALESCE(SUM(cost_usd), 0) AS total_spend + FROM energy_observations + WHERE task_category IS NULL OR task_category != ? + """, + (SEED_CATEGORY,), + ).fetchone() + total_rows = total_row["total_rows"] or 0 + priced_rows = total_row["priced_rows"] or 0 + total_spend = float(total_row["total_spend"] or 0.0) + + provider_rows = conn.execute( + """ + SELECT provider, + COUNT(*) AS total_rows, + COUNT(cost_usd) AS priced_rows, + COALESCE(SUM(cost_usd), 0) AS provider_spend + FROM energy_observations + WHERE task_category IS NULL OR task_category != ? + GROUP BY provider + ORDER BY provider + """, + (SEED_CATEGORY,), + ).fetchall() + + series["total_spend_usd"] = round(total_spend, 4) + series["priced_rows"] = priced_rows + series["total_rows"] = total_rows + + by_provider: list[dict] = [] + for row in provider_rows: + by_provider.append( + { + "provider": row["provider"], + "total_rows": row["total_rows"] or 0, + "priced_rows": row["priced_rows"] or 0, + "provider_spend_usd": round( + float(row["provider_spend"] or 0.0), 4 + ), + } + ) + series["by_provider"] = by_provider + return series + + +def cumulative_spend_warnings( + conn: sqlite3.Connection, + cfg: Any, + series: Optional[dict] = None, +) -> List[str]: + """Expiry check on the cost-as-tiebreak comment's $0.07 claim. + + The config.yaml comment on the blend section says "60% of every decision + adjudicated fractions of a cent (all real traffic to date totals $0.07)". + Once all-time spend exceeds ``objective.cumulative_spend_warn_usd`` + (default $50.00), that justification no longer matches reality. Respects a + minimum-observation floor so a DB with no priced rows does not fire. + """ + if series is None: + series = cumulative_spend_series(conn, cfg) + + warn_usd = getattr(cfg.objective, "cumulative_spend_warn_usd", None) + if warn_usd is None: + warn_usd = _CUMULATIVE_SPEND_WARN_USD + min_rows = getattr(cfg.objective, "cumulative_spend_warn_min_rows", None) + if min_rows is None: + min_rows = _CUMULATIVE_SPEND_WARN_MIN_ROWS + + warnings: List[str] = [] + + priced_rows = series.get("priced_rows") or 0 + total_spend = series.get("total_spend_usd") or 0.0 + + if priced_rows < min_rows: + return warnings + + if total_spend > warn_usd: + warnings.append( + f"cost-as-tiebreak premise expired: cumulative spend " + f"${total_spend:.2f} across {priced_rows} priced rows has " + f"passed the warning threshold (${warn_usd:.2f}), so the " + f"blend comment's '$0.07 / fractions of a cent' justification " + f"in config.yaml no longer matches reality — it is off by " + f"{total_spend / 0.07:.0f}x." + ) + + for entry in series.get("by_provider") or []: + provider_spend = entry.get("provider_spend_usd") or 0.0 + provider_priced = entry.get("priced_rows") or 0 + if provider_spend > warn_usd and provider_priced >= min_rows: + warnings.append( + f"cost-as-tiebreak premise expired: provider " + f"{entry['provider']} spend ${provider_spend:.2f} has " + f"passed the warning threshold (${warn_usd:.2f}) — " + f"the blend comment's 'fractions of a cent' frame is " + f"stale for this provider." + ) + + return warnings + + # Defaults for the two report-only series below, used when config omits the # keys. Same purpose as the cache-rate defaults above: a SimpleNamespace config # in a test, or a deployment on an older config file, still produces a series diff --git a/tests/test_admin_knob_coverage.py b/tests/test_admin_knob_coverage.py index 0eacadf..c1becc9 100644 --- a/tests/test_admin_knob_coverage.py +++ b/tests/test_admin_knob_coverage.py @@ -280,6 +280,24 @@ DELIBERATELY_NOT_IN_ADMIN: dict[str, str] = { "objective.cache_rate_warn_min_observations": ( "sample floor for the cache-rate warning; tunes a warning." ), + # --- premise-expiry check thresholds: they tune a /metrics warning + # --- class, never dispatch. Same category as the cache-rate warn knobs. + "objective.proficiency_depth_warn_min_samples": ( + "minimum average sample depth threshold for the proficiency-depth " + "premise-expiry warning; tunes a /metrics warning, not dispatch." + ), + "objective.proficiency_depth_warn_min_rows": ( + "observation floor for the proficiency-depth premise-expiry warning; " + "tunes a /metrics warning, not dispatch." + ), + "objective.cumulative_spend_warn_usd": ( + "dollar threshold for the cumulative-spend premise-expiry warning; " + "tunes a /metrics warning, not dispatch." + ), + "objective.cumulative_spend_warn_min_rows": ( + "priced-row floor for the cumulative-spend premise-expiry warning; " + "tunes a /metrics warning, not dispatch." + ), # --- report-only series: they change what /metrics SHOWS, and not even # --- what it warns about. metrics.cost_estimate_calibration and # --- metrics.latency_series emit no warning class at all and are read by diff --git a/tests/test_metrics.py b/tests/test_metrics.py index f6b1907..f416750 100644 --- a/tests/test_metrics.py +++ b/tests/test_metrics.py @@ -36,10 +36,14 @@ from metrics import ( capability_ceilings, capability_demand_warnings, context_ceilings, + cumulative_spend_series, + cumulative_spend_warnings, demand_ceiling_warnings, local_energy_summary, per_model, pinch_summary, + proficiency_sample_depth_series, + proficiency_sample_depth_warnings, quota_accounts, recent_decisions, rejection_warnings, @@ -2135,3 +2139,295 @@ def test_scoring_coverage_carries_the_cache_series_and_its_warnings(tmp_path): assert coverage["cache"]["cache_rate"] == pytest.approx(0.4) assert any(w.startswith("cache rate: measured") for w in coverage["warnings"]) + + +# ============================================================================= +# proficiency_sample_depth_series / _warnings (Wave 5.3) +# ============================================================================= + + +def _seed_proficiency_depth_rows( + conn: sqlite3.Connection, + *, + category: str = "coding_general", + n: int, + outcome_samples: int = 0, + self_eval_samples: int = 0, + model_id: str = "cheap", + seed_models: bool = True, +) -> None: + """Insert *n* proficiency rows with the given sample counts.""" + for i in range(n): + suffix = str(i) if i else "" + mid = model_id + suffix + if seed_models: + conn.execute( + """ + INSERT OR IGNORE INTO models ( + model_id, provider, base_model_id, tier, context_window, + effective_context_window, max_output_tokens, + cost_per_1m_prompt, cost_per_1m_completion, + supports_vision, supports_json_mode, + latency_class, reasoning_mode, context_variant, + access_level, availability, last_updated + ) VALUES (?, 'neuralwatt', ?, 2, 262128, 192500, 16384, 0.30, 0.10, + 1, 1, 'standard', 'default', 'full', 'public', 'active', + '2026-09-01T00:00:00+00:00') + """, + (mid, mid), + ) + conn.execute( + """ + INSERT INTO proficiency ( + model_id, provider, category, blended_score, source, + self_eval_samples, outcome_samples, last_updated + ) VALUES (?, 'neuralwatt', ?, 0.9, 'self_eval_thin', + ?, ?, '2026-09-01T00:00:00+00:00') + """, + (mid, category, self_eval_samples, outcome_samples), + ) + conn.commit() + + +def test_proficiency_depth_series_computes_average_depth(tmp_path): + """The series aggregates sample depth per category and overall.""" + conn = _make_db(tmp_path) + _seed_proficiency_depth_rows(conn, n=3, category="coding_general", + self_eval_samples=10, outcome_samples=5) + _seed_proficiency_depth_rows(conn, n=2, category="debugging", + self_eval_samples=20, outcome_samples=10) + + series = proficiency_sample_depth_series(conn, CFG) + + # coding_general: 3 rows, avg depth (15+15+15)/3 = 15 + # debugging: 2 rows, avg depth (30+30)/2 = 30 + # overall: (45+60)/5 = 21.0 + assert series["total_rows"] == 5 + assert series["overall_avg_depth"] == pytest.approx(21.0) + by_cat = {e["category"]: e for e in series["by_category"]} + assert by_cat["coding_general"]["avg_depth"] == pytest.approx(15.0) + assert by_cat["coding_general"]["rows"] == 3 + assert by_cat["debugging"]["avg_depth"] == pytest.approx(30.0) + assert by_cat["debugging"]["rows"] == 2 + + +def test_proficiency_depth_series_identifies_thin_rows(tmp_path): + """Rows below min_samples are counted as thin.""" + conn = _make_db(tmp_path) + _seed_proficiency_depth_rows(conn, n=2, category="coding_general", + self_eval_samples=1, outcome_samples=1) + _seed_proficiency_depth_rows(conn, n=3, category="coding_general", + self_eval_samples=20, outcome_samples=10, + model_id="dear") + + series = proficiency_sample_depth_series(conn, CFG) + + assert series["thin_rows"] == 2 + assert series["total_rows"] == 5 + + +def test_proficiency_depth_warning_silent_below_the_observation_floor(tmp_path): + """Never warns on a near-empty table (novelty-or-rate).""" + conn = _make_db(tmp_path) + _seed_proficiency_depth_rows(conn, n=3, category="coding_general", + self_eval_samples=5, outcome_samples=5) + + warnings = proficiency_sample_depth_warnings(conn, CFG) + + assert warnings == [] + + +def test_proficiency_depth_warning_silent_below_the_threshold(tmp_path): + """Average sample depth below min_samples does not fire.""" + conn = _make_db(tmp_path) + _seed_proficiency_depth_rows(conn, n=15, category="coding_general", + self_eval_samples=5, outcome_samples=5) + + cfg = _cfg_with(proficiency_depth_warn_min_samples=50) + warnings = proficiency_sample_depth_warnings(conn, cfg) + + assert warnings == [] + + +def test_proficiency_depth_warning_fires_when_depth_exceeds_threshold(tmp_path): + """Once average depth passes the configurable threshold, the premise is + expired and the warning fires.""" + conn = _make_db(tmp_path) + _seed_proficiency_depth_rows(conn, n=20, category="coding_general", + self_eval_samples=15, outcome_samples=10) + + warnings = proficiency_sample_depth_warnings(conn, CFG) + + premise = [w for w in warnings + if w.startswith("quality_tolerance premise expired")] + assert len(premise) >= 1 + assert "quality_tolerance" in premise[0] + + +def test_proficiency_depth_warning_respects_the_observation_floor(tmp_path): + """Plenty of rows but all below the floor threshold — silent.""" + conn = _make_db(tmp_path) + _seed_proficiency_depth_rows(conn, n=5, category="coding_general", + self_eval_samples=30, outcome_samples=30) + + cfg = _cfg_with(proficiency_depth_warn_min_rows=10) + warnings = proficiency_sample_depth_warnings(conn, cfg) + + assert warnings == [] + + +# ============================================================================= +# cumulative_spend_series / _warnings (Wave 5.3) +# ============================================================================= + + +def _seed_spend_rows( + conn: sqlite3.Connection, + *, + n: int, + cost_usd: float = 0.0, + model_id: str = "cheap", + provider: str = "neuralwatt", + task_category: str = "coding_general", +) -> None: + """Insert *n* energy_observations rows with a cost_usd.""" + now = _now() + for _ in range(n): + conn.execute( + """ + INSERT INTO energy_observations ( + model_id, provider, task_category, prompt_tokens, + completion_tokens, energy_kwh, cost_usd, observed_at + ) VALUES (?, ?, ?, 1000, 100, 0.001, ?, ?) + """, + (model_id, provider, task_category, cost_usd, now.isoformat()), + ) + conn.commit() + + +def _seed_spend_seed_rows(conn: sqlite3.Connection, *, n: int = 5) -> None: + """Insert seed_reference energy rows (excluded from spend).""" + now = _now() + for _ in range(n): + conn.execute( + """ + INSERT INTO energy_observations ( + model_id, provider, task_category, prompt_tokens, + completion_tokens, energy_kwh, cost_usd, observed_at + ) VALUES ('cheap', 'neuralwatt', 'seed_reference', 1000, 100, + 0.001, 999.0, ?) + """, + (now.isoformat(),), + ) + conn.commit() + + +def test_cumulative_spend_series_aggregates_all_time_cost(tmp_path): + """The series sums cost_usd across all non-seed energy rows.""" + conn = _make_db(tmp_path) + _seed_spend_rows(conn, n=3, cost_usd=0.05) + _seed_spend_rows(conn, n=2, cost_usd=0.10) + + series = cumulative_spend_series(conn, CFG) + + assert series["total_spend_usd"] == pytest.approx(0.35) + assert series["priced_rows"] == 5 + assert series["total_rows"] == 5 + + +def test_cumulative_spend_series_excludes_seed_reference(tmp_path): + """seed_reference rows do not inflate 'real traffic' spend.""" + conn = _make_db(tmp_path) + _seed_spend_rows(conn, n=2, cost_usd=0.05) + _seed_spend_seed_rows(conn, n=5) + + series = cumulative_spend_series(conn, CFG) + + assert series["total_spend_usd"] == pytest.approx(0.10) + assert series["priced_rows"] == 2 + # total_rows is also seed-filtered: the series describes real traffic only, + # so the excluded rows never appear in any of its counts. + assert series["total_rows"] == 2 + + +def test_cumulative_spend_series_per_provider_breakdown(tmp_path): + """Results include per-provider sub-totals.""" + conn = _make_db(tmp_path) + _seed_spend_rows(conn, n=2, cost_usd=1.0, provider="neuralwatt") + _seed_spend_rows(conn, n=3, cost_usd=2.0, provider="openrouter") + + series = cumulative_spend_series(conn, CFG) + + assert series["total_spend_usd"] == pytest.approx(8.0) + by_provider = {e["provider"]: e for e in series["by_provider"]} + assert by_provider["neuralwatt"]["provider_spend_usd"] == pytest.approx(2.0) + assert by_provider["openrouter"]["provider_spend_usd"] == pytest.approx(6.0) + + +def test_cumulative_spend_warning_silent_below_the_observation_floor(tmp_path): + """Never warns on too few priced rows (novelty-or-rate).""" + conn = _make_db(tmp_path) + _seed_spend_rows(conn, n=3, cost_usd=100.0) + + cfg = _cfg_with(cumulative_spend_warn_min_rows=10) + warnings = cumulative_spend_warnings(conn, cfg) + + assert warnings == [] + + +def test_cumulative_spend_warning_silent_below_the_threshold(tmp_path): + """Spend below the warning threshold does not fire.""" + conn = _make_db(tmp_path) + _seed_spend_rows(conn, n=15, cost_usd=0.01) + + cfg = _cfg_with(cumulative_spend_warn_usd=50.0) + warnings = cumulative_spend_warnings(conn, cfg) + + assert warnings == [] + + +def test_cumulative_spend_warning_fires_when_premise_expires(tmp_path): + """Once all-time spend exceeds warn_usd, the warning fires naming the + blend comment whose premise has expired.""" + conn = _make_db(tmp_path) + _seed_spend_rows(conn, n=15, cost_usd=50.0) + + cfg = _cfg_with(cumulative_spend_warn_usd=10.0, + cumulative_spend_warn_min_rows=5) + warnings = cumulative_spend_warnings(conn, cfg) + + premise = [w for w in warnings + if w.startswith("cost-as-tiebreak premise expired")] + assert len(premise) >= 1 + assert "cumulative spend" in premise[0] + assert "$" in premise[0] + + + +def test_scoring_coverage_carries_the_premise_expiry_series(tmp_path): + """scoring_coverage includes both new premise-expiry series and warnings.""" + conn = _make_db(tmp_path) + _seed_models(conn) + # Seed enough proficiency depth to trigger the warning + _seed_proficiency_depth_rows(conn, n=20, category="coding_general", + self_eval_samples=15, outcome_samples=10) + _seed_spend_rows(conn, n=15, cost_usd=50.0) + + cfg = _cfg_with( + cumulative_spend_warn_usd=10.0, + cumulative_spend_warn_min_rows=5, + ) + coverage = scoring_coverage(conn, cfg) + + assert "proficiency_sample_depth" in coverage + assert coverage["proficiency_sample_depth"]["overall_avg_depth"] is not None + assert "cumulative_spend" in coverage + assert coverage["cumulative_spend"]["total_spend_usd"] == pytest.approx(750.0) + assert any( + w.startswith("quality_tolerance premise expired") + for w in coverage["warnings"] + ) + assert any( + w.startswith("cost-as-tiebreak premise expired") + for w in coverage["warnings"] + ) -- 2.49.1