Files
6krrt/tests/test_metrics.py
adlee-was-taken 2c358cd40e test(metrics): stop the plan-pace tests failing on the billing reset day
test_quota_accounts_alarm_plan_pace_warning and _critical hardcoded
billing_reset_day=6 and claimed to hold "on any day". quota_accounts sets
elapsed_fraction to None below 0.02 (day 0 of a period) and skips the pace
rule there on purpose, so on the 6th UTC both tests saw the usage-floor alarm
instead of plan_pace and failed. They went red on 2026-10-06 UTC, on main at
d97588c as well as on this branch, so every gate that day was red.

The tests now take a reset day from _pace_reset_day(today), which always puts
today at least one day into the period (valid days are 1..28). A sweep test
checks every calendar day of a leap and a common year against the same 0.02
guard, using metrics' own _billing_period_start and _next_reset_date.

Product behavior is unchanged: the guard is deliberate.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01KkCGRantZsSwmcFpet6FTa
2026-10-05 21:32:02 -04:00

2928 lines
108 KiB
Python

"""Tests for metrics.py — read-only aggregation helpers.
Every test seeds a throwaway SQLite DB directly from schema.sql, never writes
to the live ``router.db``, and asserts on *actual queried aggregates* rather
than mock-call assertions (to defeat ``misleading_success_output``).
Functions tested:
- quota_accounts
- scoring_coverage
- recent_decisions
- per_model
- verdict_mix
- top_proficiency
Also verifies that ``import dispatcher`` and ``/health`` still work after the
move, and that ``import metrics`` alone succeeds (no circular import).
"""
from __future__ import annotations
import re
import sqlite3
from datetime import date, datetime, timedelta, timezone
from pathlib import Path
from types import SimpleNamespace
import pytest
from starlette.testclient import TestClient
import dispatcher
from config import load_config
from metrics import (
_billing_period_start,
_next_reset_date,
cache_rate_series,
cache_rate_warnings,
capability_ceilings,
capability_demand_warnings,
context_ceilings,
conversation_adoption,
cumulative_spend_series,
cumulative_spend_warnings,
decision_outcome_summary,
demand_ceiling_warnings,
local_energy_summary,
per_model,
pinch_summary,
proficiency_sample_depth_series,
proficiency_sample_depth_warnings,
quota_accounts,
recent_decisions,
rejection_warnings,
scoring_coverage,
selection_coverage,
selection_coverage_warnings,
top_proficiency,
unroutable_models,
verdict_mix,
)
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
CFG = load_config(str(ROOT / "config" / "config.yaml"))
# =============================================================================
# Helpers
# =============================================================================
def _now() -> datetime:
return datetime.now(timezone.utc)
def _make_db(tmp_path: Path, extra_sql: str = "") -> sqlite3.Connection:
"""Create a clean DB seeded from schema.sql, returning a Row-backed conn."""
conn = sqlite3.connect(str(tmp_path / "test.db"))
conn.row_factory = sqlite3.Row
conn.executescript(SCHEMA_SQL + extra_sql)
return conn
def _seed_models(conn: sqlite3.Connection) -> None:
"""Insert routable model rows into an (empty) DB."""
fresh = _now().isoformat()
for model_id, tier, context, cost, vision in (
("cheap", 2, 262128, 0.30, 1),
("dear", 2, 262128, 9.00, 0),
("tiny", 1, 131072, 0.10, 1),
):
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, ?, ?, 192500, 16384, ?, ?,
?, 1, 'standard', 'default', 'full', 'public', 'active',
?)
""",
(model_id, model_id, tier, context, cost, cost / 3, vision, fresh),
)
conn.commit()
def _seed_proficiency(conn: sqlite3.Connection) -> None:
"""Insert proficiency rows for the seeded models."""
for model_id, score in (
("cheap", 0.90),
("dear", 0.95),
("tiny", 0.70),
):
conn.execute(
"""
INSERT INTO proficiency (
model_id, provider, category, blended_score, source, last_updated
) VALUES (?, 'neuralwatt', 'coding_general', ?, 'self_eval_thin', '2026-01-01T00:00:00+00:00')
""",
(model_id, score),
)
conn.commit()
def _seed_energy(conn: sqlite3.Connection) -> None:
now = _now()
recent_rows = [
("cheap", 5.0e-05, 100, 0.25), # 2 days ago
("cheap", 3.0e-05, 200, 0.50), # 2 days ago
("dear", 1.0e-04, 150, 0.75), # 5 days ago
("tiny", 1.0e-05, 50, 1.00), # 10 days ago
]
for (model_id, kwh, tokens, attr) in recent_rows:
conn.execute(
"""
INSERT INTO energy_observations (
model_id, provider, task_category, prompt_tokens,
completion_tokens, energy_kwh, attribution_ratio,
observed_at
) VALUES (?, 'neuralwatt', 'coding_general', 1000, ?, ?, ?, ?)
""",
(model_id, tokens, kwh, attr, (now - timedelta(days=2)).isoformat()),
)
conn.commit()
def _telemetry_provider_cfg(**objective_overrides) -> SimpleNamespace:
"""Build a cfg whose only dispatch provider is telemetry-backed."""
return SimpleNamespace(
objective=SimpleNamespace(**objective_overrides),
dispatch_providers={
"neuralwatt": SimpleNamespace(has_energy_telemetry=True),
},
)
def _pace_reset_day(today: date) -> int:
"""A billing_reset_day that puts ``today`` at least one day into the period.
quota_accounts sets elapsed_fraction to None below 0.02, which is day 0 of a
period, and then skips the pace rule on purpose: a ratio over a few hours of
elapsed time is noise. A hardcoded reset day therefore turns the pace tests
red on that one calendar day every month. They did on 2026-10-06 UTC with
billing_reset_day=6, while their comments claimed "on any day". Valid days
are 1..28.
"""
return today.day - 1 if 2 <= today.day <= 29 else 28
def test_pace_reset_day_is_never_the_reset_day_itself():
"""Every calendar day of a leap year and a common year lands past the 0.02 guard."""
d = date(2024, 1, 1)
while d < date(2026, 1, 1):
reset_day = _pace_reset_day(d)
assert 1 <= reset_day <= 28, d
start = date.fromisoformat(_billing_period_start(reset_day, d))
nxt = date.fromisoformat(_next_reset_date(reset_day, d))
assert (d - start).days / (nxt - start).days >= 0.02, d
d += timedelta(days=1)
# --- Import / no-circular-import smoke tests -----------------------------------
def test_metrics_can_be_imported_alone():
"""metrics.py must not require dispatcher — it *is* the cycle-breaker."""
# If this import raises ImportError (circular), we fail.
import metrics # noqa: F401
def test_dispatcher_imports_after_metrics():
"""importing metrics first, then dispatcher, must not raise."""
# This test runs *after* metrics has already been imported above.
# The import chain is: dispatcher → metrics (one-way).
assert hasattr(dispatcher, "app")
# --- quota_accounts tests -------------------------------------------------------
def test_quota_accounts_returns_correct_top_level_keys(tmp_path):
"""Result has period, accounts, spend, alarm keys."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
conn = _make_db(tmp_path)
now = _now()
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, completion_tokens, observed_at) "
"VALUES ('m', 'neuralwatt', 0.10, 100, ?)",
(now.isoformat(),),
)
conn.commit()
result = quota_accounts(conn, cfg)
assert "period" in result
assert "accounts" in result
assert "spend" in result
assert "alarm" in result
def test_quota_accounts_metered_plan_has_plan_block(tmp_path):
"""metered_plan shape includes plan block with kwh_per_period, used_kwh, used_fraction."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
conn = _make_db(tmp_path)
now = _now()
for kwh in (0.10, 0.15):
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, completion_tokens, observed_at) "
"VALUES ('m', 'neuralwatt', ?, 100, ?)",
(kwh, now.isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
assert acc["shape"] == "metered_plan"
assert acc["plan"]["kwh_per_period"] == 6.25
assert acc["plan"]["used_kwh"] == pytest.approx(0.25)
assert acc["plan"]["used_fraction"] == pytest.approx(0.25 / 6.25)
def test_quota_accounts_prepaid_credit_has_pool_block(tmp_path):
"""prepaid_credit shape includes pool block with balance from provider_balance_observations."""
conn = _make_db(tmp_path)
now = _now()
cfg = SimpleNamespace(
objective=SimpleNamespace(),
dispatch_providers={
"neuralwatt": SimpleNamespace(
has_energy_telemetry=False,
balance_url="https://example.com/credits",
),
},
)
conn.execute(
"INSERT INTO provider_balance_observations "
"(provider, balance_usd, observed_at) "
"VALUES ('neuralwatt', ?, ?)",
(50.0, now.isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
assert acc["shape"] == "prepaid_credit"
assert acc["pool"]["balance_usd"] == pytest.approx(50.0)
assert acc["pool"]["age_seconds"] is not None
def test_quota_accounts_self_hosted_shape(tmp_path):
"""Provider named ollama-local gets self_hosted shape."""
conn = _make_db(tmp_path)
cfg = SimpleNamespace(
objective=SimpleNamespace(),
dispatch_providers={
"ollama-local": SimpleNamespace(has_energy_telemetry=False),
},
)
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
assert acc["shape"] == "self_hosted"
assert acc["provider"] == "ollama-local"
def test_quota_accounts_unmetered_shape(tmp_path):
"""Bare provider (no telemetry, no plan, no balance_url) gets unmetered shape."""
conn = _make_db(tmp_path)
cfg = SimpleNamespace(
objective=SimpleNamespace(),
dispatch_providers={
"bare": SimpleNamespace(has_energy_telemetry=False),
},
)
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
assert acc["shape"] == "unmetered"
def test_quota_accounts_period_with_billing_reset(tmp_path):
"""billing_reset_day=6 sets source=billing_reset_day with start and next_reset dates."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
conn = _make_db(tmp_path)
result = quota_accounts(conn, cfg)
assert result["period"]["source"] == "billing_reset_day"
assert result["period"]["start"] is not None
assert result["period"]["next_reset"] is not None
def test_quota_accounts_period_without_reset_day(tmp_path):
"""No billing_reset_day sets source=30d_rolling with next_reset=None."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
conn = _make_db(tmp_path)
result = quota_accounts(conn, cfg)
assert result["period"]["source"] == "30d_rolling"
assert result["period"]["next_reset"] is None
def test_quota_accounts_spend_block(tmp_path):
"""Spend aggregates cost_usd from energy rows per provider."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
conn = _make_db(tmp_path)
now = _now()
for cost in (0.05, 0.03):
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, cost_usd, completion_tokens, observed_at) "
"VALUES ('m', 'neuralwatt', 0.001, ?, 100, ?)",
(cost, now.isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
assert result["spend"]["total_usd"] == pytest.approx(0.08)
assert "by_provider_usd" in result["spend"]
assert result["spend"]["by_provider_usd"]["neuralwatt"] == pytest.approx(0.08)
def test_quota_accounts_burn_rate_computed(tmp_path):
"""Burn rate computed from allowance_remaining_usd series over 3 hours."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
conn = _make_db(tmp_path)
now = _now()
rows = [
("m", 1.60, now - timedelta(hours=3)),
("m", 1.20, now - timedelta(hours=2)),
("m", 0.80, now - timedelta(hours=1)),
("m", 0.60, now - timedelta(minutes=30)),
]
for model_id, balance, ts in rows:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
(model_id, balance, ts.isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
assert acc["burn"]["burn_rate_usd_per_hour"] == pytest.approx(0.4)
assert acc["burn"]["projected_hours_remaining"] == pytest.approx(1.5)
def test_quota_accounts_top_up_resets_segment(tmp_path):
"""A credit top-up splits the window; burn uses only the latest segment."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
conn = _make_db(tmp_path)
now = _now()
rows = [
("m", 3.00, now - timedelta(hours=5)),
("m", 2.00, now - timedelta(hours=4)),
("m", 1.00, now - timedelta(hours=3)),
("m", 10.00, now - timedelta(hours=2)),
("m", 9.25, now - timedelta(minutes=90)),
("m", 8.50, now - timedelta(minutes=30)),
]
for model_id, balance, ts in rows:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
(model_id, balance, ts.isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
# Burn ~1.0 from post-top-up segment, not ~7.0 from overall MAX-MIN
assert acc["burn"]["burn_rate_usd_per_hour"] == pytest.approx(1.0)
def test_quota_accounts_min_samples_guard(tmp_path):
"""Segment with only 2 samples cannot compute a burn rate."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
conn = _make_db(tmp_path)
now = _now()
for bal, hours_ago in ((10.0, 2), (9.0, 0.5)):
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
("m", bal, (now - timedelta(hours=hours_ago)).isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
assert acc["burn"] is None
def test_quota_accounts_min_hours_guard(tmp_path):
"""3 samples spanning only 10 minutes cannot compute a burn rate."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
conn = _make_db(tmp_path)
now = _now()
for value, minutes_ago in ((10.0, 10), (9.0, 7), (8.0, 1)):
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
("m", value, (now - timedelta(minutes=minutes_ago)).isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
assert acc["burn"] is None
def test_quota_accounts_all_null_allowance(tmp_path):
"""All allowance_remaining_usd NULL -> burn is None, pool is None for metered_plan."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
conn = _make_db(tmp_path)
now = _now()
for kwh, hours_ago in ((2.0, 2), (3.0, 1)):
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', ?, NULL, ?)",
("m", kwh, (now - timedelta(hours=hours_ago)).isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
acc = result["accounts"][0]
assert acc["burn"] is None
assert acc["pool"] is None
def test_quota_accounts_interleaved_providers_independent(tmp_path):
"""Two providers' allowance series compute independently despite interleaved rows."""
conn = _make_db(tmp_path)
cfg = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6),
dispatch_providers={
"neuralwatt": SimpleNamespace(has_energy_telemetry=True),
"openrouter": SimpleNamespace(
has_energy_telemetry=False,
balance_url="https://openrouter.ai/api/v1/credits",
),
},
)
now = _now()
# NeuralWatt telemetry: burn 1.0 USD/h
for bal, at in (
(5.0, now - timedelta(hours=4)),
(4.0, now - timedelta(hours=3)),
(3.0, now - timedelta(hours=2)),
(2.0, now - timedelta(hours=1)),
):
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
("m", bal, at.isoformat()),
)
# OpenRouter polled: burn 4.0 USD/h
for bal, at in (
(20.0, now - timedelta(minutes=210)),
(16.0, now - timedelta(minutes=150)),
(12.0, now - timedelta(minutes=90)),
(8.0, now - timedelta(minutes=30)),
):
conn.execute(
"INSERT INTO provider_balance_observations "
"(provider, balance_usd, observed_at) "
"VALUES ('openrouter', ?, ?)",
(bal, at.isoformat()),
)
conn.commit()
result = quota_accounts(conn, cfg)
by_provider = {a["provider"]: a for a in result["accounts"]}
assert by_provider["neuralwatt"]["burn"]["burn_rate_usd_per_hour"] == pytest.approx(1.0)
assert by_provider["openrouter"]["burn"]["burn_rate_usd_per_hour"] == pytest.approx(4.0)
assert by_provider["neuralwatt"]["shape"] == "metered_plan"
assert by_provider["openrouter"]["shape"] == "prepaid_credit"
def test_quota_accounts_alarm_plan_pace_warning(tmp_path):
"""Usage pace > 1.25x triggers plan_pace alarm."""
cfg = _telemetry_provider_cfg(
plan_kwh_per_period=6.25, billing_reset_day=_pace_reset_day(_now().date())
)
conn = _make_db(tmp_path)
now = _now()
# Seed 8.0 kWh to guarantee used_fraction > 1.25 * elapsed_fraction; the reset
# day keeps today past the 0.02 elapsed guard on every calendar day
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, completion_tokens, observed_at) "
"VALUES ('m', 'neuralwatt', 8.0, 100, ?)",
(now.isoformat(),),
)
conn.commit()
result = quota_accounts(conn, cfg)
assert result["alarm"] is not None
assert result["alarm"]["kind"] == "plan_pace"
assert result["alarm"]["severity"] in ("warning", "critical")
def test_quota_accounts_alarm_stale_reading(tmp_path):
"""A balance reading 4+ hours old triggers stale_reading alarm."""
conn = _make_db(tmp_path)
cfg = SimpleNamespace(
objective=SimpleNamespace(),
dispatch_providers={
"neuralwatt": SimpleNamespace(
has_energy_telemetry=False,
balance_url="https://example.com/credits",
),
},
)
now = _now()
conn.execute(
"INSERT INTO provider_balance_observations "
"(provider, balance_usd, observed_at) "
"VALUES ('neuralwatt', 50.0, ?)",
((now - timedelta(hours=5)).isoformat(),),
)
conn.commit()
result = quota_accounts(conn, cfg)
assert result["alarm"] is not None
assert result["alarm"]["kind"] == "stale_reading"
def test_quota_accounts_alarm_plan_pace_critical(tmp_path):
"""Usage pace > 2.0 triggers critical severity."""
cfg = _telemetry_provider_cfg(
plan_kwh_per_period=6.25, billing_reset_day=_pace_reset_day(_now().date())
)
conn = _make_db(tmp_path)
now = _now()
# Seed 15 kWh to guarantee pace > 2.0; the reset day keeps today past the
# 0.02 elapsed guard on every calendar day
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, completion_tokens, observed_at) "
"VALUES ('m', 'neuralwatt', 15.0, 100, ?)",
(now.isoformat(),),
)
conn.commit()
result = quota_accounts(conn, cfg)
assert result["alarm"]["severity"] == "critical"
def test_quota_accounts_empty_db(tmp_path):
"""Empty DB: zero spend, zero energy, accounts match provider count."""
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
conn = _make_db(tmp_path)
result = quota_accounts(conn, cfg)
assert len(result["accounts"]) == 1
assert result["accounts"][0]["spend_usd"]["period"] == 0.0
assert result["accounts"][0]["energy"]["kwh_30d"] == 0.0
assert result["accounts"][0]["energy"]["calls_30d"] == 0
def test_local_energy_summary_reset_date_is_billing_period_start(tmp_path):
"""local_energy_summary mirrors the same reset_date/window_start split."""
cfg = SimpleNamespace(
local_energy=SimpleNamespace(enabled=True),
objective=SimpleNamespace(billing_reset_day=6),
)
conn = _make_db(tmp_path)
now = datetime.now(timezone.utc)
today = now.date()
reset_day = 6
if today.day >= reset_day:
expected_period_start = date(today.year, today.month, reset_day).isoformat()
else:
if today.month == 1:
expected_period_start = date(today.year - 1, 12, reset_day).isoformat()
else:
expected_period_start = date(today.year, today.month - 1, reset_day).isoformat()
expected_rolling_start = (today - timedelta(days=30)).isoformat()
result = local_energy_summary(conn, cfg)
assert result["reset_date"] == expected_period_start
assert result["window_start_30d"] == expected_rolling_start
cfg_unconfigured = SimpleNamespace(
local_energy=SimpleNamespace(enabled=True),
objective=SimpleNamespace(),
)
result_unconfigured = local_energy_summary(conn, cfg_unconfigured)
assert "reset_date" in result_unconfigured
assert result_unconfigured["reset_date"] is None
assert result_unconfigured["window_start_30d"] == expected_rolling_start
def test_pinch_summary_reset_date_is_billing_period_start(tmp_path):
"""pinch_summary mirrors the same reset_date/window_start split."""
cfg = SimpleNamespace(
pinch=SimpleNamespace(enabled=True),
objective=SimpleNamespace(assumed_cache_rate=0.917, billing_reset_day=6),
)
conn = _make_db(tmp_path)
now = datetime.now(timezone.utc)
today = now.date()
reset_day = 6
if today.day >= reset_day:
expected_period_start = date(today.year, today.month, reset_day).isoformat()
else:
if today.month == 1:
expected_period_start = date(today.year - 1, 12, reset_day).isoformat()
else:
expected_period_start = date(today.year, today.month - 1, reset_day).isoformat()
expected_rolling_start = (today - timedelta(days=30)).isoformat()
result = pinch_summary(conn, cfg)
assert result["reset_date"] == expected_period_start
assert result["window_start_30d"] == expected_rolling_start
cfg_unconfigured = SimpleNamespace(
pinch=SimpleNamespace(enabled=True),
objective=SimpleNamespace(assumed_cache_rate=0.917),
)
result_unconfigured = pinch_summary(conn, cfg_unconfigured)
assert "reset_date" in result_unconfigured
assert result_unconfigured["reset_date"] is None
assert result_unconfigured["window_start_30d"] == expected_rolling_start
# --- _next_reset_date helper --------------------------------------------------
def test_next_reset_date_this_month():
"""When today is before the anchor day, reset lands this month."""
from datetime import date
assert _next_reset_date(6, date(2026, 9, 1)) == "2026-09-06"
def test_next_reset_date_rolls_to_next_month():
"""When today's day-of-month is on or past the anchor, roll forward."""
from datetime import date
assert _next_reset_date(6, date(2026, 9, 6)) == "2026-10-06"
assert _next_reset_date(6, date(2026, 9, 15)) == "2026-10-06"
def test_next_reset_date_handles_december_rollover():
"""December rolls over to January of the next year."""
from datetime import date
assert _next_reset_date(6, date(2026, 12, 7)) == "2027-01-06"
assert _next_reset_date(28, date(2026, 12, 29)) == "2027-01-28"
# --- scoring_coverage tests ---------------------------------------------------
def test_scoring_coverage_has_all_keys(tmp_path):
"""The return dict always has these keys, even when empty."""
conn = _make_db(tmp_path)
# Seed routable models
_seed_models(conn)
result = scoring_coverage(conn, CFG)
assert "routable_models" in result
assert "with_energy_data" in result
assert "with_proficiency_data" in result
assert "quota" in result
assert "warnings" in result
def test_scoring_coverage_warning_when_no_energy(tmp_path):
"""Models with no seed_reference observations produce a warning."""
conn = _make_db(tmp_path)
_seed_models(conn)
# No energy_observations rows at all
result = scoring_coverage(conn, CFG)
warning_texts = result["warnings"]
assert any("no reference-workload observations" in w for w in warning_texts)
assert result["with_energy_data"] == 0
def test_scoring_coverage_no_warning_when_full_coverage(tmp_path):
"""When every routable model has energy + proficiency, no warnings."""
conn = _make_db(tmp_path)
_seed_models(conn)
# Seed SEED_CATEGORY energy observations
now = _now()
for model in ["cheap", "dear"]:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, task_category, prompt_tokens, "
"completion_tokens, energy_kwh, attribution_ratio, observed_at) "
"VALUES (?, 'neuralwatt', 'seed_reference', 1000, 100, 0.001, 0.25, ?)",
(model, now.isoformat()),
)
conn.execute(
"INSERT INTO proficiency "
"(model_id, provider, category, blended_score, source, last_updated) "
"VALUES (?, 'neuralwatt', 'coding_general', 0.9, 'self_eval', ?)",
(model, now.isoformat()),
)
conn.commit()
result = scoring_coverage(conn, CFG)
# 'tiny' has no data so we expect warnings. Let's also add 'tiny'.
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, task_category, prompt_tokens, "
"completion_tokens, energy_kwh, attribution_ratio, observed_at) "
"VALUES ('tiny', 'neuralwatt', 'seed_reference', 1000, 100, 0.001, 0.25, ?)",
(now.isoformat(),),
)
conn.execute(
"INSERT INTO proficiency "
"(model_id, provider, category, blended_score, source, last_updated) "
"VALUES ('tiny', 'neuralwatt', 'coding_general', 0.7, 'self_eval', ?)",
(now.isoformat(),),
)
conn.commit()
result = scoring_coverage(conn, CFG)
assert result["with_energy_data"] == 3
assert result["with_proficiency_data"] == 3
assert len(result["warnings"]) == 0
def test_scoring_coverage_empty_db(tmp_path):
"""Empty DB: 0 routable, no warnings, quota=None."""
conn = _make_db(tmp_path)
result = scoring_coverage(conn, CFG)
assert result["routable_models"] == 0
assert result["with_energy_data"] == 0
assert result["with_proficiency_data"] == 0
# quota returns None because plan_kwh_per_period may be None in default cfg
# (it is 6.25 by default, but let's just check the structure)
assert result["warnings"] == []
def test_scoring_coverage_warns_per_provider_staleness(tmp_path):
"""A stale second provider must produce a warning naming that provider."""
conn = _make_db(tmp_path)
fresh = _now().isoformat()
stale = (_now() - timedelta(days=2)).isoformat()
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES ('fresh-model', 'neuralwatt', 'fresh-model', 2, 262128, 192500, 16384,
0.30, 0.10, 1, 1, 'standard', 'default', 'full', 'public', 'active', ?)
""",
(fresh,),
)
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES ('old-model', 'other-provider', 'old-model', 2, 262128, 192500, 16384,
0.30, 0.10, 1, 1, 'standard', 'default', 'full', 'public', 'active', ?)
""",
(stale,),
)
conn.commit()
result = scoring_coverage(conn, CFG)
neuralwatt_warnings = [w for w in result["warnings"] if "days ago" in w]
assert any("[other-provider]" in w for w in neuralwatt_warnings)
assert not any("[neuralwatt]" in w for w in neuralwatt_warnings)
# One formatted number. This read "1.0.8 days ago" live, because an int
# and a rounded fraction joined by a dot keeps the fraction's own "0.".
assert re.search(r"catalog last polled \d+\.\d days ago", neuralwatt_warnings[0])
_ADMIN_TABLE_SQL = """
CREATE TABLE IF NOT EXISTS admin_model_overrides (
model_id TEXT NOT NULL,
provider TEXT NOT NULL,
availability TEXT NOT NULL,
reason TEXT,
updated_at TEXT NOT NULL,
PRIMARY KEY (model_id, provider)
);
CREATE INDEX IF NOT EXISTS idx_admin_model_overrides_availability
ON admin_model_overrides (availability);
"""
def test_context_ceilings_excludes_admin_deprecated_models(tmp_path):
"""Admin-deprecated models are excluded from context ceilings even when the
raw ``models.availability`` column still reads ``active``.
Regression for the 2026-09-01 outage: ``admin_model_overrides`` carries
the override while the catalog row stays active, so ``exclude_deprecated``
on the raw row is not enough.
"""
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
fresh = _now().isoformat()
for model_id, eff_ctx in (("big", 200000), ("small", 100000)):
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, 3, ?, ?, 16384, 0.30, 0.10,
1, 1, 'standard', 'default', 'full', 'public', 'active',
?)
""",
(model_id, model_id, eff_ctx, eff_ctx, fresh),
)
conn.execute(
"""
INSERT INTO admin_model_overrides (
model_id, provider, availability, reason, updated_at
) VALUES (?, 'neuralwatt', 'deprecated', 'test', ?)
""",
("big", fresh),
)
conn.commit()
ctx = context_ceilings(conn, CFG)
key = (3, "interactive")
assert key in ctx
assert ctx[key]["ceiling"] == 100000
assert ctx[key]["count"] == 1
# --- capability sub-ceiling demand warnings (2026-09-04 incident) ---------------
def _insert_model_row(
conn: sqlite3.Connection,
*,
model_id: str,
tier: int,
effective_context_window: int,
supports_vision: int,
supports_json_mode: int = 1,
) -> None:
"""Insert one active, routable catalog row with explicit capability flags.
``_seed_models`` covers the common case but hardcodes its capability
flags and effective window; the incident-reproduction tests need both
per-row.
"""
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, ?, ?, ?, 16384, 0.30, 0.10,
?, ?, 'standard', 'default', 'full', 'public', 'active',
?)
""",
(
model_id,
model_id,
tier,
effective_context_window,
effective_context_window,
supports_vision,
supports_json_mode,
_now().isoformat(),
),
)
conn.commit()
def _deprecate_model(conn: sqlite3.Connection, model_id: str) -> None:
"""Insert an admin override marking *model_id* deprecated (reason: cost)."""
conn.execute(
"INSERT INTO admin_model_overrides "
"(model_id, provider, availability, reason, updated_at) "
"VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)",
(model_id, _now().isoformat()),
)
conn.commit()
def _insert_rejected_decision(
conn: sqlite3.Connection,
*,
tier: int,
reason: str,
observed_at: str,
tokens: int = 200000,
images: int = 0,
json_mode: int = 0,
) -> None:
"""Insert a route_decisions row that selected no model (a 422 rejection).
``kind='chat'`` so the demand queries (which filter on kind) see it;
``latency_tolerance='interactive'`` to match the tier-1 bucket key the
ceiling checks use.
"""
conn.execute(
"""
INSERT INTO route_decisions (
observed_at, kind, task_category, task_tier,
required_context_tokens, confidence, classifier_ms,
classification_source, latency_tolerance, candidates_considered,
selected_model, selected_provider, rejected_reason,
session_key, tools, images, json_mode, streamed
) VALUES (?, 'chat', 'coding_general', ?, ?, NULL, NULL,
'override', 'interactive', 0, NULL, NULL, ?,
'sess', 0, ?, ?, 0)
""",
(observed_at, tier, tokens, reason, images, json_mode),
)
conn.commit()
def _seed_incident_state(conn: sqlite3.Connection) -> str:
"""Seed the 2026-09-04 incident model state; returns the shared 'now' stamp.
Three tier-1 rows: kimi-k3 and kimi-k3-fast — the only vision-capable
large-context models — deprecated through admin overrides, and
deepseek-v4-flash (no vision, 262,128 effective tokens) left active.
The overall tier-1 interactive ceiling stays high while the vision
sub-ceiling collapses to zero, which is exactly the shape every
pre-2026-09-05 ceiling check was blind to.
"""
for model_id, vision, eff_ctx in (
("kimi-k3", 1, 782324),
("kimi-k3-fast", 1, 782324),
("deepseek-v4-flash", 0, 262128),
):
_insert_model_row(
conn,
model_id=model_id,
tier=1,
effective_context_window=eff_ctx,
supports_vision=vision,
)
for model_id in ("kimi-k3", "kimi-k3-fast"):
_deprecate_model(conn, model_id)
return _now().isoformat()
def test_capability_demand_warning_incident_reproduction(tmp_path):
"""The 2026-09-04 incident state warns on the vision sub-ceiling while
every existing ``(tier, latency_tolerance)`` check stays silent.
In the incident, kimi-k3 and kimi-k3-fast — the only vision-capable
tier-1 models with a large context — were deprecated through admin
overrides, so a 200,000-token image request was unservable even though
the overall tier-1 ceiling (262,128, from non-vision deepseek-v4-flash)
comfortably covered that demand. The new capability check must fire;
the existing demand_ceiling_warnings must NOT — that silence is what
hid the incident for 19 hours.
"""
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
ts = _seed_incident_state(conn)
_insert_rejected_decision(
conn,
tier=1,
reason=(
"context >= 242486 tokens; "
"vision-capable model (request carries image(s))"
),
observed_at=ts,
tokens=200000,
images=1,
)
# The overall tier-1 interactive ceiling is untouched by the deprecations.
ctx = context_ceilings(conn, CFG)
assert ctx[(1, "interactive")]["ceiling"] == 262128
# The vision-gated subset collapsed: no candidate survives the gate.
cap_ctx = capability_ceilings(conn, CFG)
assert cap_ctx["vision"][(1, "interactive")]["ceiling"] == 0
assert cap_ctx["vision"][(1, "interactive")]["count"] == 0
# deepseek-v4-flash still serves the json_mode subset.
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 262128
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
assert len(warnings) == 1
w = warnings[0]
assert "vision" in w
assert "tier 1" in w
assert "ceiling (0)" in w
assert "200000" in w
# The pre-existing check must stay silent on the same state: the overall
# tier-1 ceiling (262128) still exceeds the observed max demand (200000).
assert demand_ceiling_warnings(conn, CFG, ctx, [1]) == []
def test_capability_demand_no_demand_no_warning(tmp_path):
"""A state with demand but no capability-flagged demand warns about nothing.
Requests exist at the tier — far above every ceiling — but none carry
images or ask for JSON mode, so both capability subsets are unexercised:
an unexercised sub-ceiling is not a warning.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens",
observed_at=_now().isoformat(),
tokens=999999,
)
assert capability_demand_warnings(conn, CFG, capability_ceilings(conn, CFG)) == []
def test_capability_json_mode_demand_warning(tmp_path):
"""Deprecating the only json_mode-capable model trips the json_mode warning.
Same incident shape as the vision case, different capability dimension:
the json_mode sub-ceiling collapses to zero while the overall ceiling
stays fine, and a json_mode request that cannot be served produces a
warning naming json_mode.
"""
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
_insert_model_row(
conn,
model_id="jm-big",
tier=1,
effective_context_window=782324,
supports_vision=0,
supports_json_mode=1,
)
_insert_model_row(
conn,
model_id="plain-fallback",
tier=1,
effective_context_window=262128,
supports_vision=0,
supports_json_mode=0,
)
_deprecate_model(conn, "jm-big")
_insert_rejected_decision(
conn,
tier=1,
reason=(
"context >= 242486 tokens; json-mode-capable model "
"(json_mode requested)"
),
observed_at=_now().isoformat(),
tokens=200000,
json_mode=1,
)
cap_ctx = capability_ceilings(conn, CFG)
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 0
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
assert len(warnings) == 1
w = warnings[0]
assert "json_mode" in w
assert "tier 1" in w
assert "ceiling (0)" in w
assert "200000" in w
# --- NULL required_context_tokens guards (regression) ---------------------------
#
# route_decisions.required_context_tokens is a nullable column (rows written
# before the column existed, or decisions persisted without a measured token
# count). SQLite MAX() over an all-NULL group returns NULL, not 0, so a
# window whose every row lacks a token count yields observed_max=None —
# which used to raise ``None > ceiling`` TypeError inside the demand-ceiling
# checks, and crashed _percentile on the escalation path. Semantics: NULL is
# "no demand recorded for this row", not zero — so a NULL row is excluded
# from comparisons, never fabricated into a warning.
def test_demand_ceiling_all_null_7d_window_no_crash_no_warning(tmp_path):
"""A tier whose 7d rows are all NULL does not crash demand_ceiling_warnings.
Site A regression: ``MAX(required_context_tokens)`` over an all-NULL
window returns NULL, so ``observed_max`` used to become None and the
``None > ceiling`` comparison raised TypeError. Seeding the tier's
demand proof in the 30d window (one non-NULL row at 20 days ago) plus
12 NULL rows inside 7d puts the tier in ``tiers_with_demand`` while
keeping the observed-max window at 7d — exactly the shape that used to
crash. With the guard, the comparison is skipped: no crash, and no
false "0 > ceiling" warning either.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
old = (_now() - timedelta(days=20)).isoformat()
for _ in range(12):
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens",
observed_at=ts,
tokens=None,
)
# Non-NULL demand inside 30d (outside 7d) proves the tier has demand,
# while every row inside the 7d observed-max window stays NULL.
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens",
observed_at=old,
tokens=200000,
)
ctx = context_ceilings(conn, CFG)
assert ctx[(2, "interactive")]["ceiling"] > 0
# Old code: None > ceiling -> TypeError. New code: comparison skipped.
assert demand_ceiling_warnings(conn, CFG, ctx, [1, 2]) == []
def test_demand_ceiling_escalation_percentile_skips_null_tokens(tmp_path):
"""NULL tokens in the escalation p95 query are excluded, not sorted.
Site B regression: the escalation-hazard probe fetched every tier's 7d
``required_context_tokens`` raw and fed the list to _percentile, whose
``sorted()`` raises TypeError the moment a None meets an int. A mixed
NULL/non-NULL 7d window for tier 1 used to crash; after the fix the
NULL row is filtered out and the percentile is computed over the
remaining values. 100000 < the tier-2 ceiling, so no escalation
warning fires either — the assertion proves both.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
_insert_rejected_decision(
conn, tier=1, reason="context >= 999999 tokens", observed_at=ts, tokens=None
)
_insert_rejected_decision(
conn, tier=1, reason="context >= 999999 tokens", observed_at=ts, tokens=100000
)
ctx = context_ceilings(conn, CFG)
# Escalation enabled in the default CFG; tier-2 bucket must exist and
# carry a non-zero ceiling or the percentile probe would be skipped.
assert getattr(CFG.escalation, "enabled", False)
assert ctx[(2, "interactive")]["ceiling"] > 0
# Old code: _percentile([None, 100000], 95) -> TypeError on sorted().
# New code: None filtered out, p95 = 100000 < ceiling, no warning.
assert demand_ceiling_warnings(conn, CFG, ctx, [1, 2]) == []
def test_capability_demand_all_null_7d_window_no_crash_no_warning(tmp_path):
"""A capability-gated tier whose 7d rows are all NULL does not crash.
Site C regression: same all-NULL MAX shape as the tier-wide demand
check, but inside capability_demand_warnings' vision dimension. One
vision-carrying row with a non-NULL token count at 20 days ago puts
the tier's vision bucket past the unexercised-silence guard, and 12
NULL-token vision rows inside 7d keep the observed-max window at 7d —
the exact shape that used to raise ``None > ceiling`` TypeError. With
the guard, no crash and no fabricated warning.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
old = (_now() - timedelta(days=20)).isoformat()
for _ in range(12):
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens; vision-capable model",
observed_at=ts,
tokens=None,
images=1,
)
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens; vision-capable model",
observed_at=old,
tokens=200000,
images=1,
)
cap_ctx = capability_ceilings(conn, CFG)
assert cap_ctx["vision"][(2, "interactive")]["ceiling"] > 0
# Old code: None > ceiling -> TypeError. New code: comparison skipped.
assert capability_demand_warnings(conn, CFG, cap_ctx) == []
# --- rejection warnings (reactive rejection detector) --------------------------
def test_rejection_warnings_zero_rejections(tmp_path):
"""Zero NULL-selection rows produce no rejection warnings.
The absence of rejections is not a signal worth reporting — the
detector watches rates and new patterns, not mere existence.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
assert rejection_warnings(conn, CFG) == []
def test_rejection_warnings_novel_group_at_incident_volume_fires(tmp_path):
"""The 2026-09-04 incident shape — a novel group of only 2 — must fire.
The incident produced exactly 2 image rejections, below the routine
noise floor (about 3/hour on this deployment), so no rate threshold can
catch it; the novelty prong — absent from the 24h baseline, count
>= 2 — is what must fire here. Both rows share one timestamp so the
rendered most-recent value is deterministic.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
for reason in (
"tier >= 1; context >= 242486 tokens; interactive; "
"vision-capable model (request carries image(s))",
"tier >= 1; context >= 195000 tokens; interactive; "
"vision-capable model (request carries image(s))",
):
_insert_rejected_decision(conn, tier=1, reason=reason, observed_at=ts)
warnings = rejection_warnings(conn, CFG)
assert len(warnings) == 1
w = warnings[0]
assert w.startswith("new rejection pattern:")
# Token counts collapse: both raw reasons normalize to "context >= N tokens".
assert "2 rejection(s)" in w
assert "context >= N tokens" in w
# The tier is rendered from the structured task_tier column, not the string.
assert "(tier 1;" in w
assert ts in w
def test_rejection_warnings_familiar_group_below_threshold_silent(tmp_path):
"""Routine over-large tier-3 traffic — 3/hour and familiar — stays silent.
Regression for the measured routine floor: this deployment routinely
sees about 3 rejections per hour of ordinary over-large tier-3 requests
that behave as designed. The group is familiar (present in the 24h
baseline window) and below ``rejection_warning_min_count`` (6), so
neither the novelty prong nor the rate prong fires.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
now = _now()
# Baseline occurrence: 2h ago — inside the 24h baseline, outside the 1h
# alert window.
_insert_rejected_decision(
conn,
tier=3,
reason="tier >= 3; context >= 40000 tokens; interactive",
observed_at=(now - timedelta(hours=2)).isoformat(),
)
# The live-DB-measured routine hourly volume, with distinct token counts
# so digit normalization does real grouping work.
for tokens in (30000, 31000, 32000):
_insert_rejected_decision(
conn,
tier=3,
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
observed_at=now.isoformat(),
)
assert rejection_warnings(conn, CFG) == []
def test_rejection_warnings_familiar_group_at_threshold_fires(tmp_path):
"""A familiar group at ``rejection_warning_min_count`` (6) fires as a rate."""
conn = _make_db(tmp_path)
_seed_models(conn)
now = _now()
_insert_rejected_decision(
conn,
tier=3,
reason="tier >= 3; context >= 40000 tokens; interactive",
observed_at=(now - timedelta(hours=2)).isoformat(),
)
for tokens in (30000, 31000, 32000, 33000, 34000, 35000):
_insert_rejected_decision(
conn,
tier=3,
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
observed_at=now.isoformat(),
)
warnings = rejection_warnings(conn, CFG)
assert len(warnings) == 1
w = warnings[0]
assert w.startswith("rejection rate:")
assert "6 rejection(s)" in w
def test_rejection_warnings_novel_single_rejection_silent(tmp_path):
"""A single novel rejection stays silent.
A genuinely impossible one-off request SHOULD 422; presence of one
rejection is not a signal. n=1 is below NOVEL_GROUP_MIN_COUNT (2).
"""
conn = _make_db(tmp_path)
_seed_models(conn)
_insert_rejected_decision(
conn,
tier=1,
reason=(
"tier >= 1; context >= 242486 tokens; interactive; "
"vision-capable model (request carries image(s))"
),
observed_at=_now().isoformat(),
)
assert rejection_warnings(conn, CFG) == []
def test_rejection_warnings_groups_by_structured_tier(tmp_path):
"""Tier-1 and tier-3 rejections of the same normalized shape never merge.
Regression for the sibling finding of the incident review: digit
normalization erases the tier embedded in the reason string ("tier >= 1"
and "tier >= 3" both become "tier >= N"), so grouping by normalized
string alone would produce ONE merged count-4 group, hiding which
tier's candidate set went empty. The discriminator is the structured
``task_tier`` column, rendered back into the body as "(tier N; ...)".
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
for tier, tokens in ((1, 242486), (1, 50000), (3, 229702), (3, 30000)):
_insert_rejected_decision(
conn,
tier=tier,
reason=f"tier >= {tier}; context >= {tokens} tokens; interactive",
observed_at=ts,
)
warnings = rejection_warnings(conn, CFG)
assert len(warnings) == 2
tier1 = [w for w in warnings if "(tier 1;" in w]
tier3 = [w for w in warnings if "(tier 3;" in w]
assert len(tier1) == 1
assert len(tier3) == 1
assert "2 rejection(s)" in tier1[0]
assert "2 rejection(s)" in tier3[0]
def test_rejection_warnings_old_rejections_not_counted(tmp_path):
"""A rejection outside both windows is not counted at all.
Two same-shape rows 3 days old reach the novelty floor of 2, so the
only thing that keeps them silent is their age — outside the 1h alert
window and beyond the 25h total pull window they are not even read.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
for tokens in (242486, 195000):
_insert_rejected_decision(
conn,
tier=1,
reason=f"tier >= 1; context >= {tokens} tokens; interactive",
observed_at=(_now() - timedelta(days=3)).isoformat(),
)
assert rejection_warnings(conn, CFG) == []
def test_rejection_warnings_custom_window(tmp_path):
"""``rejection_warning_window_hours`` widens the alert window.
Two same-shape rows 2h old sit outside the default 1h window but
inside a 3h window; with the knob set they form a novel group at the
incident volume and must fire.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
two_hours_ago = (_now() - timedelta(hours=2)).isoformat()
for tokens in (242486, 195000):
_insert_rejected_decision(
conn,
tier=1,
reason=(
f"tier >= 1; context >= {tokens} tokens; interactive; "
"vision-capable model (request carries image(s))"
),
observed_at=two_hours_ago,
)
wide_cfg = SimpleNamespace(
objective=SimpleNamespace(
rejection_warning_window_hours=3,
rejection_warning_baseline_hours=24,
rejection_warning_min_count=6,
)
)
warnings = rejection_warnings(conn, wide_cfg)
assert len(warnings) == 1
assert warnings[0].startswith("new rejection pattern:")
assert "2 rejection(s)" in warnings[0]
def test_scoring_coverage_includes_new_warnings(tmp_path):
"""``scoring_coverage`` surfaces both new warning classes on the incident state.
The 2026-09-04 incident produced two image rejections against a catalog
whose vision sub-ceiling had collapsed. Both the capability demand
warning and the novel rejection pattern must reach the same warnings
list that /metrics, the admin portal and the TUI read.
"""
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
ts = _seed_incident_state(conn)
for tokens in (242486, 195000):
_insert_rejected_decision(
conn,
tier=1,
reason=(
f"context >= {tokens} tokens; "
"vision-capable model (request carries image(s))"
),
observed_at=ts,
images=1,
)
result = scoring_coverage(conn, CFG)
warnings = result["warnings"]
assert any("vision" in w for w in warnings)
assert any(w.startswith("new rejection pattern:") for w in warnings)
# --- recent_decisions tests ---------------------------------------------------
def test_recent_decisions_returns_rows_in_desc_order(tmp_path, monkeypatch):
"""Rows come back ordered by id DESC, and selected_provider is included."""
conn = _make_db(tmp_path)
# Seed two models
_seed_models(conn)
# Create fake dispatcher state to use TestClient, but we'll insert
# route_decisions rows directly and call recent_decisions(conn).
conn.execute(
"""
INSERT INTO route_decisions (
observed_at, kind, task_category, task_tier, required_context_tokens,
confidence, classifier_ms, classification_source, latency_tolerance,
candidates_considered, selected_model, selected_provider,
runner_up_models, est_cost_usd, est_proficiency,
session_key, tools, images, json_mode, streamed
) VALUES (?, 'route', 'coding_general', 2, 100, 0.95, 200,
'classifier', 'interactive', 5, 'cheap', 'neuralwatt',
'[{"model_id":"dear","provider":"neuralwatt"}]',
0.001, 0.9, 'abc123', 0, 0, 0, 0)
""",
(datetime.now(timezone.utc).isoformat(),),
)
conn.execute(
"""
INSERT INTO route_decisions (
observed_at, kind, task_category, task_tier, required_context_tokens,
confidence, classifier_ms, classification_source, latency_tolerance,
candidates_considered, selected_model, selected_provider,
runner_up_models, est_cost_usd, est_proficiency,
session_key, tools, images, json_mode, streamed
) VALUES (?, 'route', 'docs_writing', 1, 50, 0.88, 150,
'classifier', 'interactive', 3, 'dear', 'neuralwatt',
NULL, 0.005, 0.95, 'def456', 0, 0, 0, 0)
""",
(datetime.now(timezone.utc).isoformat(),),
)
conn.commit()
rows = recent_decisions(conn, limit=50)
assert len(rows) == 2
# DESC order: dear (id=2) first
assert rows[0]["selected_model"] == "dear"
assert rows[0]["kind"] == "route"
assert rows[0]["selected_provider"] == "neuralwatt"
assert rows[1]["selected_model"] == "cheap"
def test_recent_decisions_respects_limit(tmp_path):
"""limit=1 should return only one row regardless of DB content."""
conn = _make_db(tmp_path)
_seed_models(conn)
now = _now().isoformat()
for i in range(5):
conn.execute(
"INSERT INTO route_decisions "
"(observed_at, kind, selected_model, selected_provider) "
"VALUES (?, 'route', 'm', 'neuralwatt')",
(now,),
)
conn.commit()
rows = recent_decisions(conn, limit=1)
assert len(rows) == 1
def test_recent_decisions_empty_db(tmp_path):
"""No rows: returns an empty list, not an error."""
conn = _make_db(tmp_path)
assert recent_decisions(conn) == []
# --- per_model tests ----------------------------------------------------------
def test_per_model_aggregates_correctly(tmp_path):
"""Sum/cost/energy/carbon/tokens match hand-computed values."""
conn = _make_db(tmp_path)
now = _now()
rows = [
("cheap", 0.001, 5.0e-05, 2.4e-03, 100, 0.25),
("cheap", 0.002, 3.0e-05, 1.2e-03, 200, 0.50),
("dear", 0.010, 1.0e-04, 5.0e-03, 150, 0.75),
]
for (model_id, cost, kwh, carbon, tokens, attr) in rows:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, cost_usd, energy_kwh, carbon_g_co2eq, "
"completion_tokens, attribution_ratio, observed_at) "
"VALUES (?, 'neuralwatt', ?, ?, ?, ?, ?, ?)",
(model_id, cost, kwh, carbon, tokens, attr, now.isoformat()),
)
conn.commit()
results = per_model(conn)
by_model = {r["model_id"]: r for r in results}
cheap = by_model["cheap"]
assert cheap["calls"] == 2
assert cheap["sum_cost_usd"] == pytest.approx(0.003)
assert cheap["sum_energy_kwh"] == pytest.approx(8.0e-05)
assert cheap["sum_carbon_g_co2eq"] == pytest.approx(3.6e-03)
assert cheap["avg_completion_tokens"] == pytest.approx(150.0)
assert cheap["avg_attribution_ratio"] == pytest.approx(0.375)
dear = by_model["dear"]
assert dear["calls"] == 1
assert dear["sum_cost_usd"] == pytest.approx(0.010)
def test_per_model_empty_db(tmp_path):
"""No energy rows: returns empty list."""
conn = _make_db(tmp_path)
assert per_model(conn) == []
# --- verdict_mix tests --------------------------------------------------------
def test_verdict_mix_counts_by_verdict(tmp_path):
"""Counts match the inserted rows."""
conn = _make_db(tmp_path)
now = _now()
conn.execute(
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
"VALUES ('m', 'neuralwatt', 'structural', 'ok', ?)",
(now.isoformat(),),
)
conn.execute(
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
"VALUES ('m', 'neuralwatt', 'structural', 'ok', ?)",
(now.isoformat(),),
)
conn.execute(
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
"VALUES ('m', 'neuralwatt', 'local_llm', 'malformed', ?)",
(now.isoformat(),),
)
conn.execute(
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
"VALUES ('m', 'neuralwatt', 'structural', 'unverifiable', ?)",
(now.isoformat(),),
)
# Stale row outside the window
conn.execute(
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
"VALUES ('m', 'neuralwatt', 'structural', 'truncated', ?)",
((now - timedelta(days=30)).isoformat(),),
)
conn.commit()
result = verdict_mix(conn, since_days=7)
assert result["ok"] == 2
assert result["malformed"] == 1
assert result["unverifiable"] == 1
# truncated is > 7 days ago, so excluded
assert "truncated" not in result or result["truncated"] == 0
def test_verdict_mix_empty_db(tmp_path):
"""No rows: returns empty dict."""
conn = _make_db(tmp_path)
assert verdict_mix(conn) == {}
# --- top_proficiency tests ----------------------------------------------------
def test_top_proficiency_ordered_correctly(tmp_path):
"""Models are returned ordered by blended_score DESC."""
conn = _make_db(tmp_path)
_seed_models(conn)
_seed_proficiency(conn)
conn.commit()
results = top_proficiency(conn, "coding_general")
assert len(results) == 3
assert results[0]["model_id"] == "dear" # 0.95
assert results[1]["model_id"] == "cheap" # 0.90
assert results[2]["model_id"] == "tiny" # 0.70
def test_top_proficiency_filter_by_category(tmp_path):
"""Requests for one category exclude models that have scores only for another."""
conn = _make_db(tmp_path)
_seed_models(conn)
# Only code categories
conn.execute(
"""
INSERT INTO proficiency (
model_id, provider, category, blended_score, source, last_updated
) VALUES ('cheap', 'neuralwatt', 'coding_general', 0.90, 'self_eval_thin', '2026-01-01T00:00:00+00:00')
""",
)
conn.execute(
"""
INSERT INTO proficiency (
model_id, provider, category, blended_score, source, last_updated
) VALUES ('cheap', 'neuralwatt', 'docs_writing', 0.85, 'self_eval_thin', '2026-01-01T00:00:00+00:00')
""",
)
conn.commit()
coding = top_proficiency(conn, "coding_general")
docs = top_proficiency(conn, "docs_writing")
assert len(coding) == 1
assert coding[0]["model_id"] == "cheap"
assert len(docs) == 1
assert docs[0]["model_id"] == "cheap"
def test_top_proficiency_empty_for_missing_category(tmp_path):
"""No proficiency rows for category → empty list."""
conn = _make_db(tmp_path)
_seed_models(conn)
assert top_proficiency(conn, "nonexistent_category") == []
def test_top_proficiency_passes_all_sources_through_to_json(tmp_path):
"""Every proficiency source value survives the row to dict to JSON path.
The empirical-Bayes outcome conversion introduced two new sources
(outcome_prior, outcome_blended) alongside the four legacy ones.
top_proficiency does not filter source strings, so this is a guard
against any future JSON schema or typed enum accidentally discarding one.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
sources = [
"self_eval",
"self_eval_thin",
"leaderboard",
"blended",
"outcome_prior",
"outcome_blended",
]
for i, source in enumerate(sources):
model_id = f"m-{source}"
conn.execute(
"""
INSERT INTO models (model_id, provider, base_model_id, availability, last_updated)
VALUES (?, 'neuralwatt', ?, 'active', '2026-01-01T00:00:00+00:00')
""",
(model_id, model_id),
)
conn.execute(
"""
INSERT INTO proficiency (
model_id, provider, category, blended_score, source, last_updated
) VALUES (?, 'neuralwatt', 'coding_general', ?, ?, '2026-01-01T00:00:00+00:00')
""",
(model_id, 0.8 - i * 0.01, source),
)
conn.commit()
results = top_proficiency(conn, "coding_general")
returned = {r["source"] for r in results}
assert returned == set(sources), f"missing sources: {set(sources) - returned}"
for r in results:
assert "model_id" in r
assert "blended_score" in r
assert "source" in r
assert isinstance(r["source"], str)
# --- /health endpoint compatibility -------------------------------------------
def _seed_pinch_models(conn: sqlite3.Connection) -> None:
"""Insert two cheap/dear models plus a cheap-no-cached-price variant."""
fresh = _now().isoformat()
rows = [
# (model_id, tier, context, cost_per_1m_prompt, cost_per_1m_prompt_cached)
("cheap", 2, 262128, 1.0, 0.5),
("dear", 2, 262128, 5.0, None),
("tiny", 1, 131072, 0.1, 0.1),
]
for model_id, tier, context, cost, cached in rows:
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
cost_per_1m_prompt_cached,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, ?, ?, 192500, 16384, ?, ?, ?,
1, 1, 'standard', 'default', 'full', 'public', 'active',
?)
""",
(model_id, model_id, tier, context, cost, cost / 3, cached, fresh),
)
conn.commit()
def _insert_pinch_decision(conn, *, model, provider, orig, final):
"""Insert a route_decisions row with observed_at = now and given pinch values."""
conn.execute(
"""
INSERT INTO route_decisions (
observed_at, kind, task_category, task_tier,
required_context_tokens, confidence, classifier_ms,
classification_source, latency_tolerance, candidates_considered,
selected_model, selected_provider, runner_up_models,
est_cost_usd, est_proficiency, rejected_reason, session_key,
tools, images, json_mode, streamed,
pinch_original_tokens, pinch_final_tokens
) VALUES (?, 'chat', 'coding_general', 2, 100, 0.9, 100,
'classifier', 'interactive', 3, ?, ?, NULL,
0.001, 0.9, NULL, 'sess', 0, 0, 0, 0, ?, ?)
""",
(_now().isoformat(), model, provider, orig, final),
)
conn.commit()
def test_pinch_summary_aggregates_correctly(tmp_path, monkeypatch):
"""Hand-computed aggregates over a seeded 30-day window."""
cfg = load_config(str(ROOT / "config" / "config.yaml"))
monkeypatch.setattr(cfg.pinch, "enabled", True)
cache_rate = cfg.objective.assumed_cache_rate
conn = _make_db(tmp_path)
_seed_pinch_models(conn)
# Two pruned cheap rows: saved tokens 50 and 30.
_insert_pinch_decision(conn, model="cheap", provider="neuralwatt", orig=100, final=50)
_insert_pinch_decision(conn, model="cheap", provider="neuralwatt", orig=80, final=50)
# One non-pruned cheap row (final == orig).
_insert_pinch_decision(conn, model="cheap", provider="neuralwatt", orig=60, final=60)
# One pruned dear row with NULL cost_per_1m_prompt_cached -> falls back to prompt price.
_insert_pinch_decision(conn, model="dear", provider="neuralwatt", orig=200, final=100)
# One rejection row with NULL selected_model -> counted in share/tokens but excluded from dollars.
_insert_pinch_decision(conn, model=None, provider=None, orig=1000, final=500)
result = pinch_summary(conn, cfg)
assert result is not None
assert result["calls_30d"] == 5
assert result["pruned_calls_30d"] == 4
assert result["share_pruned"] == round(4 / 5, 4)
# saved tokens: 50 + 30 + 100 + 500 = 680
assert result["total_tokens_saved"] == 680
# median of [30, 50, 100, 500] = 75
assert result["median_tokens_saved"] == 75
# dollar math: cheap saved 80 tokens at blended rate over cached prompt.
cheap_blended = (1 - cache_rate) * 1.0 + cache_rate * 0.5
cheap_dollars = 80 * cheap_blended / 1_000_000
# dear saved 100 tokens; cached price NULL -> fallback to prompt price 5.0.
dear_dollars = 100 * 5.0 / 1_000_000
# rejection row with NULL selected_model is excluded from dollar math.
assert result["dollars_saved_usd_30d"] == round(cheap_dollars + dear_dollars, 6)
def test_pinch_summary_zero_rows(tmp_path, monkeypatch):
"""Enabled with no decisions in the window returns zeros and None median."""
cfg = load_config(str(ROOT / "config" / "config.yaml"))
monkeypatch.setattr(cfg.pinch, "enabled", True)
conn = _make_db(tmp_path)
result = pinch_summary(conn, cfg)
assert result is not None
assert result["calls_30d"] == 0
assert result["pruned_calls_30d"] == 0
assert result["share_pruned"] == 0.0
assert result["total_tokens_saved"] == 0
assert result["median_tokens_saved"] is None
assert result["dollars_saved_usd_30d"] == 0.0
def test_pinch_summary_disabled_returns_none(tmp_path, monkeypatch):
"""When cfg.pinch.enabled is False, None is returned."""
conn = _make_db(tmp_path)
cfg = SimpleNamespace(pinch=SimpleNamespace(enabled=False))
assert pinch_summary(conn, cfg) is None
# --- /health endpoint compatibility -------------------------------------------
def test_health_endpoint_returns_scoring_key(tmp_path, monkeypatch):
"""/health still returns the same SHAPE after moving functions to metrics."""
db_path = tmp_path / "test.db"
conn = _make_db(tmp_path)
_seed_models(conn)
conn.commit()
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(db_path))
# Disable local verification to avoid Ollama dependency
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
with TestClient(dispatcher.app) as client:
resp = client.get("/health")
assert resp.status_code == 200
data = resp.json()
assert "scoring" in data
assert "routable_models" in data["scoring"]
assert "with_energy_data" in data["scoring"]
assert "with_proficiency_data" in data["scoring"]
assert "quota" in data["scoring"]
assert "warnings" in data["scoring"]
# --- unroutable_models / selection_coverage -----------------------------------
#
# The gap these close: an ineligible candidate produces NO signal. A ceiling is
# a max() and a broken row contributes 0 to it; a rejection warning counts
# decisions with no selected model, and nothing is rejected while the other
# rows serve the traffic. Live on 2026-09-08, 12 of 30 active OpenRouter rows
# had a zero effective context window and had never been selected once across
# 23,000+ decisions.
def _seed_zero_context(conn: sqlite3.Connection, model_id: str = "broken") -> None:
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, 1, 1048576, 0, 16384, 0.1, 0.2,
1, 1, 'standard', 'default', 'full', 'public', 'active', ?)
""",
(model_id, model_id, _now().isoformat()),
)
conn.commit()
def test_a_zero_context_row_is_reported_unroutable(tmp_path):
conn = _make_db(tmp_path)
_seed_models(conn)
_seed_zero_context(conn)
found = unroutable_models(conn, CFG)
assert [d["model_id"] for d in found] == ["broken"]
assert found[0]["reason"].startswith("context(")
def test_the_probe_uses_one_token_not_zero(tmp_path):
"""context_ceilings probes with zero, which is exactly why it could not
see this: a zero-token window passes `0 < 0` and is counted a candidate,
then contributes 0 to a max()."""
conn = _make_db(tmp_path)
_seed_models(conn)
_seed_zero_context(conn)
# The broken row IS in the candidate set the ceiling is taken over...
ceilings = context_ceilings(conn, CFG)
assert any(b["count"] for b in ceilings.values())
# ...and yet contributes nothing, so no ceiling moves and nothing warns.
assert max(b["ceiling"] for b in ceilings.values()) == 192500
assert unroutable_models(conn, CFG)
def test_a_healthy_catalog_reports_nothing(tmp_path):
"""The check must be quiet on a normal catalog or it is worthless."""
conn = _make_db(tmp_path)
_seed_models(conn)
assert unroutable_models(conn, CFG) == []
assert selection_coverage_warnings(selection_coverage(conn, CFG)) == []
def test_an_admin_deprecation_is_not_reported_as_a_defect(tmp_path):
"""Switching a model off is a choice. Reporting it would make this warn
about the exact thing it exists to distinguish from."""
conn = _make_db(tmp_path)
_seed_models(conn)
_seed_zero_context(conn)
conn.executescript(_ADMIN_TABLE_SQL)
conn.execute(
"INSERT INTO admin_model_overrides "
"(model_id, provider, availability, updated_at) "
"VALUES ('broken', 'neuralwatt', 'deprecated', ?)",
(_now().isoformat(),),
)
conn.commit()
assert unroutable_models(conn, CFG) == []
def test_a_restricted_category_row_is_not_called_unroutable(tmp_path):
"""eligible_categories is a restrict-only gate -- a narrowing, not a
defect. The probe asks with a category the row itself admits."""
conn = _make_db(tmp_path)
_seed_models(conn)
conn.execute(
"UPDATE models SET eligible_categories = 'file_summarization,"
"diff_checking' WHERE model_id = 'tiny'"
)
conn.commit()
assert unroutable_models(conn, CFG) == []
def test_one_warning_per_gate_not_per_model(tmp_path):
"""The live incident produced 11 rows failing the identical gate. Eleven
near-identical warnings is a wall nobody reads, and they share one root
cause and one fix."""
conn = _make_db(tmp_path)
_seed_models(conn)
for i in range(11):
_seed_zero_context(conn, f"broken-{i}")
warnings = selection_coverage_warnings(selection_coverage(conn, CFG))
assert len(warnings) == 1
assert "11 active models can never be selected" in warnings[0]
# Named enough to recognize, summarized past that.
assert "+7 more" in warnings[0]
def test_never_selected_is_data_and_never_a_warning(tmp_path):
"""22 of 38 routable models were unselected on the live DB. Enumerating
those as a warning would fire permanently and mean 'you have more models
than winners', which is normal."""
conn = _make_db(tmp_path)
_seed_models(conn)
coverage = selection_coverage(conn, CFG)
assert len(coverage["never_selected"]) == 3
assert coverage["selected_models"] == 0
assert selection_coverage_warnings(coverage) == []
def test_selection_coverage_counts_only_inside_the_window(tmp_path):
conn = _make_db(tmp_path)
_seed_models(conn)
for model_id, age_hours in (("cheap", 1), ("dear", 24 * 30)):
conn.execute(
"INSERT INTO route_decisions (observed_at, kind, selected_model,"
" selected_provider) VALUES (?, 'chat', ?, 'neuralwatt')",
((_now() - timedelta(hours=age_hours)).isoformat(), model_id),
)
conn.commit()
coverage = selection_coverage(conn, CFG)
# 'dear' was picked, but a month ago -- outside the 168h window.
assert coverage["selected_models"] == 1
assert coverage["by_provider"]["neuralwatt"] == {"routable": 3, "selected": 1}
assert {r["model_id"] for r in coverage["never_selected"]} == {"dear", "tiny"}
def test_an_access_gated_row_is_not_counted_routable(tmp_path):
"""A canary or preview row can never be selected under this config BY
DESIGN, so counting it routable-but-unselected pads never_selected and
makes the two routable_models figures in one /metrics payload disagree.
Live after the detector shipped: selection said 51 while scoring_coverage
said 46 -- exactly the catalog's 1 canary + 4 preview rows.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
conn.execute(
"UPDATE models SET access_level = 'canary' WHERE model_id = 'tiny'"
)
conn.commit()
coverage = selection_coverage(conn, CFG)
assert coverage["routable_models"] == 2
assert "tiny" not in {r["model_id"] for r in coverage["never_selected"]}
# The two figures in one payload must agree.
assert (
coverage["routable_models"]
== scoring_coverage(conn, CFG)["routable_models"]
)
# =============================================================================
# cache_rate_series / cache_rate_warnings (Wave 1 item 1.4)
# =============================================================================
def _seed_cache_rows(
conn: sqlite3.Connection,
*,
model_id: str = "cheap",
provider: str = "neuralwatt",
n: int,
prompt_tokens: int,
cached: object,
source: str = "reported",
task_category: str = "coding_general",
age_hours: float = 1.0,
) -> None:
"""Insert *n* energy_observations rows shaped for the cache-rate query."""
at = (_now() - timedelta(hours=age_hours)).isoformat()
for _ in range(n):
conn.execute(
"""
INSERT INTO energy_observations (
model_id, provider, task_category, prompt_tokens,
completion_tokens, energy_kwh, cached_prompt_tokens,
cached_tokens_source, observed_at
) VALUES (?, ?, ?, ?, 100, 0.0, ?, ?, ?)
""",
(model_id, provider, task_category, prompt_tokens, cached, source, at),
)
conn.commit()
def _cfg_with(**objective_overrides):
cfg = CFG.model_copy(deep=True)
for key, value in objective_overrides.items():
setattr(cfg.objective, key, value)
return cfg
def test_cache_rate_series_is_token_weighted_not_a_mean_of_rates(tmp_path):
"""sum(cached)/sum(prompt), not the average of per-request rates.
The bill is denominated in tokens, so one 100k-token turn is not one
observation's worth of evidence against a 1k one. A mean of per-request
rates here reads 0.667; the token-weighted answer is 0.505.
"""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=1, prompt_tokens=1000, cached=500)
_seed_cache_rows(conn, n=1, prompt_tokens=1000, cached=1000)
# One huge mostly-uncached turn, which a mean of rates would let vanish.
_seed_cache_rows(conn, n=1, prompt_tokens=100000, cached=50000)
series = cache_rate_series(conn, CFG)
assert series["prompt_tokens"] == 102000
assert series["cached_prompt_tokens"] == 51500
assert series["cache_rate"] == pytest.approx(51500 / 102000)
assert series["observations"] == 3
def test_cache_rate_series_counts_only_reported_rows(tmp_path):
"""A row saying nothing must not enter the denominator.
A 'details_no_count' or 'no_details' row carries a real prompt_tokens
figure, so including it would divide a genuine cached total by a
denominator full of prompts nobody measured -- understating the rate
exactly where coverage is thinnest.
"""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=2, prompt_tokens=1000, cached=900)
_seed_cache_rows(
conn, n=5, prompt_tokens=1000, cached=None, source="details_no_count"
)
_seed_cache_rows(
conn, n=5, prompt_tokens=1000, cached=None, source="no_details"
)
series = cache_rate_series(conn, CFG)
assert series["observations"] == 2
assert series["prompt_tokens"] == 2000
assert series["cache_rate"] == pytest.approx(0.9)
def test_cache_rate_series_counts_an_explicit_zero(tmp_path):
"""A reported 0 IS a measured full miss and must drag the rate down.
This is what af18009 exists for: "reported, and the number was zero" is
evidence, while NULL is an absence. Dropping the zeros would leave every
rate conditioned on a hit having occurred.
"""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=1, prompt_tokens=1000, cached=1000)
_seed_cache_rows(conn, n=1, prompt_tokens=1000, cached=0)
series = cache_rate_series(conn, CFG)
assert series["observations"] == 2
assert series["cache_rate"] == pytest.approx(0.5)
def test_cache_rate_series_excludes_seed_reference_sweeps(tmp_path):
"""seed_energy.py's sweep is not user traffic.
Those rows were the entire reason NeuralWatt's cached-token coverage read
12.8% while real dispatch traffic reads ~98%.
"""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=2, prompt_tokens=1000, cached=900)
_seed_cache_rows(
conn, n=40, prompt_tokens=400, cached=0, task_category="seed_reference"
)
series = cache_rate_series(conn, CFG)
assert series["observations"] == 2
assert series["cache_rate"] == pytest.approx(0.9)
def test_cache_rate_series_is_bounded_by_the_trailing_window(tmp_path):
"""Rows older than the window are out, so the pre-capture era ages away by
itself rather than needing a hardcoded start date."""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=2, prompt_tokens=1000, cached=900, age_hours=1)
_seed_cache_rows(conn, n=9, prompt_tokens=1000, cached=0, age_hours=200)
series = cache_rate_series(conn, CFG)
assert series["observations"] == 2
assert series["cache_rate"] == pytest.approx(0.9)
assert series["window_hours"] == 168
def test_cache_rate_series_groups_on_provider_and_model(tmp_path):
"""Two providers serving the same base model are two groups, not one."""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, model_id="dv4", provider="neuralwatt", n=3,
prompt_tokens=1000, cached=900)
_seed_cache_rows(conn, model_id="dv4", provider="openrouter", n=3,
prompt_tokens=1000, cached=400)
series = cache_rate_series(conn, CFG)
by_key = {(r["provider"], r["model_id"]): r for r in series["by_model"]}
assert set(by_key) == {("neuralwatt", "dv4"), ("openrouter", "dv4")}
assert by_key[("neuralwatt", "dv4")]["cache_rate"] == pytest.approx(0.9)
assert by_key[("openrouter", "dv4")]["cache_rate"] == pytest.approx(0.4)
def test_cache_rate_series_survives_a_pre_migration_schema(tmp_path):
"""cached_tokens_source arrives by ALTER at dispatcher start-up, and
admin.py can open a database that has not been through it. A missing
column must degrade this series, not 500 the whole /metrics payload."""
conn = _make_db(tmp_path)
conn.execute("DROP TABLE energy_observations")
conn.execute(
"CREATE TABLE energy_observations ("
" id INTEGER PRIMARY KEY, model_id TEXT, provider TEXT, observed_at TEXT)"
)
conn.commit()
series = cache_rate_series(conn, CFG)
assert series["cache_rate"] is None
assert series["by_model"] == []
assert cache_rate_warnings(conn, CFG, series) == []
def test_cache_rate_warning_silent_below_the_observation_floor(tmp_path):
"""Never mere presence: a divergence over a handful of requests is one
session's luck, and a warning that flaps on low traffic goes unread."""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=5, prompt_tokens=1000, cached=100)
assert cache_rate_warnings(conn, CFG) == []
def test_cache_rate_warning_silent_inside_the_margin(tmp_path):
"""0.900 against an assumed 0.917 is ordinary session-mix drift."""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=900)
assert cache_rate_warnings(conn, CFG) == []
def test_cache_rate_warning_fires_when_the_premise_expires(tmp_path):
"""The aggregate class names the constant it is expiring, not a floor."""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=400)
warnings = cache_rate_warnings(conn, CFG)
premise = [w for w in warnings if w.startswith("cache rate: measured")]
assert len(premise) == 1
assert "0.400" in premise[0]
assert "assumed_cache_rate" in premise[0]
assert "below" in premise[0]
def test_cache_rate_warning_fires_in_the_other_direction_too(tmp_path):
"""A rate ABOVE the assumption mis-prices just as surely -- it overstates
the cache and underprices every candidate. Shown against a lower assumed
rate because at the shipped 0.917 a 0.10 margin cannot be exceeded
upward: the measured rate is bounded by 1.0."""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=900)
warnings = cache_rate_warnings(conn, _cfg_with(assumed_cache_rate=0.5))
premise = [w for w in warnings if w.startswith("cache rate: measured")]
assert len(premise) == 1
assert "above" in premise[0]
def test_cache_rate_outlier_fires_while_the_aggregate_stays_quiet(tmp_path):
"""The per-group half is the one that catches a single bad route.
The aggregate sits inside the margin; the openrouter group at 0.400 does
not. Only the outlier class fires, and it names the structured
(provider, model) pair rather than a formatted label.
"""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, model_id="hot", provider="neuralwatt", n=300,
prompt_tokens=1000, cached=920)
_seed_cache_rows(conn, model_id="cold", provider="openrouter", n=30,
prompt_tokens=1000, cached=400)
series = cache_rate_series(conn, CFG)
warnings = cache_rate_warnings(conn, CFG, series)
assert series["cache_rate"] == pytest.approx(
(300 * 920 + 30 * 400) / (330 * 1000)
)
assert not [w for w in warnings if w.startswith("cache rate: measured")]
outliers = [w for w in warnings if w.startswith("cache rate outlier:")]
assert len(outliers) == 1
assert "cold" in outliers[0] and "openrouter" in outliers[0]
def test_cache_rate_outlier_respects_the_per_group_observation_floor(tmp_path):
"""A four-request group that looks catastrophic is four requests."""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, model_id="hot", provider="neuralwatt", n=300,
prompt_tokens=1000, cached=920)
_seed_cache_rows(conn, model_id="cold", provider="openrouter", n=4,
prompt_tokens=1000, cached=100)
warnings = cache_rate_warnings(conn, CFG)
assert not [w for w in warnings if w.startswith("cache rate outlier:")]
def test_cache_rate_warning_silent_without_an_assumption_to_expire(tmp_path):
"""Nothing to diverge from means nothing to say."""
conn = _make_db(tmp_path)
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=100)
assert cache_rate_warnings(conn, _cfg_with(assumed_cache_rate=None)) == []
def test_scoring_coverage_carries_the_cache_series_and_its_warnings(tmp_path):
"""The series and the warning are one reading, not two queries that can
disagree about the number on screen."""
conn = _make_db(tmp_path)
_seed_models(conn)
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=400)
coverage = scoring_coverage(conn, CFG)
assert coverage["cache"]["cache_rate"] == pytest.approx(0.4)
assert any(w.startswith("cache rate: measured") for w in coverage["warnings"])
# =============================================================================
# proficiency_sample_depth_series / _warnings (Wave 5.3)
# =============================================================================
def _seed_proficiency_depth_rows(
conn: sqlite3.Connection,
*,
category: str = "coding_general",
n: int,
outcome_samples: int = 0,
self_eval_samples: int = 0,
model_id: str = "cheap",
seed_models: bool = True,
) -> None:
"""Insert *n* proficiency rows with the given sample counts."""
for i in range(n):
suffix = str(i) if i else ""
mid = model_id + suffix
if seed_models:
conn.execute(
"""
INSERT OR IGNORE INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, 2, 262128, 192500, 16384, 0.30, 0.10,
1, 1, 'standard', 'default', 'full', 'public', 'active',
'2026-09-01T00:00:00+00:00')
""",
(mid, mid),
)
conn.execute(
"""
INSERT INTO proficiency (
model_id, provider, category, blended_score, source,
self_eval_samples, outcome_samples, last_updated
) VALUES (?, 'neuralwatt', ?, 0.9, 'self_eval_thin',
?, ?, '2026-09-01T00:00:00+00:00')
""",
(mid, category, self_eval_samples, outcome_samples),
)
conn.commit()
def test_proficiency_depth_series_computes_average_depth(tmp_path):
"""The series aggregates sample depth per category and overall."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
self_eval_samples=10, outcome_samples=5)
_seed_proficiency_depth_rows(conn, n=2, category="debugging",
self_eval_samples=20, outcome_samples=10)
series = proficiency_sample_depth_series(conn, CFG)
# coding_general: 3 rows, avg depth (15+15+15)/3 = 15
# debugging: 2 rows, avg depth (30+30)/2 = 30
# overall: (45+60)/5 = 21.0
assert series["total_rows"] == 5
assert series["overall_avg_depth"] == pytest.approx(21.0)
by_cat = {e["category"]: e for e in series["by_category"]}
assert by_cat["coding_general"]["avg_depth"] == pytest.approx(15.0)
assert by_cat["coding_general"]["rows"] == 3
assert by_cat["debugging"]["avg_depth"] == pytest.approx(30.0)
assert by_cat["debugging"]["rows"] == 2
def test_proficiency_depth_series_identifies_thin_rows(tmp_path):
"""Rows below min_samples are counted as thin."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=2, category="coding_general",
self_eval_samples=1, outcome_samples=1)
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
self_eval_samples=20, outcome_samples=10,
model_id="dear")
series = proficiency_sample_depth_series(conn, CFG)
assert series["thin_rows"] == 2
assert series["total_rows"] == 5
def test_proficiency_depth_warning_silent_below_the_observation_floor(tmp_path):
"""Never warns on a near-empty table (novelty-or-rate)."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=3, category="coding_general",
self_eval_samples=5, outcome_samples=5)
warnings = proficiency_sample_depth_warnings(conn, CFG)
assert warnings == []
def test_proficiency_depth_warning_silent_below_the_threshold(tmp_path):
"""Average sample depth below min_samples does not fire."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=15, category="coding_general",
self_eval_samples=5, outcome_samples=5)
cfg = _cfg_with(proficiency_depth_warn_min_samples=50)
warnings = proficiency_sample_depth_warnings(conn, cfg)
assert warnings == []
def test_proficiency_depth_warning_fires_when_depth_exceeds_threshold(tmp_path):
"""Once average depth passes the configurable threshold, the premise is
expired and the warning fires."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=20, category="coding_general",
self_eval_samples=15, outcome_samples=10)
warnings = proficiency_sample_depth_warnings(conn, CFG)
premise = [w for w in warnings
if w.startswith("quality_tolerance premise expired")]
assert len(premise) >= 1
assert "quality_tolerance" in premise[0]
def test_proficiency_depth_warning_respects_the_observation_floor(tmp_path):
"""Plenty of rows but all below the floor threshold — silent."""
conn = _make_db(tmp_path)
_seed_proficiency_depth_rows(conn, n=5, category="coding_general",
self_eval_samples=30, outcome_samples=30)
cfg = _cfg_with(proficiency_depth_warn_min_rows=10)
warnings = proficiency_sample_depth_warnings(conn, cfg)
assert warnings == []
# =============================================================================
# cumulative_spend_series / _warnings (Wave 5.3)
# =============================================================================
def _seed_spend_rows(
conn: sqlite3.Connection,
*,
n: int,
cost_usd: float = 0.0,
model_id: str = "cheap",
provider: str = "neuralwatt",
task_category: str = "coding_general",
) -> None:
"""Insert *n* energy_observations rows with a cost_usd."""
now = _now()
for _ in range(n):
conn.execute(
"""
INSERT INTO energy_observations (
model_id, provider, task_category, prompt_tokens,
completion_tokens, energy_kwh, cost_usd, observed_at
) VALUES (?, ?, ?, 1000, 100, 0.001, ?, ?)
""",
(model_id, provider, task_category, cost_usd, now.isoformat()),
)
conn.commit()
def _seed_spend_seed_rows(conn: sqlite3.Connection, *, n: int = 5) -> None:
"""Insert seed_reference energy rows (excluded from spend)."""
now = _now()
for _ in range(n):
conn.execute(
"""
INSERT INTO energy_observations (
model_id, provider, task_category, prompt_tokens,
completion_tokens, energy_kwh, cost_usd, observed_at
) VALUES ('cheap', 'neuralwatt', 'seed_reference', 1000, 100,
0.001, 999.0, ?)
""",
(now.isoformat(),),
)
conn.commit()
def test_cumulative_spend_series_aggregates_all_time_cost(tmp_path):
"""The series sums cost_usd across all non-seed energy rows."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=3, cost_usd=0.05)
_seed_spend_rows(conn, n=2, cost_usd=0.10)
series = cumulative_spend_series(conn, CFG)
assert series["total_spend_usd"] == pytest.approx(0.35)
assert series["priced_rows"] == 5
assert series["total_rows"] == 5
def test_cumulative_spend_series_excludes_seed_reference(tmp_path):
"""seed_reference rows do not inflate 'real traffic' spend."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=2, cost_usd=0.05)
_seed_spend_seed_rows(conn, n=5)
series = cumulative_spend_series(conn, CFG)
assert series["total_spend_usd"] == pytest.approx(0.10)
assert series["priced_rows"] == 2
# total_rows is also seed-filtered: the series describes real traffic only,
# so the excluded rows never appear in any of its counts.
assert series["total_rows"] == 2
def test_cumulative_spend_series_per_provider_breakdown(tmp_path):
"""Results include per-provider sub-totals."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=2, cost_usd=1.0, provider="neuralwatt")
_seed_spend_rows(conn, n=3, cost_usd=2.0, provider="openrouter")
series = cumulative_spend_series(conn, CFG)
assert series["total_spend_usd"] == pytest.approx(8.0)
by_provider = {e["provider"]: e for e in series["by_provider"]}
assert by_provider["neuralwatt"]["provider_spend_usd"] == pytest.approx(2.0)
assert by_provider["openrouter"]["provider_spend_usd"] == pytest.approx(6.0)
def test_cumulative_spend_warning_silent_below_the_observation_floor(tmp_path):
"""Never warns on too few priced rows (novelty-or-rate)."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=3, cost_usd=100.0)
cfg = _cfg_with(cumulative_spend_warn_min_rows=10)
warnings = cumulative_spend_warnings(conn, cfg)
assert warnings == []
def test_cumulative_spend_warning_silent_below_the_threshold(tmp_path):
"""Spend below the warning threshold does not fire."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=15, cost_usd=0.01)
cfg = _cfg_with(cumulative_spend_warn_usd=50.0)
warnings = cumulative_spend_warnings(conn, cfg)
assert warnings == []
def test_cumulative_spend_warning_fires_when_premise_expires(tmp_path):
"""Once all-time spend exceeds warn_usd, the warning fires naming the
blend comment whose premise has expired."""
conn = _make_db(tmp_path)
_seed_spend_rows(conn, n=15, cost_usd=50.0)
cfg = _cfg_with(cumulative_spend_warn_usd=10.0,
cumulative_spend_warn_min_rows=5)
warnings = cumulative_spend_warnings(conn, cfg)
premise = [w for w in warnings
if w.startswith("cost-as-tiebreak premise expired")]
assert len(premise) >= 1
assert "cumulative spend" in premise[0]
assert "$" in premise[0]
def test_scoring_coverage_carries_the_premise_expiry_series(tmp_path):
"""scoring_coverage includes both new premise-expiry series and warnings."""
conn = _make_db(tmp_path)
_seed_models(conn)
# Seed enough proficiency depth to trigger the warning
_seed_proficiency_depth_rows(conn, n=20, category="coding_general",
self_eval_samples=15, outcome_samples=10)
_seed_spend_rows(conn, n=15, cost_usd=50.0)
cfg = _cfg_with(
cumulative_spend_warn_usd=10.0,
cumulative_spend_warn_min_rows=5,
)
coverage = scoring_coverage(conn, cfg)
assert "proficiency_sample_depth" in coverage
assert coverage["proficiency_sample_depth"]["overall_avg_depth"] is not None
assert "cumulative_spend" in coverage
assert coverage["cumulative_spend"]["total_spend_usd"] == pytest.approx(750.0)
assert any(
w.startswith("quality_tolerance premise expired")
for w in coverage["warnings"]
)
assert any(
w.startswith("cost-as-tiebreak premise expired")
for w in coverage["warnings"]
)
# --- conversation adoption tests ----------------------------------------------
def _seed_conversation_adoption_rows(
conn: sqlite3.Connection,
*,
c_rows: tuple = (),
fingerprint_in_window: tuple = (),
) -> None:
"""Seed route_decisions with c:-prefixed and fingerprint session_keys.
Each c-row is a (session_key, in_window) pair; a fingerprint row needs
only its in-window flag.
"""
now = _now()
for key, in_window in c_rows:
conn.execute(
"INSERT INTO route_decisions "
"(observed_at, kind, session_key, agent, parent_key) "
"VALUES (?, 'chat', ?, 'classifier', NULL)",
(
(now - timedelta(hours=1)).isoformat() if in_window
else (now - timedelta(days=30)).isoformat(),
key,
),
)
for i, in_window in enumerate(fingerprint_in_window):
conn.execute(
"INSERT INTO route_decisions "
"(observed_at, kind, session_key, agent, parent_key) "
"VALUES (?, 'chat', ?, 'classifier', NULL)",
(
(now - timedelta(hours=1)).isoformat() if in_window
else (now - timedelta(days=30)).isoformat(),
f"fp-{i}",
),
)
conn.commit()
def test_conversation_adoption_counts_conversations_and_decisions(tmp_path):
"""3 c: rows over 2 keys + 1 fingerprint row -> 2 conversations,
3 c: decisions, 4 decisions, share 0.75."""
conn = _make_db(tmp_path)
_seed_conversation_adoption_rows(
conn,
c_rows=(("c:conv-a", True), ("c:conv-a", True), ("c:conv-b", True)),
fingerprint_in_window=(True,),
)
result = conversation_adoption(conn, CFG)
conn.close()
assert result["n_c_conversations"] == 2
assert result["n_c_decisions"] == 3
assert result["n_decisions"] == 4
assert result["share"] == pytest.approx(0.75)
assert result["window_seconds"] == 604800 # from config.yaml
def test_conversation_adoption_empty_table(tmp_path):
"""An empty route_decisions -> 0, 0, 0 and share None."""
conn = _make_db(tmp_path)
result = conversation_adoption(conn, CFG)
conn.close()
assert result["n_c_conversations"] == 0
assert result["n_c_decisions"] == 0
assert result["n_decisions"] == 0
assert result["share"] is None
def test_conversation_adoption_window_excludes_old_rows(tmp_path):
"""The window applies to every count: a 30 d old c: row and fingerprint
row stay out of n_c_* AND the n_decisions denominator."""
conn = _make_db(tmp_path)
_seed_conversation_adoption_rows(
conn,
c_rows=(
("c:conv-a", True),
("c:conv-b", True),
("c:conv-old", False),
),
fingerprint_in_window=(False,),
)
result = conversation_adoption(conn, CFG)
conn.close()
assert result["n_c_conversations"] == 2
assert result["n_c_decisions"] == 2
assert result["n_decisions"] == 2
def test_conversation_adoption_db_without_session_key_column(tmp_path):
"""DB whose route_decisions predates the identity migration -> the gate
dict: zeros, share None, no error."""
conn = sqlite3.connect(str(tmp_path / "no_session_key.db"))
conn.row_factory = sqlite3.Row
conn.execute(
"CREATE TABLE route_decisions "
"(id INTEGER PRIMARY KEY, observed_at TEXT, kind TEXT)"
)
conn.commit()
result = conversation_adoption(conn, CFG)
conn.close()
assert result["n_c_conversations"] == 0
assert result["n_c_decisions"] == 0
assert result["n_decisions"] == 0
assert result["share"] is None
assert result["window_seconds"] == 604800 # from config.yaml
# --- [H3] migration-path tests ------------------------------------------------
def test_ensure_tables_adds_route_decisions_columns_to_old_db(tmp_path, monkeypatch):
"""A DB created from the OLD schema gains both agent and parent_key columns."""
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
monkeypatch.setenv("OPENROUTER_API_KEY", "test-key")
old_route_decisions = """
CREATE TABLE IF NOT EXISTS route_decisions (
id INTEGER PRIMARY KEY AUTOINCREMENT,
observed_at TEXT NOT NULL,
kind TEXT NOT NULL,
task_category TEXT,
task_tier INTEGER,
required_context_tokens INTEGER,
confidence REAL,
classifier_ms INTEGER,
classification_source TEXT,
latency_tolerance TEXT,
candidates_considered INTEGER,
selected_model TEXT,
selected_provider TEXT,
runner_up_models TEXT,
est_cost_usd REAL,
est_proficiency REAL,
rejected_reason TEXT,
session_key TEXT,
tools INTEGER,
images INTEGER,
json_mode INTEGER,
streamed INTEGER,
flex_preference TEXT,
flex_swapped INTEGER,
flex_forced INTEGER,
request_id TEXT,
exploration INTEGER DEFAULT 0,
pinch_original_tokens INTEGER,
pinch_final_tokens INTEGER,
profile TEXT,
prefix_divergence_index INTEGER,
prefix_tokens_after_divergence INTEGER,
prefix_prev_message_count INTEGER
)
"""
db_path = tmp_path / "old-schema.db"
old_conn = sqlite3.connect(str(db_path))
old_conn.executescript(old_route_decisions)
old_conn.execute(
"CREATE TABLE local_energy_observations ("
"id INTEGER PRIMARY KEY, model_id TEXT, call_type TEXT, request_id TEXT,"
"session_dir TEXT, avg_power_watts REAL, duration_seconds REAL,"
"energy_kwh REAL, cost_usd REAL, carbon_g_co2eq REAL, meter TEXT,"
"observed_at TEXT)"
)
old_conn.execute(
"CREATE TABLE energy_observations (id INTEGER PRIMARY KEY, model_id TEXT, provider TEXT)"
)
old_conn.close()
# Point dispatcher at this DB and run _ensure_tables
import dispatcher as d
original_path = d.cfg.database.path
try:
d.cfg.database.path = str(db_path)
d._ensure_tables()
finally:
d.cfg.database.path = original_path
# Verify both columns exist
check = sqlite3.connect(str(db_path))
check.row_factory = sqlite3.Row
cols = {row[1] for row in check.execute("PRAGMA table_info(route_decisions)")}
assert "agent" in cols, "agent column should exist after migration"
assert "parent_key" in cols, "parent_key column should exist after migration"
# Verify local_energy_observations has session_key
local_cols = {row[1] for row in check.execute("PRAGMA table_info(local_energy_observations)")}
assert "session_key" in local_cols, "session_key column should exist after migration"
# Also test that recent_decisions runs cleanly on the migrated DB
rows = recent_decisions(check)
check.close()
assert isinstance(rows, list)
def test_log_local_energy_persists_session_key(tmp_path):
"""log_local_energy stores the session_key value given."""
conn = sqlite3.connect(str(tmp_path / "le-test.db"))
conn.row_factory = sqlite3.Row
import local_energy
local_energy.ensure_local_energy_table(conn)
local_energy.log_local_energy(
conn,
model_id="test-model",
call_type="classify",
avg_power_watts=50.0,
duration_seconds=10.0,
energy_kwh=0.00014,
cost_usd=0.0,
carbon_g_co2eq=0.0,
meter="test",
observed_at=_now().isoformat(),
session_key="c:conv-42",
)
row = conn.execute(
"SELECT session_key FROM local_energy_observations LIMIT 1"
).fetchone()
conn.close()
assert row is not None
assert row["session_key"] == "c:conv-42"
# --- decision_outcome_summary (admin Decisions tile) ---------------------------
def _insert_client_outcome(
conn: sqlite3.Connection,
verdict: str,
observed_at: str,
) -> None:
conn.execute(
"INSERT INTO verifications (model_id, provider, kind, verdict, "
"observed_at) VALUES ('cheap', 'neuralwatt', 'client_outcome', ?, ?)",
(verdict, observed_at),
)
def _insert_routed_decision(conn: sqlite3.Connection, observed_at: str) -> None:
conn.execute(
"""
INSERT INTO route_decisions (
observed_at, kind, task_category, task_tier, required_context_tokens,
confidence, classifier_ms, classification_source, latency_tolerance,
candidates_considered, selected_model, selected_provider,
runner_up_models, est_cost_usd, est_proficiency,
session_key, tools, images, json_mode, streamed
) VALUES (?, 'route', 'coding_general', 2, 100, 0.95, 200,
'classifier', 'interactive', 5, 'cheap', 'neuralwatt',
NULL, 0.001, 0.9, 'abc123', 0, 0, 0, 0)
""",
(observed_at,),
)
def test_decision_outcome_summary_counts_only_client_outcome_kinds(tmp_path):
"""client_* counts verifications kind='client_outcome' ONLY.
structural and local_llm verdicts are diagnostics, not ground truth:
a structural 'ok' must never raise client_ok, and neither diagnostic
kind may raise client_reports.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
now = _now().isoformat()
for kind, verdict in (
("structural", "ok"),
("structural", "failed"),
("local_llm", "ok"),
("local_llm", "malformed"),
("client_outcome", "succeeded"),
("client_outcome", "succeeded"),
("client_outcome", "failed"),
):
conn.execute(
"INSERT INTO verifications (model_id, provider, kind, verdict, "
"observed_at) VALUES ('cheap', 'neuralwatt', ?, ?, ?)",
(kind, verdict, now),
)
conn.commit()
summary = decision_outcome_summary(conn)
assert summary["client_reports"] == 3
assert summary["client_ok"] == 2
assert summary["client_failed"] == 1
def test_decision_outcome_summary_decisions_come_from_route_decisions(tmp_path):
"""decisions counts route_decisions rows, independent of verifications."""
conn = _make_db(tmp_path)
_seed_models(conn)
now = _now().isoformat()
for _ in range(4):
_insert_routed_decision(conn, now)
_insert_client_outcome(conn, "succeeded", now)
conn.commit()
summary = decision_outcome_summary(conn)
assert summary["decisions"] == 4
assert summary["client_reports"] == 1
assert summary["client_ok"] == 1
assert summary["client_failed"] == 0
def test_decision_outcome_summary_excludes_out_of_window_rows(tmp_path):
"""A 10-day-old decision and a 10-day-old client report both miss the
default 7-day window."""
conn = _make_db(tmp_path)
_seed_models(conn)
now = _now().isoformat()
old = (datetime.now(timezone.utc) - timedelta(days=10)).isoformat()
_insert_routed_decision(conn, old)
_insert_routed_decision(conn, now)
_insert_client_outcome(conn, "failed", old)
_insert_client_outcome(conn, "succeeded", now)
conn.commit()
summary = decision_outcome_summary(conn)
assert summary["decisions"] == 1
assert summary["client_reports"] == 1
assert summary["client_ok"] == 1
assert summary["client_failed"] == 0
def test_decision_outcome_summary_empty_db_is_all_zeros(tmp_path):
"""An empty DB yields a dict of zeros -- not None, not missing keys."""
conn = _make_db(tmp_path)
assert decision_outcome_summary(conn) == {
"decisions": 0,
"client_reports": 0,
"client_ok": 0,
"client_failed": 0,
}