2263 lines
84 KiB
Python
2263 lines
84 KiB
Python
"""Tests for metrics.py — read-only aggregation helpers.
|
|
|
|
Every test seeds a throwaway SQLite DB directly from schema.sql, never writes
|
|
to the live ``router.db``, and asserts on *actual queried aggregates* rather
|
|
than mock-call assertions (to defeat ``misleading_success_output``).
|
|
|
|
Functions tested:
|
|
- quota_accounts
|
|
- scoring_coverage
|
|
- recent_decisions
|
|
- per_model
|
|
- verdict_mix
|
|
- top_proficiency
|
|
|
|
Also verifies that ``import dispatcher`` and ``/health`` still work after the
|
|
move, and that ``import metrics`` alone succeeds (no circular import).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import sqlite3
|
|
from datetime import date, datetime, timedelta, timezone
|
|
from pathlib import Path
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
from starlette.testclient import TestClient
|
|
|
|
import dispatcher
|
|
from config import load_config
|
|
from metrics import (
|
|
_next_reset_date,
|
|
cache_rate_series,
|
|
cache_rate_warnings,
|
|
capability_ceilings,
|
|
capability_demand_warnings,
|
|
context_ceilings,
|
|
demand_ceiling_warnings,
|
|
local_energy_summary,
|
|
per_model,
|
|
pinch_summary,
|
|
quota_accounts,
|
|
recent_decisions,
|
|
rejection_warnings,
|
|
scoring_coverage,
|
|
selection_coverage,
|
|
selection_coverage_warnings,
|
|
top_proficiency,
|
|
unroutable_models,
|
|
verdict_mix,
|
|
)
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
|
CFG = load_config(str(ROOT / "config" / "config.yaml"))
|
|
|
|
|
|
# =============================================================================
|
|
# Helpers
|
|
# =============================================================================
|
|
|
|
|
|
def _now() -> datetime:
|
|
return datetime.now(timezone.utc)
|
|
|
|
|
|
def _make_db(tmp_path: Path, extra_sql: str = "") -> sqlite3.Connection:
|
|
"""Create a clean DB seeded from schema.sql, returning a Row-backed conn."""
|
|
conn = sqlite3.connect(str(tmp_path / "test.db"))
|
|
conn.row_factory = sqlite3.Row
|
|
conn.executescript(SCHEMA_SQL + extra_sql)
|
|
return conn
|
|
|
|
|
|
def _seed_models(conn: sqlite3.Connection) -> None:
|
|
"""Insert routable model rows into an (empty) DB."""
|
|
fresh = _now().isoformat()
|
|
for model_id, tier, context, cost, vision in (
|
|
("cheap", 2, 262128, 0.30, 1),
|
|
("dear", 2, 262128, 9.00, 0),
|
|
("tiny", 1, 131072, 0.10, 1),
|
|
):
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES (?, 'neuralwatt', ?, ?, ?, 192500, 16384, ?, ?,
|
|
?, 1, 'standard', 'default', 'full', 'public', 'active',
|
|
?)
|
|
""",
|
|
(model_id, model_id, tier, context, cost, cost / 3, vision, fresh),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _seed_proficiency(conn: sqlite3.Connection) -> None:
|
|
"""Insert proficiency rows for the seeded models."""
|
|
for model_id, score in (
|
|
("cheap", 0.90),
|
|
("dear", 0.95),
|
|
("tiny", 0.70),
|
|
):
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO proficiency (
|
|
model_id, provider, category, blended_score, source, last_updated
|
|
) VALUES (?, 'neuralwatt', 'coding_general', ?, 'self_eval_thin', '2026-01-01T00:00:00+00:00')
|
|
""",
|
|
(model_id, score),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _seed_energy(conn: sqlite3.Connection) -> None:
|
|
now = _now()
|
|
recent_rows = [
|
|
("cheap", 5.0e-05, 100, 0.25), # 2 days ago
|
|
("cheap", 3.0e-05, 200, 0.50), # 2 days ago
|
|
("dear", 1.0e-04, 150, 0.75), # 5 days ago
|
|
("tiny", 1.0e-05, 50, 1.00), # 10 days ago
|
|
]
|
|
for (model_id, kwh, tokens, attr) in recent_rows:
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO energy_observations (
|
|
model_id, provider, task_category, prompt_tokens,
|
|
completion_tokens, energy_kwh, attribution_ratio,
|
|
observed_at
|
|
) VALUES (?, 'neuralwatt', 'coding_general', 1000, ?, ?, ?, ?)
|
|
""",
|
|
(model_id, tokens, kwh, attr, (now - timedelta(days=2)).isoformat()),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _telemetry_provider_cfg(**objective_overrides) -> SimpleNamespace:
|
|
"""Build a cfg whose only dispatch provider is telemetry-backed."""
|
|
return SimpleNamespace(
|
|
objective=SimpleNamespace(**objective_overrides),
|
|
dispatch_providers={
|
|
"neuralwatt": SimpleNamespace(has_energy_telemetry=True),
|
|
},
|
|
)
|
|
|
|
|
|
# --- Import / no-circular-import smoke tests -----------------------------------
|
|
|
|
|
|
def test_metrics_can_be_imported_alone():
|
|
"""metrics.py must not require dispatcher — it *is* the cycle-breaker."""
|
|
# If this import raises ImportError (circular), we fail.
|
|
import metrics # noqa: F401
|
|
|
|
|
|
def test_dispatcher_imports_after_metrics():
|
|
"""importing metrics first, then dispatcher, must not raise."""
|
|
# This test runs *after* metrics has already been imported above.
|
|
# The import chain is: dispatcher → metrics (one-way).
|
|
assert hasattr(dispatcher, "app")
|
|
|
|
|
|
# --- quota_accounts tests -------------------------------------------------------
|
|
|
|
|
|
def test_quota_accounts_returns_correct_top_level_keys(tmp_path):
|
|
"""Result has period, accounts, spend, alarm keys."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, completion_tokens, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 0.10, 100, ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
assert "period" in result
|
|
assert "accounts" in result
|
|
assert "spend" in result
|
|
assert "alarm" in result
|
|
|
|
|
|
def test_quota_accounts_metered_plan_has_plan_block(tmp_path):
|
|
"""metered_plan shape includes plan block with kwh_per_period, used_kwh, used_fraction."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
for kwh in (0.10, 0.15):
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, completion_tokens, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', ?, 100, ?)",
|
|
(kwh, now.isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
assert acc["shape"] == "metered_plan"
|
|
assert acc["plan"]["kwh_per_period"] == 6.25
|
|
assert acc["plan"]["used_kwh"] == pytest.approx(0.25)
|
|
assert acc["plan"]["used_fraction"] == pytest.approx(0.25 / 6.25)
|
|
|
|
|
|
def test_quota_accounts_prepaid_credit_has_pool_block(tmp_path):
|
|
"""prepaid_credit shape includes pool block with balance from provider_balance_observations."""
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
cfg = SimpleNamespace(
|
|
objective=SimpleNamespace(),
|
|
dispatch_providers={
|
|
"neuralwatt": SimpleNamespace(
|
|
has_energy_telemetry=False,
|
|
balance_url="https://example.com/credits",
|
|
),
|
|
},
|
|
)
|
|
conn.execute(
|
|
"INSERT INTO provider_balance_observations "
|
|
"(provider, balance_usd, observed_at) "
|
|
"VALUES ('neuralwatt', ?, ?)",
|
|
(50.0, now.isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
assert acc["shape"] == "prepaid_credit"
|
|
assert acc["pool"]["balance_usd"] == pytest.approx(50.0)
|
|
assert acc["pool"]["age_seconds"] is not None
|
|
|
|
|
|
def test_quota_accounts_self_hosted_shape(tmp_path):
|
|
"""Provider named ollama-local gets self_hosted shape."""
|
|
conn = _make_db(tmp_path)
|
|
cfg = SimpleNamespace(
|
|
objective=SimpleNamespace(),
|
|
dispatch_providers={
|
|
"ollama-local": SimpleNamespace(has_energy_telemetry=False),
|
|
},
|
|
)
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
assert acc["shape"] == "self_hosted"
|
|
assert acc["provider"] == "ollama-local"
|
|
|
|
|
|
def test_quota_accounts_unmetered_shape(tmp_path):
|
|
"""Bare provider (no telemetry, no plan, no balance_url) gets unmetered shape."""
|
|
conn = _make_db(tmp_path)
|
|
cfg = SimpleNamespace(
|
|
objective=SimpleNamespace(),
|
|
dispatch_providers={
|
|
"bare": SimpleNamespace(has_energy_telemetry=False),
|
|
},
|
|
)
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
assert acc["shape"] == "unmetered"
|
|
|
|
|
|
def test_quota_accounts_period_with_billing_reset(tmp_path):
|
|
"""billing_reset_day=6 sets source=billing_reset_day with start and next_reset dates."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
|
|
conn = _make_db(tmp_path)
|
|
result = quota_accounts(conn, cfg)
|
|
assert result["period"]["source"] == "billing_reset_day"
|
|
assert result["period"]["start"] is not None
|
|
assert result["period"]["next_reset"] is not None
|
|
|
|
|
|
def test_quota_accounts_period_without_reset_day(tmp_path):
|
|
"""No billing_reset_day sets source=30d_rolling with next_reset=None."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
|
|
conn = _make_db(tmp_path)
|
|
result = quota_accounts(conn, cfg)
|
|
assert result["period"]["source"] == "30d_rolling"
|
|
assert result["period"]["next_reset"] is None
|
|
|
|
|
|
def test_quota_accounts_spend_block(tmp_path):
|
|
"""Spend aggregates cost_usd from energy rows per provider."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
for cost in (0.05, 0.03):
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, cost_usd, completion_tokens, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 0.001, ?, 100, ?)",
|
|
(cost, now.isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
assert result["spend"]["total_usd"] == pytest.approx(0.08)
|
|
assert "by_provider_usd" in result["spend"]
|
|
assert result["spend"]["by_provider_usd"]["neuralwatt"] == pytest.approx(0.08)
|
|
|
|
|
|
def test_quota_accounts_burn_rate_computed(tmp_path):
|
|
"""Burn rate computed from allowance_remaining_usd series over 3 hours."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
rows = [
|
|
("m", 1.60, now - timedelta(hours=3)),
|
|
("m", 1.20, now - timedelta(hours=2)),
|
|
("m", 0.80, now - timedelta(hours=1)),
|
|
("m", 0.60, now - timedelta(minutes=30)),
|
|
]
|
|
for model_id, balance, ts in rows:
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
|
|
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
|
|
(model_id, balance, ts.isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
assert acc["burn"]["burn_rate_usd_per_hour"] == pytest.approx(0.4)
|
|
assert acc["burn"]["projected_hours_remaining"] == pytest.approx(1.5)
|
|
|
|
|
|
def test_quota_accounts_top_up_resets_segment(tmp_path):
|
|
"""A credit top-up splits the window; burn uses only the latest segment."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
rows = [
|
|
("m", 3.00, now - timedelta(hours=5)),
|
|
("m", 2.00, now - timedelta(hours=4)),
|
|
("m", 1.00, now - timedelta(hours=3)),
|
|
("m", 10.00, now - timedelta(hours=2)),
|
|
("m", 9.25, now - timedelta(minutes=90)),
|
|
("m", 8.50, now - timedelta(minutes=30)),
|
|
]
|
|
for model_id, balance, ts in rows:
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
|
|
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
|
|
(model_id, balance, ts.isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
# Burn ~1.0 from post-top-up segment, not ~7.0 from overall MAX-MIN
|
|
assert acc["burn"]["burn_rate_usd_per_hour"] == pytest.approx(1.0)
|
|
|
|
|
|
def test_quota_accounts_min_samples_guard(tmp_path):
|
|
"""Segment with only 2 samples cannot compute a burn rate."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
for bal, hours_ago in ((10.0, 2), (9.0, 0.5)):
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
|
|
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
|
|
("m", bal, (now - timedelta(hours=hours_ago)).isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
assert acc["burn"] is None
|
|
|
|
|
|
def test_quota_accounts_min_hours_guard(tmp_path):
|
|
"""3 samples spanning only 10 minutes cannot compute a burn rate."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
for value, minutes_ago in ((10.0, 10), (9.0, 7), (8.0, 1)):
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
|
|
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
|
|
("m", value, (now - timedelta(minutes=minutes_ago)).isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
assert acc["burn"] is None
|
|
|
|
|
|
def test_quota_accounts_all_null_allowance(tmp_path):
|
|
"""All allowance_remaining_usd NULL -> burn is None, pool is None for metered_plan."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
for kwh, hours_ago in ((2.0, 2), (3.0, 1)):
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
|
|
"VALUES (?, 'neuralwatt', ?, NULL, ?)",
|
|
("m", kwh, (now - timedelta(hours=hours_ago)).isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
acc = result["accounts"][0]
|
|
assert acc["burn"] is None
|
|
assert acc["pool"] is None
|
|
|
|
|
|
def test_quota_accounts_interleaved_providers_independent(tmp_path):
|
|
"""Two providers' allowance series compute independently despite interleaved rows."""
|
|
conn = _make_db(tmp_path)
|
|
cfg = SimpleNamespace(
|
|
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6),
|
|
dispatch_providers={
|
|
"neuralwatt": SimpleNamespace(has_energy_telemetry=True),
|
|
"openrouter": SimpleNamespace(
|
|
has_energy_telemetry=False,
|
|
balance_url="https://openrouter.ai/api/v1/credits",
|
|
),
|
|
},
|
|
)
|
|
now = _now()
|
|
# NeuralWatt telemetry: burn 1.0 USD/h
|
|
for bal, at in (
|
|
(5.0, now - timedelta(hours=4)),
|
|
(4.0, now - timedelta(hours=3)),
|
|
(3.0, now - timedelta(hours=2)),
|
|
(2.0, now - timedelta(hours=1)),
|
|
):
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
|
|
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
|
|
("m", bal, at.isoformat()),
|
|
)
|
|
# OpenRouter polled: burn 4.0 USD/h
|
|
for bal, at in (
|
|
(20.0, now - timedelta(minutes=210)),
|
|
(16.0, now - timedelta(minutes=150)),
|
|
(12.0, now - timedelta(minutes=90)),
|
|
(8.0, now - timedelta(minutes=30)),
|
|
):
|
|
conn.execute(
|
|
"INSERT INTO provider_balance_observations "
|
|
"(provider, balance_usd, observed_at) "
|
|
"VALUES ('openrouter', ?, ?)",
|
|
(bal, at.isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
by_provider = {a["provider"]: a for a in result["accounts"]}
|
|
assert by_provider["neuralwatt"]["burn"]["burn_rate_usd_per_hour"] == pytest.approx(1.0)
|
|
assert by_provider["openrouter"]["burn"]["burn_rate_usd_per_hour"] == pytest.approx(4.0)
|
|
assert by_provider["neuralwatt"]["shape"] == "metered_plan"
|
|
assert by_provider["openrouter"]["shape"] == "prepaid_credit"
|
|
|
|
|
|
def test_quota_accounts_alarm_plan_pace_warning(tmp_path):
|
|
"""Usage pace > 1.25x triggers plan_pace alarm."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
# Seed 8.0 kWh to guarantee used_fraction > 1.25 * elapsed_fraction on any day
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, completion_tokens, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 8.0, 100, ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
assert result["alarm"] is not None
|
|
assert result["alarm"]["kind"] == "plan_pace"
|
|
assert result["alarm"]["severity"] in ("warning", "critical")
|
|
|
|
|
|
def test_quota_accounts_alarm_stale_reading(tmp_path):
|
|
"""A balance reading 4+ hours old triggers stale_reading alarm."""
|
|
conn = _make_db(tmp_path)
|
|
cfg = SimpleNamespace(
|
|
objective=SimpleNamespace(),
|
|
dispatch_providers={
|
|
"neuralwatt": SimpleNamespace(
|
|
has_energy_telemetry=False,
|
|
balance_url="https://example.com/credits",
|
|
),
|
|
},
|
|
)
|
|
now = _now()
|
|
conn.execute(
|
|
"INSERT INTO provider_balance_observations "
|
|
"(provider, balance_usd, observed_at) "
|
|
"VALUES ('neuralwatt', 50.0, ?)",
|
|
((now - timedelta(hours=5)).isoformat(),),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
assert result["alarm"] is not None
|
|
assert result["alarm"]["kind"] == "stale_reading"
|
|
|
|
|
|
def test_quota_accounts_alarm_plan_pace_critical(tmp_path):
|
|
"""Usage pace > 2.0 triggers critical severity."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25, billing_reset_day=6)
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
# Seed 15 kWh to guarantee pace > 2.0 on any day
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, energy_kwh, completion_tokens, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 15.0, 100, ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
conn.commit()
|
|
result = quota_accounts(conn, cfg)
|
|
assert result["alarm"]["severity"] == "critical"
|
|
|
|
|
|
def test_quota_accounts_empty_db(tmp_path):
|
|
"""Empty DB: zero spend, zero energy, accounts match provider count."""
|
|
cfg = _telemetry_provider_cfg(plan_kwh_per_period=6.25)
|
|
conn = _make_db(tmp_path)
|
|
result = quota_accounts(conn, cfg)
|
|
assert len(result["accounts"]) == 1
|
|
assert result["accounts"][0]["spend_usd"]["period"] == 0.0
|
|
assert result["accounts"][0]["energy"]["kwh_30d"] == 0.0
|
|
assert result["accounts"][0]["energy"]["calls_30d"] == 0
|
|
|
|
|
|
def test_local_energy_summary_reset_date_is_billing_period_start(tmp_path):
|
|
"""local_energy_summary mirrors the same reset_date/window_start split."""
|
|
cfg = SimpleNamespace(
|
|
local_energy=SimpleNamespace(enabled=True),
|
|
objective=SimpleNamespace(billing_reset_day=6),
|
|
)
|
|
conn = _make_db(tmp_path)
|
|
now = datetime.now(timezone.utc)
|
|
today = now.date()
|
|
reset_day = 6
|
|
if today.day >= reset_day:
|
|
expected_period_start = date(today.year, today.month, reset_day).isoformat()
|
|
else:
|
|
if today.month == 1:
|
|
expected_period_start = date(today.year - 1, 12, reset_day).isoformat()
|
|
else:
|
|
expected_period_start = date(today.year, today.month - 1, reset_day).isoformat()
|
|
expected_rolling_start = (today - timedelta(days=30)).isoformat()
|
|
|
|
result = local_energy_summary(conn, cfg)
|
|
assert result["reset_date"] == expected_period_start
|
|
assert result["window_start_30d"] == expected_rolling_start
|
|
|
|
cfg_unconfigured = SimpleNamespace(
|
|
local_energy=SimpleNamespace(enabled=True),
|
|
objective=SimpleNamespace(),
|
|
)
|
|
result_unconfigured = local_energy_summary(conn, cfg_unconfigured)
|
|
assert "reset_date" in result_unconfigured
|
|
assert result_unconfigured["reset_date"] is None
|
|
assert result_unconfigured["window_start_30d"] == expected_rolling_start
|
|
|
|
|
|
def test_pinch_summary_reset_date_is_billing_period_start(tmp_path):
|
|
"""pinch_summary mirrors the same reset_date/window_start split."""
|
|
cfg = SimpleNamespace(
|
|
pinch=SimpleNamespace(enabled=True),
|
|
objective=SimpleNamespace(assumed_cache_rate=0.917, billing_reset_day=6),
|
|
)
|
|
conn = _make_db(tmp_path)
|
|
now = datetime.now(timezone.utc)
|
|
today = now.date()
|
|
reset_day = 6
|
|
if today.day >= reset_day:
|
|
expected_period_start = date(today.year, today.month, reset_day).isoformat()
|
|
else:
|
|
if today.month == 1:
|
|
expected_period_start = date(today.year - 1, 12, reset_day).isoformat()
|
|
else:
|
|
expected_period_start = date(today.year, today.month - 1, reset_day).isoformat()
|
|
expected_rolling_start = (today - timedelta(days=30)).isoformat()
|
|
|
|
result = pinch_summary(conn, cfg)
|
|
assert result["reset_date"] == expected_period_start
|
|
assert result["window_start_30d"] == expected_rolling_start
|
|
|
|
cfg_unconfigured = SimpleNamespace(
|
|
pinch=SimpleNamespace(enabled=True),
|
|
objective=SimpleNamespace(assumed_cache_rate=0.917),
|
|
)
|
|
result_unconfigured = pinch_summary(conn, cfg_unconfigured)
|
|
assert "reset_date" in result_unconfigured
|
|
assert result_unconfigured["reset_date"] is None
|
|
assert result_unconfigured["window_start_30d"] == expected_rolling_start
|
|
|
|
|
|
# --- _next_reset_date helper --------------------------------------------------
|
|
|
|
|
|
def test_next_reset_date_this_month():
|
|
"""When today is before the anchor day, reset lands this month."""
|
|
from datetime import date
|
|
|
|
assert _next_reset_date(6, date(2026, 9, 1)) == "2026-09-06"
|
|
|
|
|
|
def test_next_reset_date_rolls_to_next_month():
|
|
"""When today's day-of-month is on or past the anchor, roll forward."""
|
|
from datetime import date
|
|
|
|
assert _next_reset_date(6, date(2026, 9, 6)) == "2026-10-06"
|
|
assert _next_reset_date(6, date(2026, 9, 15)) == "2026-10-06"
|
|
|
|
|
|
def test_next_reset_date_handles_december_rollover():
|
|
"""December rolls over to January of the next year."""
|
|
from datetime import date
|
|
|
|
assert _next_reset_date(6, date(2026, 12, 7)) == "2027-01-06"
|
|
assert _next_reset_date(28, date(2026, 12, 29)) == "2027-01-28"
|
|
|
|
|
|
# --- scoring_coverage tests ---------------------------------------------------
|
|
|
|
|
|
def test_scoring_coverage_has_all_keys(tmp_path):
|
|
"""The return dict always has these keys, even when empty."""
|
|
conn = _make_db(tmp_path)
|
|
# Seed routable models
|
|
_seed_models(conn)
|
|
result = scoring_coverage(conn, CFG)
|
|
assert "routable_models" in result
|
|
assert "with_energy_data" in result
|
|
assert "with_proficiency_data" in result
|
|
assert "quota" in result
|
|
assert "warnings" in result
|
|
|
|
|
|
def test_scoring_coverage_warning_when_no_energy(tmp_path):
|
|
"""Models with no seed_reference observations produce a warning."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
# No energy_observations rows at all
|
|
result = scoring_coverage(conn, CFG)
|
|
warning_texts = result["warnings"]
|
|
assert any("no reference-workload observations" in w for w in warning_texts)
|
|
assert result["with_energy_data"] == 0
|
|
|
|
|
|
def test_scoring_coverage_no_warning_when_full_coverage(tmp_path):
|
|
"""When every routable model has energy + proficiency, no warnings."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
# Seed SEED_CATEGORY energy observations
|
|
now = _now()
|
|
for model in ["cheap", "dear"]:
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, task_category, prompt_tokens, "
|
|
"completion_tokens, energy_kwh, attribution_ratio, observed_at) "
|
|
"VALUES (?, 'neuralwatt', 'seed_reference', 1000, 100, 0.001, 0.25, ?)",
|
|
(model, now.isoformat()),
|
|
)
|
|
conn.execute(
|
|
"INSERT INTO proficiency "
|
|
"(model_id, provider, category, blended_score, source, last_updated) "
|
|
"VALUES (?, 'neuralwatt', 'coding_general', 0.9, 'self_eval', ?)",
|
|
(model, now.isoformat()),
|
|
)
|
|
conn.commit()
|
|
result = scoring_coverage(conn, CFG)
|
|
# 'tiny' has no data so we expect warnings. Let's also add 'tiny'.
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, task_category, prompt_tokens, "
|
|
"completion_tokens, energy_kwh, attribution_ratio, observed_at) "
|
|
"VALUES ('tiny', 'neuralwatt', 'seed_reference', 1000, 100, 0.001, 0.25, ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
conn.execute(
|
|
"INSERT INTO proficiency "
|
|
"(model_id, provider, category, blended_score, source, last_updated) "
|
|
"VALUES ('tiny', 'neuralwatt', 'coding_general', 0.7, 'self_eval', ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
conn.commit()
|
|
result = scoring_coverage(conn, CFG)
|
|
assert result["with_energy_data"] == 3
|
|
assert result["with_proficiency_data"] == 3
|
|
assert len(result["warnings"]) == 0
|
|
|
|
|
|
def test_scoring_coverage_empty_db(tmp_path):
|
|
"""Empty DB: 0 routable, no warnings, quota=None."""
|
|
conn = _make_db(tmp_path)
|
|
result = scoring_coverage(conn, CFG)
|
|
assert result["routable_models"] == 0
|
|
assert result["with_energy_data"] == 0
|
|
assert result["with_proficiency_data"] == 0
|
|
# quota returns None because plan_kwh_per_period may be None in default cfg
|
|
# (it is 6.25 by default, but let's just check the structure)
|
|
assert result["warnings"] == []
|
|
|
|
|
|
def test_scoring_coverage_warns_per_provider_staleness(tmp_path):
|
|
"""A stale second provider must produce a warning naming that provider."""
|
|
conn = _make_db(tmp_path)
|
|
fresh = _now().isoformat()
|
|
stale = (_now() - timedelta(days=2)).isoformat()
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES ('fresh-model', 'neuralwatt', 'fresh-model', 2, 262128, 192500, 16384,
|
|
0.30, 0.10, 1, 1, 'standard', 'default', 'full', 'public', 'active', ?)
|
|
""",
|
|
(fresh,),
|
|
)
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES ('old-model', 'other-provider', 'old-model', 2, 262128, 192500, 16384,
|
|
0.30, 0.10, 1, 1, 'standard', 'default', 'full', 'public', 'active', ?)
|
|
""",
|
|
(stale,),
|
|
)
|
|
conn.commit()
|
|
result = scoring_coverage(conn, CFG)
|
|
neuralwatt_warnings = [w for w in result["warnings"] if "days ago" in w]
|
|
assert any("[other-provider]" in w for w in neuralwatt_warnings)
|
|
assert not any("[neuralwatt]" in w for w in neuralwatt_warnings)
|
|
# One formatted number. This read "1.0.8 days ago" live, because an int
|
|
# and a rounded fraction joined by a dot keeps the fraction's own "0.".
|
|
assert re.search(r"catalog last polled \d+\.\d days ago", neuralwatt_warnings[0])
|
|
|
|
|
|
_ADMIN_TABLE_SQL = """
|
|
CREATE TABLE IF NOT EXISTS admin_model_overrides (
|
|
model_id TEXT NOT NULL,
|
|
provider TEXT NOT NULL,
|
|
availability TEXT NOT NULL,
|
|
reason TEXT,
|
|
updated_at TEXT NOT NULL,
|
|
PRIMARY KEY (model_id, provider)
|
|
);
|
|
CREATE INDEX IF NOT EXISTS idx_admin_model_overrides_availability
|
|
ON admin_model_overrides (availability);
|
|
"""
|
|
|
|
|
|
def test_context_ceilings_excludes_admin_deprecated_models(tmp_path):
|
|
"""Admin-deprecated models are excluded from context ceilings even when the
|
|
raw ``models.availability`` column still reads ``active``.
|
|
|
|
Regression for the 2026-09-01 outage: ``admin_model_overrides`` carries
|
|
the override while the catalog row stays active, so ``exclude_deprecated``
|
|
on the raw row is not enough.
|
|
"""
|
|
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
|
fresh = _now().isoformat()
|
|
for model_id, eff_ctx in (("big", 200000), ("small", 100000)):
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES (?, 'neuralwatt', ?, 3, ?, ?, 16384, 0.30, 0.10,
|
|
1, 1, 'standard', 'default', 'full', 'public', 'active',
|
|
?)
|
|
""",
|
|
(model_id, model_id, eff_ctx, eff_ctx, fresh),
|
|
)
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO admin_model_overrides (
|
|
model_id, provider, availability, reason, updated_at
|
|
) VALUES (?, 'neuralwatt', 'deprecated', 'test', ?)
|
|
""",
|
|
("big", fresh),
|
|
)
|
|
conn.commit()
|
|
|
|
ctx = context_ceilings(conn, CFG)
|
|
|
|
key = (3, "interactive")
|
|
assert key in ctx
|
|
assert ctx[key]["ceiling"] == 100000
|
|
assert ctx[key]["count"] == 1
|
|
|
|
|
|
# --- capability sub-ceiling demand warnings (2026-09-04 incident) ---------------
|
|
|
|
|
|
def _insert_model_row(
|
|
conn: sqlite3.Connection,
|
|
*,
|
|
model_id: str,
|
|
tier: int,
|
|
effective_context_window: int,
|
|
supports_vision: int,
|
|
supports_json_mode: int = 1,
|
|
) -> None:
|
|
"""Insert one active, routable catalog row with explicit capability flags.
|
|
|
|
``_seed_models`` covers the common case but hardcodes its capability
|
|
flags and effective window; the incident-reproduction tests need both
|
|
per-row.
|
|
"""
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES (?, 'neuralwatt', ?, ?, ?, ?, 16384, 0.30, 0.10,
|
|
?, ?, 'standard', 'default', 'full', 'public', 'active',
|
|
?)
|
|
""",
|
|
(
|
|
model_id,
|
|
model_id,
|
|
tier,
|
|
effective_context_window,
|
|
effective_context_window,
|
|
supports_vision,
|
|
supports_json_mode,
|
|
_now().isoformat(),
|
|
),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _deprecate_model(conn: sqlite3.Connection, model_id: str) -> None:
|
|
"""Insert an admin override marking *model_id* deprecated (reason: cost)."""
|
|
conn.execute(
|
|
"INSERT INTO admin_model_overrides "
|
|
"(model_id, provider, availability, reason, updated_at) "
|
|
"VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)",
|
|
(model_id, _now().isoformat()),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _insert_rejected_decision(
|
|
conn: sqlite3.Connection,
|
|
*,
|
|
tier: int,
|
|
reason: str,
|
|
observed_at: str,
|
|
tokens: int = 200000,
|
|
images: int = 0,
|
|
json_mode: int = 0,
|
|
) -> None:
|
|
"""Insert a route_decisions row that selected no model (a 422 rejection).
|
|
|
|
``kind='chat'`` so the demand queries (which filter on kind) see it;
|
|
``latency_tolerance='interactive'`` to match the tier-1 bucket key the
|
|
ceiling checks use.
|
|
"""
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO route_decisions (
|
|
observed_at, kind, task_category, task_tier,
|
|
required_context_tokens, confidence, classifier_ms,
|
|
classification_source, latency_tolerance, candidates_considered,
|
|
selected_model, selected_provider, rejected_reason,
|
|
session_key, tools, images, json_mode, streamed
|
|
) VALUES (?, 'chat', 'coding_general', ?, ?, NULL, NULL,
|
|
'override', 'interactive', 0, NULL, NULL, ?,
|
|
'sess', 0, ?, ?, 0)
|
|
""",
|
|
(observed_at, tier, tokens, reason, images, json_mode),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _seed_incident_state(conn: sqlite3.Connection) -> str:
|
|
"""Seed the 2026-09-04 incident model state; returns the shared 'now' stamp.
|
|
|
|
Three tier-1 rows: kimi-k3 and kimi-k3-fast — the only vision-capable
|
|
large-context models — deprecated through admin overrides, and
|
|
deepseek-v4-flash (no vision, 262,128 effective tokens) left active.
|
|
The overall tier-1 interactive ceiling stays high while the vision
|
|
sub-ceiling collapses to zero, which is exactly the shape every
|
|
pre-2026-09-05 ceiling check was blind to.
|
|
"""
|
|
for model_id, vision, eff_ctx in (
|
|
("kimi-k3", 1, 782324),
|
|
("kimi-k3-fast", 1, 782324),
|
|
("deepseek-v4-flash", 0, 262128),
|
|
):
|
|
_insert_model_row(
|
|
conn,
|
|
model_id=model_id,
|
|
tier=1,
|
|
effective_context_window=eff_ctx,
|
|
supports_vision=vision,
|
|
)
|
|
for model_id in ("kimi-k3", "kimi-k3-fast"):
|
|
_deprecate_model(conn, model_id)
|
|
return _now().isoformat()
|
|
|
|
|
|
def test_capability_demand_warning_incident_reproduction(tmp_path):
|
|
"""The 2026-09-04 incident state warns on the vision sub-ceiling while
|
|
every existing ``(tier, latency_tolerance)`` check stays silent.
|
|
|
|
In the incident, kimi-k3 and kimi-k3-fast — the only vision-capable
|
|
tier-1 models with a large context — were deprecated through admin
|
|
overrides, so a 200,000-token image request was unservable even though
|
|
the overall tier-1 ceiling (262,128, from non-vision deepseek-v4-flash)
|
|
comfortably covered that demand. The new capability check must fire;
|
|
the existing demand_ceiling_warnings must NOT — that silence is what
|
|
hid the incident for 19 hours.
|
|
"""
|
|
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
|
ts = _seed_incident_state(conn)
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=1,
|
|
reason=(
|
|
"context >= 242486 tokens; "
|
|
"vision-capable model (request carries image(s))"
|
|
),
|
|
observed_at=ts,
|
|
tokens=200000,
|
|
images=1,
|
|
)
|
|
|
|
# The overall tier-1 interactive ceiling is untouched by the deprecations.
|
|
ctx = context_ceilings(conn, CFG)
|
|
assert ctx[(1, "interactive")]["ceiling"] == 262128
|
|
|
|
# The vision-gated subset collapsed: no candidate survives the gate.
|
|
cap_ctx = capability_ceilings(conn, CFG)
|
|
assert cap_ctx["vision"][(1, "interactive")]["ceiling"] == 0
|
|
assert cap_ctx["vision"][(1, "interactive")]["count"] == 0
|
|
# deepseek-v4-flash still serves the json_mode subset.
|
|
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 262128
|
|
|
|
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
|
|
assert len(warnings) == 1
|
|
w = warnings[0]
|
|
assert "vision" in w
|
|
assert "tier 1" in w
|
|
assert "ceiling (0)" in w
|
|
assert "200000" in w
|
|
|
|
# The pre-existing check must stay silent on the same state: the overall
|
|
# tier-1 ceiling (262128) still exceeds the observed max demand (200000).
|
|
assert demand_ceiling_warnings(conn, CFG, ctx, [1]) == []
|
|
|
|
|
|
def test_capability_demand_no_demand_no_warning(tmp_path):
|
|
"""A state with demand but no capability-flagged demand warns about nothing.
|
|
|
|
Requests exist at the tier — far above every ceiling — but none carry
|
|
images or ask for JSON mode, so both capability subsets are unexercised:
|
|
an unexercised sub-ceiling is not a warning.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=2,
|
|
reason="context >= 999999 tokens",
|
|
observed_at=_now().isoformat(),
|
|
tokens=999999,
|
|
)
|
|
|
|
assert capability_demand_warnings(conn, CFG, capability_ceilings(conn, CFG)) == []
|
|
|
|
|
|
def test_capability_json_mode_demand_warning(tmp_path):
|
|
"""Deprecating the only json_mode-capable model trips the json_mode warning.
|
|
|
|
Same incident shape as the vision case, different capability dimension:
|
|
the json_mode sub-ceiling collapses to zero while the overall ceiling
|
|
stays fine, and a json_mode request that cannot be served produces a
|
|
warning naming json_mode.
|
|
"""
|
|
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
|
_insert_model_row(
|
|
conn,
|
|
model_id="jm-big",
|
|
tier=1,
|
|
effective_context_window=782324,
|
|
supports_vision=0,
|
|
supports_json_mode=1,
|
|
)
|
|
_insert_model_row(
|
|
conn,
|
|
model_id="plain-fallback",
|
|
tier=1,
|
|
effective_context_window=262128,
|
|
supports_vision=0,
|
|
supports_json_mode=0,
|
|
)
|
|
_deprecate_model(conn, "jm-big")
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=1,
|
|
reason=(
|
|
"context >= 242486 tokens; json-mode-capable model "
|
|
"(json_mode requested)"
|
|
),
|
|
observed_at=_now().isoformat(),
|
|
tokens=200000,
|
|
json_mode=1,
|
|
)
|
|
|
|
cap_ctx = capability_ceilings(conn, CFG)
|
|
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 0
|
|
|
|
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
|
|
assert len(warnings) == 1
|
|
w = warnings[0]
|
|
assert "json_mode" in w
|
|
assert "tier 1" in w
|
|
assert "ceiling (0)" in w
|
|
assert "200000" in w
|
|
|
|
|
|
# --- NULL required_context_tokens guards (regression) ---------------------------
|
|
#
|
|
# route_decisions.required_context_tokens is a nullable column (rows written
|
|
# before the column existed, or decisions persisted without a measured token
|
|
# count). SQLite MAX() over an all-NULL group returns NULL, not 0, so a
|
|
# window whose every row lacks a token count yields observed_max=None —
|
|
# which used to raise ``None > ceiling`` TypeError inside the demand-ceiling
|
|
# checks, and crashed _percentile on the escalation path. Semantics: NULL is
|
|
# "no demand recorded for this row", not zero — so a NULL row is excluded
|
|
# from comparisons, never fabricated into a warning.
|
|
|
|
|
|
def test_demand_ceiling_all_null_7d_window_no_crash_no_warning(tmp_path):
|
|
"""A tier whose 7d rows are all NULL does not crash demand_ceiling_warnings.
|
|
|
|
Site A regression: ``MAX(required_context_tokens)`` over an all-NULL
|
|
window returns NULL, so ``observed_max`` used to become None and the
|
|
``None > ceiling`` comparison raised TypeError. Seeding the tier's
|
|
demand proof in the 30d window (one non-NULL row at 20 days ago) plus
|
|
12 NULL rows inside 7d puts the tier in ``tiers_with_demand`` while
|
|
keeping the observed-max window at 7d — exactly the shape that used to
|
|
crash. With the guard, the comparison is skipped: no crash, and no
|
|
false "0 > ceiling" warning either.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
ts = _now().isoformat()
|
|
old = (_now() - timedelta(days=20)).isoformat()
|
|
for _ in range(12):
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=2,
|
|
reason="context >= 999999 tokens",
|
|
observed_at=ts,
|
|
tokens=None,
|
|
)
|
|
# Non-NULL demand inside 30d (outside 7d) proves the tier has demand,
|
|
# while every row inside the 7d observed-max window stays NULL.
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=2,
|
|
reason="context >= 999999 tokens",
|
|
observed_at=old,
|
|
tokens=200000,
|
|
)
|
|
|
|
ctx = context_ceilings(conn, CFG)
|
|
assert ctx[(2, "interactive")]["ceiling"] > 0
|
|
|
|
# Old code: None > ceiling -> TypeError. New code: comparison skipped.
|
|
assert demand_ceiling_warnings(conn, CFG, ctx, [1, 2]) == []
|
|
|
|
|
|
def test_demand_ceiling_escalation_percentile_skips_null_tokens(tmp_path):
|
|
"""NULL tokens in the escalation p95 query are excluded, not sorted.
|
|
|
|
Site B regression: the escalation-hazard probe fetched every tier's 7d
|
|
``required_context_tokens`` raw and fed the list to _percentile, whose
|
|
``sorted()`` raises TypeError the moment a None meets an int. A mixed
|
|
NULL/non-NULL 7d window for tier 1 used to crash; after the fix the
|
|
NULL row is filtered out and the percentile is computed over the
|
|
remaining values. 100000 < the tier-2 ceiling, so no escalation
|
|
warning fires either — the assertion proves both.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
ts = _now().isoformat()
|
|
_insert_rejected_decision(
|
|
conn, tier=1, reason="context >= 999999 tokens", observed_at=ts, tokens=None
|
|
)
|
|
_insert_rejected_decision(
|
|
conn, tier=1, reason="context >= 999999 tokens", observed_at=ts, tokens=100000
|
|
)
|
|
|
|
ctx = context_ceilings(conn, CFG)
|
|
# Escalation enabled in the default CFG; tier-2 bucket must exist and
|
|
# carry a non-zero ceiling or the percentile probe would be skipped.
|
|
assert getattr(CFG.escalation, "enabled", False)
|
|
assert ctx[(2, "interactive")]["ceiling"] > 0
|
|
|
|
# Old code: _percentile([None, 100000], 95) -> TypeError on sorted().
|
|
# New code: None filtered out, p95 = 100000 < ceiling, no warning.
|
|
assert demand_ceiling_warnings(conn, CFG, ctx, [1, 2]) == []
|
|
|
|
|
|
def test_capability_demand_all_null_7d_window_no_crash_no_warning(tmp_path):
|
|
"""A capability-gated tier whose 7d rows are all NULL does not crash.
|
|
|
|
Site C regression: same all-NULL MAX shape as the tier-wide demand
|
|
check, but inside capability_demand_warnings' vision dimension. One
|
|
vision-carrying row with a non-NULL token count at 20 days ago puts
|
|
the tier's vision bucket past the unexercised-silence guard, and 12
|
|
NULL-token vision rows inside 7d keep the observed-max window at 7d —
|
|
the exact shape that used to raise ``None > ceiling`` TypeError. With
|
|
the guard, no crash and no fabricated warning.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
ts = _now().isoformat()
|
|
old = (_now() - timedelta(days=20)).isoformat()
|
|
for _ in range(12):
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=2,
|
|
reason="context >= 999999 tokens; vision-capable model",
|
|
observed_at=ts,
|
|
tokens=None,
|
|
images=1,
|
|
)
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=2,
|
|
reason="context >= 999999 tokens; vision-capable model",
|
|
observed_at=old,
|
|
tokens=200000,
|
|
images=1,
|
|
)
|
|
|
|
cap_ctx = capability_ceilings(conn, CFG)
|
|
assert cap_ctx["vision"][(2, "interactive")]["ceiling"] > 0
|
|
|
|
# Old code: None > ceiling -> TypeError. New code: comparison skipped.
|
|
assert capability_demand_warnings(conn, CFG, cap_ctx) == []
|
|
|
|
|
|
# --- rejection warnings (reactive rejection detector) --------------------------
|
|
|
|
|
|
def test_rejection_warnings_zero_rejections(tmp_path):
|
|
"""Zero NULL-selection rows produce no rejection warnings.
|
|
|
|
The absence of rejections is not a signal worth reporting — the
|
|
detector watches rates and new patterns, not mere existence.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
assert rejection_warnings(conn, CFG) == []
|
|
|
|
|
|
def test_rejection_warnings_novel_group_at_incident_volume_fires(tmp_path):
|
|
"""The 2026-09-04 incident shape — a novel group of only 2 — must fire.
|
|
|
|
The incident produced exactly 2 image rejections, below the routine
|
|
noise floor (about 3/hour on this deployment), so no rate threshold can
|
|
catch it; the novelty prong — absent from the 24h baseline, count
|
|
>= 2 — is what must fire here. Both rows share one timestamp so the
|
|
rendered most-recent value is deterministic.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
ts = _now().isoformat()
|
|
for reason in (
|
|
"tier >= 1; context >= 242486 tokens; interactive; "
|
|
"vision-capable model (request carries image(s))",
|
|
"tier >= 1; context >= 195000 tokens; interactive; "
|
|
"vision-capable model (request carries image(s))",
|
|
):
|
|
_insert_rejected_decision(conn, tier=1, reason=reason, observed_at=ts)
|
|
|
|
warnings = rejection_warnings(conn, CFG)
|
|
assert len(warnings) == 1
|
|
w = warnings[0]
|
|
assert w.startswith("new rejection pattern:")
|
|
# Token counts collapse: both raw reasons normalize to "context >= N tokens".
|
|
assert "2 rejection(s)" in w
|
|
assert "context >= N tokens" in w
|
|
# The tier is rendered from the structured task_tier column, not the string.
|
|
assert "(tier 1;" in w
|
|
assert ts in w
|
|
|
|
|
|
def test_rejection_warnings_familiar_group_below_threshold_silent(tmp_path):
|
|
"""Routine over-large tier-3 traffic — 3/hour and familiar — stays silent.
|
|
|
|
Regression for the measured routine floor: this deployment routinely
|
|
sees about 3 rejections per hour of ordinary over-large tier-3 requests
|
|
that behave as designed. The group is familiar (present in the 24h
|
|
baseline window) and below ``rejection_warning_min_count`` (6), so
|
|
neither the novelty prong nor the rate prong fires.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
now = _now()
|
|
# Baseline occurrence: 2h ago — inside the 24h baseline, outside the 1h
|
|
# alert window.
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=3,
|
|
reason="tier >= 3; context >= 40000 tokens; interactive",
|
|
observed_at=(now - timedelta(hours=2)).isoformat(),
|
|
)
|
|
# The live-DB-measured routine hourly volume, with distinct token counts
|
|
# so digit normalization does real grouping work.
|
|
for tokens in (30000, 31000, 32000):
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=3,
|
|
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
|
|
observed_at=now.isoformat(),
|
|
)
|
|
|
|
assert rejection_warnings(conn, CFG) == []
|
|
|
|
|
|
def test_rejection_warnings_familiar_group_at_threshold_fires(tmp_path):
|
|
"""A familiar group at ``rejection_warning_min_count`` (6) fires as a rate."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
now = _now()
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=3,
|
|
reason="tier >= 3; context >= 40000 tokens; interactive",
|
|
observed_at=(now - timedelta(hours=2)).isoformat(),
|
|
)
|
|
for tokens in (30000, 31000, 32000, 33000, 34000, 35000):
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=3,
|
|
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
|
|
observed_at=now.isoformat(),
|
|
)
|
|
|
|
warnings = rejection_warnings(conn, CFG)
|
|
assert len(warnings) == 1
|
|
w = warnings[0]
|
|
assert w.startswith("rejection rate:")
|
|
assert "6 rejection(s)" in w
|
|
|
|
|
|
def test_rejection_warnings_novel_single_rejection_silent(tmp_path):
|
|
"""A single novel rejection stays silent.
|
|
|
|
A genuinely impossible one-off request SHOULD 422; presence of one
|
|
rejection is not a signal. n=1 is below NOVEL_GROUP_MIN_COUNT (2).
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=1,
|
|
reason=(
|
|
"tier >= 1; context >= 242486 tokens; interactive; "
|
|
"vision-capable model (request carries image(s))"
|
|
),
|
|
observed_at=_now().isoformat(),
|
|
)
|
|
|
|
assert rejection_warnings(conn, CFG) == []
|
|
|
|
|
|
def test_rejection_warnings_groups_by_structured_tier(tmp_path):
|
|
"""Tier-1 and tier-3 rejections of the same normalized shape never merge.
|
|
|
|
Regression for the sibling finding of the incident review: digit
|
|
normalization erases the tier embedded in the reason string ("tier >= 1"
|
|
and "tier >= 3" both become "tier >= N"), so grouping by normalized
|
|
string alone would produce ONE merged count-4 group, hiding which
|
|
tier's candidate set went empty. The discriminator is the structured
|
|
``task_tier`` column, rendered back into the body as "(tier N; ...)".
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
ts = _now().isoformat()
|
|
for tier, tokens in ((1, 242486), (1, 50000), (3, 229702), (3, 30000)):
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=tier,
|
|
reason=f"tier >= {tier}; context >= {tokens} tokens; interactive",
|
|
observed_at=ts,
|
|
)
|
|
|
|
warnings = rejection_warnings(conn, CFG)
|
|
assert len(warnings) == 2
|
|
tier1 = [w for w in warnings if "(tier 1;" in w]
|
|
tier3 = [w for w in warnings if "(tier 3;" in w]
|
|
assert len(tier1) == 1
|
|
assert len(tier3) == 1
|
|
assert "2 rejection(s)" in tier1[0]
|
|
assert "2 rejection(s)" in tier3[0]
|
|
|
|
|
|
def test_rejection_warnings_old_rejections_not_counted(tmp_path):
|
|
"""A rejection outside both windows is not counted at all.
|
|
|
|
Two same-shape rows 3 days old reach the novelty floor of 2, so the
|
|
only thing that keeps them silent is their age — outside the 1h alert
|
|
window and beyond the 25h total pull window they are not even read.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
for tokens in (242486, 195000):
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=1,
|
|
reason=f"tier >= 1; context >= {tokens} tokens; interactive",
|
|
observed_at=(_now() - timedelta(days=3)).isoformat(),
|
|
)
|
|
|
|
assert rejection_warnings(conn, CFG) == []
|
|
|
|
|
|
def test_rejection_warnings_custom_window(tmp_path):
|
|
"""``rejection_warning_window_hours`` widens the alert window.
|
|
|
|
Two same-shape rows 2h old sit outside the default 1h window but
|
|
inside a 3h window; with the knob set they form a novel group at the
|
|
incident volume and must fire.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
two_hours_ago = (_now() - timedelta(hours=2)).isoformat()
|
|
for tokens in (242486, 195000):
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=1,
|
|
reason=(
|
|
f"tier >= 1; context >= {tokens} tokens; interactive; "
|
|
"vision-capable model (request carries image(s))"
|
|
),
|
|
observed_at=two_hours_ago,
|
|
)
|
|
|
|
wide_cfg = SimpleNamespace(
|
|
objective=SimpleNamespace(
|
|
rejection_warning_window_hours=3,
|
|
rejection_warning_baseline_hours=24,
|
|
rejection_warning_min_count=6,
|
|
)
|
|
)
|
|
|
|
warnings = rejection_warnings(conn, wide_cfg)
|
|
assert len(warnings) == 1
|
|
assert warnings[0].startswith("new rejection pattern:")
|
|
assert "2 rejection(s)" in warnings[0]
|
|
|
|
|
|
def test_scoring_coverage_includes_new_warnings(tmp_path):
|
|
"""``scoring_coverage`` surfaces both new warning classes on the incident state.
|
|
|
|
The 2026-09-04 incident produced two image rejections against a catalog
|
|
whose vision sub-ceiling had collapsed. Both the capability demand
|
|
warning and the novel rejection pattern must reach the same warnings
|
|
list that /metrics, the admin portal and the TUI read.
|
|
"""
|
|
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
|
ts = _seed_incident_state(conn)
|
|
for tokens in (242486, 195000):
|
|
_insert_rejected_decision(
|
|
conn,
|
|
tier=1,
|
|
reason=(
|
|
f"context >= {tokens} tokens; "
|
|
"vision-capable model (request carries image(s))"
|
|
),
|
|
observed_at=ts,
|
|
images=1,
|
|
)
|
|
|
|
result = scoring_coverage(conn, CFG)
|
|
warnings = result["warnings"]
|
|
assert any("vision" in w for w in warnings)
|
|
assert any(w.startswith("new rejection pattern:") for w in warnings)
|
|
|
|
|
|
# --- recent_decisions tests ---------------------------------------------------
|
|
|
|
|
|
def test_recent_decisions_returns_rows_in_desc_order(tmp_path, monkeypatch):
|
|
"""Rows come back ordered by id DESC, and selected_provider is included."""
|
|
conn = _make_db(tmp_path)
|
|
# Seed two models
|
|
_seed_models(conn)
|
|
|
|
# Create fake dispatcher state to use TestClient, but we'll insert
|
|
# route_decisions rows directly and call recent_decisions(conn).
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO route_decisions (
|
|
observed_at, kind, task_category, task_tier, required_context_tokens,
|
|
confidence, classifier_ms, classification_source, latency_tolerance,
|
|
candidates_considered, selected_model, selected_provider,
|
|
runner_up_models, est_cost_usd, est_proficiency,
|
|
session_key, tools, images, json_mode, streamed
|
|
) VALUES (?, 'route', 'coding_general', 2, 100, 0.95, 200,
|
|
'classifier', 'interactive', 5, 'cheap', 'neuralwatt',
|
|
'[{"model_id":"dear","provider":"neuralwatt"}]',
|
|
0.001, 0.9, 'abc123', 0, 0, 0, 0)
|
|
""",
|
|
(datetime.now(timezone.utc).isoformat(),),
|
|
)
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO route_decisions (
|
|
observed_at, kind, task_category, task_tier, required_context_tokens,
|
|
confidence, classifier_ms, classification_source, latency_tolerance,
|
|
candidates_considered, selected_model, selected_provider,
|
|
runner_up_models, est_cost_usd, est_proficiency,
|
|
session_key, tools, images, json_mode, streamed
|
|
) VALUES (?, 'route', 'docs_writing', 1, 50, 0.88, 150,
|
|
'classifier', 'interactive', 3, 'dear', 'neuralwatt',
|
|
NULL, 0.005, 0.95, 'def456', 0, 0, 0, 0)
|
|
""",
|
|
(datetime.now(timezone.utc).isoformat(),),
|
|
)
|
|
conn.commit()
|
|
|
|
rows = recent_decisions(conn, limit=50)
|
|
assert len(rows) == 2
|
|
# DESC order: dear (id=2) first
|
|
assert rows[0]["selected_model"] == "dear"
|
|
assert rows[0]["kind"] == "route"
|
|
assert rows[0]["selected_provider"] == "neuralwatt"
|
|
assert rows[1]["selected_model"] == "cheap"
|
|
|
|
|
|
def test_recent_decisions_respects_limit(tmp_path):
|
|
"""limit=1 should return only one row regardless of DB content."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
now = _now().isoformat()
|
|
for i in range(5):
|
|
conn.execute(
|
|
"INSERT INTO route_decisions "
|
|
"(observed_at, kind, selected_model, selected_provider) "
|
|
"VALUES (?, 'route', 'm', 'neuralwatt')",
|
|
(now,),
|
|
)
|
|
conn.commit()
|
|
rows = recent_decisions(conn, limit=1)
|
|
assert len(rows) == 1
|
|
|
|
|
|
def test_recent_decisions_empty_db(tmp_path):
|
|
"""No rows: returns an empty list, not an error."""
|
|
conn = _make_db(tmp_path)
|
|
assert recent_decisions(conn) == []
|
|
|
|
|
|
# --- per_model tests ----------------------------------------------------------
|
|
|
|
|
|
def test_per_model_aggregates_correctly(tmp_path):
|
|
"""Sum/cost/energy/carbon/tokens match hand-computed values."""
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
rows = [
|
|
("cheap", 0.001, 5.0e-05, 2.4e-03, 100, 0.25),
|
|
("cheap", 0.002, 3.0e-05, 1.2e-03, 200, 0.50),
|
|
("dear", 0.010, 1.0e-04, 5.0e-03, 150, 0.75),
|
|
]
|
|
for (model_id, cost, kwh, carbon, tokens, attr) in rows:
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, cost_usd, energy_kwh, carbon_g_co2eq, "
|
|
"completion_tokens, attribution_ratio, observed_at) "
|
|
"VALUES (?, 'neuralwatt', ?, ?, ?, ?, ?, ?)",
|
|
(model_id, cost, kwh, carbon, tokens, attr, now.isoformat()),
|
|
)
|
|
conn.commit()
|
|
|
|
results = per_model(conn)
|
|
by_model = {r["model_id"]: r for r in results}
|
|
|
|
cheap = by_model["cheap"]
|
|
assert cheap["calls"] == 2
|
|
assert cheap["sum_cost_usd"] == pytest.approx(0.003)
|
|
assert cheap["sum_energy_kwh"] == pytest.approx(8.0e-05)
|
|
assert cheap["sum_carbon_g_co2eq"] == pytest.approx(3.6e-03)
|
|
assert cheap["avg_completion_tokens"] == pytest.approx(150.0)
|
|
assert cheap["avg_attribution_ratio"] == pytest.approx(0.375)
|
|
|
|
dear = by_model["dear"]
|
|
assert dear["calls"] == 1
|
|
assert dear["sum_cost_usd"] == pytest.approx(0.010)
|
|
|
|
|
|
def test_per_model_empty_db(tmp_path):
|
|
"""No energy rows: returns empty list."""
|
|
conn = _make_db(tmp_path)
|
|
assert per_model(conn) == []
|
|
|
|
|
|
# --- verdict_mix tests --------------------------------------------------------
|
|
|
|
|
|
def test_verdict_mix_counts_by_verdict(tmp_path):
|
|
"""Counts match the inserted rows."""
|
|
conn = _make_db(tmp_path)
|
|
now = _now()
|
|
conn.execute(
|
|
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 'structural', 'ok', ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
conn.execute(
|
|
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 'structural', 'ok', ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
conn.execute(
|
|
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 'local_llm', 'malformed', ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
conn.execute(
|
|
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 'structural', 'unverifiable', ?)",
|
|
(now.isoformat(),),
|
|
)
|
|
# Stale row outside the window
|
|
conn.execute(
|
|
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
|
|
"VALUES ('m', 'neuralwatt', 'structural', 'truncated', ?)",
|
|
((now - timedelta(days=30)).isoformat(),),
|
|
)
|
|
conn.commit()
|
|
|
|
result = verdict_mix(conn, since_days=7)
|
|
assert result["ok"] == 2
|
|
assert result["malformed"] == 1
|
|
assert result["unverifiable"] == 1
|
|
# truncated is > 7 days ago, so excluded
|
|
assert "truncated" not in result or result["truncated"] == 0
|
|
|
|
|
|
def test_verdict_mix_empty_db(tmp_path):
|
|
"""No rows: returns empty dict."""
|
|
conn = _make_db(tmp_path)
|
|
assert verdict_mix(conn) == {}
|
|
|
|
|
|
# --- top_proficiency tests ----------------------------------------------------
|
|
|
|
|
|
def test_top_proficiency_ordered_correctly(tmp_path):
|
|
"""Models are returned ordered by blended_score DESC."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
_seed_proficiency(conn)
|
|
conn.commit()
|
|
|
|
results = top_proficiency(conn, "coding_general")
|
|
assert len(results) == 3
|
|
assert results[0]["model_id"] == "dear" # 0.95
|
|
assert results[1]["model_id"] == "cheap" # 0.90
|
|
assert results[2]["model_id"] == "tiny" # 0.70
|
|
|
|
|
|
def test_top_proficiency_filter_by_category(tmp_path):
|
|
"""Requests for one category exclude models that have scores only for another."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
# Only code categories
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO proficiency (
|
|
model_id, provider, category, blended_score, source, last_updated
|
|
) VALUES ('cheap', 'neuralwatt', 'coding_general', 0.90, 'self_eval_thin', '2026-01-01T00:00:00+00:00')
|
|
""",
|
|
)
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO proficiency (
|
|
model_id, provider, category, blended_score, source, last_updated
|
|
) VALUES ('cheap', 'neuralwatt', 'docs_writing', 0.85, 'self_eval_thin', '2026-01-01T00:00:00+00:00')
|
|
""",
|
|
)
|
|
conn.commit()
|
|
|
|
coding = top_proficiency(conn, "coding_general")
|
|
docs = top_proficiency(conn, "docs_writing")
|
|
assert len(coding) == 1
|
|
assert coding[0]["model_id"] == "cheap"
|
|
assert len(docs) == 1
|
|
assert docs[0]["model_id"] == "cheap"
|
|
|
|
|
|
def test_top_proficiency_empty_for_missing_category(tmp_path):
|
|
"""No proficiency rows for category → empty list."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
assert top_proficiency(conn, "nonexistent_category") == []
|
|
|
|
|
|
def test_top_proficiency_passes_all_sources_through_to_json(tmp_path):
|
|
"""Every proficiency source value survives the row to dict to JSON path.
|
|
|
|
The empirical-Bayes outcome conversion introduced two new sources
|
|
(outcome_prior, outcome_blended) alongside the four legacy ones.
|
|
top_proficiency does not filter source strings, so this is a guard
|
|
against any future JSON schema or typed enum accidentally discarding one.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
sources = [
|
|
"self_eval",
|
|
"self_eval_thin",
|
|
"leaderboard",
|
|
"blended",
|
|
"outcome_prior",
|
|
"outcome_blended",
|
|
]
|
|
for i, source in enumerate(sources):
|
|
model_id = f"m-{source}"
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (model_id, provider, base_model_id, availability, last_updated)
|
|
VALUES (?, 'neuralwatt', ?, 'active', '2026-01-01T00:00:00+00:00')
|
|
""",
|
|
(model_id, model_id),
|
|
)
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO proficiency (
|
|
model_id, provider, category, blended_score, source, last_updated
|
|
) VALUES (?, 'neuralwatt', 'coding_general', ?, ?, '2026-01-01T00:00:00+00:00')
|
|
""",
|
|
(model_id, 0.8 - i * 0.01, source),
|
|
)
|
|
conn.commit()
|
|
|
|
results = top_proficiency(conn, "coding_general")
|
|
returned = {r["source"] for r in results}
|
|
assert returned == set(sources), f"missing sources: {set(sources) - returned}"
|
|
for r in results:
|
|
assert "model_id" in r
|
|
assert "blended_score" in r
|
|
assert "source" in r
|
|
assert isinstance(r["source"], str)
|
|
|
|
|
|
# --- /health endpoint compatibility -------------------------------------------
|
|
|
|
|
|
def _seed_pinch_models(conn: sqlite3.Connection) -> None:
|
|
"""Insert two cheap/dear models plus a cheap-no-cached-price variant."""
|
|
fresh = _now().isoformat()
|
|
rows = [
|
|
# (model_id, tier, context, cost_per_1m_prompt, cost_per_1m_prompt_cached)
|
|
("cheap", 2, 262128, 1.0, 0.5),
|
|
("dear", 2, 262128, 5.0, None),
|
|
("tiny", 1, 131072, 0.1, 0.1),
|
|
]
|
|
for model_id, tier, context, cost, cached in rows:
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
cost_per_1m_prompt_cached,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES (?, 'neuralwatt', ?, ?, ?, 192500, 16384, ?, ?, ?,
|
|
1, 1, 'standard', 'default', 'full', 'public', 'active',
|
|
?)
|
|
""",
|
|
(model_id, model_id, tier, context, cost, cost / 3, cached, fresh),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _insert_pinch_decision(conn, *, model, provider, orig, final):
|
|
"""Insert a route_decisions row with observed_at = now and given pinch values."""
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO route_decisions (
|
|
observed_at, kind, task_category, task_tier,
|
|
required_context_tokens, confidence, classifier_ms,
|
|
classification_source, latency_tolerance, candidates_considered,
|
|
selected_model, selected_provider, runner_up_models,
|
|
est_cost_usd, est_proficiency, rejected_reason, session_key,
|
|
tools, images, json_mode, streamed,
|
|
pinch_original_tokens, pinch_final_tokens
|
|
) VALUES (?, 'chat', 'coding_general', 2, 100, 0.9, 100,
|
|
'classifier', 'interactive', 3, ?, ?, NULL,
|
|
0.001, 0.9, NULL, 'sess', 0, 0, 0, 0, ?, ?)
|
|
""",
|
|
(_now().isoformat(), model, provider, orig, final),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def test_pinch_summary_aggregates_correctly(tmp_path, monkeypatch):
|
|
"""Hand-computed aggregates over a seeded 30-day window."""
|
|
cfg = load_config(str(ROOT / "config" / "config.yaml"))
|
|
monkeypatch.setattr(cfg.pinch, "enabled", True)
|
|
cache_rate = cfg.objective.assumed_cache_rate
|
|
|
|
conn = _make_db(tmp_path)
|
|
_seed_pinch_models(conn)
|
|
|
|
# Two pruned cheap rows: saved tokens 50 and 30.
|
|
_insert_pinch_decision(conn, model="cheap", provider="neuralwatt", orig=100, final=50)
|
|
_insert_pinch_decision(conn, model="cheap", provider="neuralwatt", orig=80, final=50)
|
|
# One non-pruned cheap row (final == orig).
|
|
_insert_pinch_decision(conn, model="cheap", provider="neuralwatt", orig=60, final=60)
|
|
# One pruned dear row with NULL cost_per_1m_prompt_cached -> falls back to prompt price.
|
|
_insert_pinch_decision(conn, model="dear", provider="neuralwatt", orig=200, final=100)
|
|
# One rejection row with NULL selected_model -> counted in share/tokens but excluded from dollars.
|
|
_insert_pinch_decision(conn, model=None, provider=None, orig=1000, final=500)
|
|
|
|
result = pinch_summary(conn, cfg)
|
|
assert result is not None
|
|
assert result["calls_30d"] == 5
|
|
assert result["pruned_calls_30d"] == 4
|
|
assert result["share_pruned"] == round(4 / 5, 4)
|
|
# saved tokens: 50 + 30 + 100 + 500 = 680
|
|
assert result["total_tokens_saved"] == 680
|
|
# median of [30, 50, 100, 500] = 75
|
|
assert result["median_tokens_saved"] == 75
|
|
|
|
# dollar math: cheap saved 80 tokens at blended rate over cached prompt.
|
|
cheap_blended = (1 - cache_rate) * 1.0 + cache_rate * 0.5
|
|
cheap_dollars = 80 * cheap_blended / 1_000_000
|
|
# dear saved 100 tokens; cached price NULL -> fallback to prompt price 5.0.
|
|
dear_dollars = 100 * 5.0 / 1_000_000
|
|
# rejection row with NULL selected_model is excluded from dollar math.
|
|
assert result["dollars_saved_usd_30d"] == round(cheap_dollars + dear_dollars, 6)
|
|
|
|
|
|
def test_pinch_summary_zero_rows(tmp_path, monkeypatch):
|
|
"""Enabled with no decisions in the window returns zeros and None median."""
|
|
cfg = load_config(str(ROOT / "config" / "config.yaml"))
|
|
monkeypatch.setattr(cfg.pinch, "enabled", True)
|
|
conn = _make_db(tmp_path)
|
|
result = pinch_summary(conn, cfg)
|
|
assert result is not None
|
|
assert result["calls_30d"] == 0
|
|
assert result["pruned_calls_30d"] == 0
|
|
assert result["share_pruned"] == 0.0
|
|
assert result["total_tokens_saved"] == 0
|
|
assert result["median_tokens_saved"] is None
|
|
assert result["dollars_saved_usd_30d"] == 0.0
|
|
|
|
|
|
def test_pinch_summary_disabled_returns_none(tmp_path, monkeypatch):
|
|
"""When cfg.pinch.enabled is False, None is returned."""
|
|
conn = _make_db(tmp_path)
|
|
cfg = SimpleNamespace(pinch=SimpleNamespace(enabled=False))
|
|
assert pinch_summary(conn, cfg) is None
|
|
|
|
|
|
# --- /health endpoint compatibility -------------------------------------------
|
|
|
|
|
|
def test_health_endpoint_returns_scoring_key(tmp_path, monkeypatch):
|
|
"""/health still returns the same SHAPE after moving functions to metrics."""
|
|
db_path = tmp_path / "test.db"
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
monkeypatch.setattr(dispatcher.cfg.database, "path", str(db_path))
|
|
# Disable local verification to avoid Ollama dependency
|
|
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
|
|
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
|
|
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
|
|
|
|
with TestClient(dispatcher.app) as client:
|
|
resp = client.get("/health")
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
assert "scoring" in data
|
|
assert "routable_models" in data["scoring"]
|
|
assert "with_energy_data" in data["scoring"]
|
|
assert "with_proficiency_data" in data["scoring"]
|
|
assert "quota" in data["scoring"]
|
|
assert "warnings" in data["scoring"]
|
|
|
|
|
|
# --- unroutable_models / selection_coverage -----------------------------------
|
|
#
|
|
# The gap these close: an ineligible candidate produces NO signal. A ceiling is
|
|
# a max() and a broken row contributes 0 to it; a rejection warning counts
|
|
# decisions with no selected model, and nothing is rejected while the other
|
|
# rows serve the traffic. Live on 2026-09-08, 12 of 30 active OpenRouter rows
|
|
# had a zero effective context window and had never been selected once across
|
|
# 23,000+ decisions.
|
|
|
|
|
|
def _seed_zero_context(conn: sqlite3.Connection, model_id: str = "broken") -> None:
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES (?, 'neuralwatt', ?, 1, 1048576, 0, 16384, 0.1, 0.2,
|
|
1, 1, 'standard', 'default', 'full', 'public', 'active', ?)
|
|
""",
|
|
(model_id, model_id, _now().isoformat()),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def test_a_zero_context_row_is_reported_unroutable(tmp_path):
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
_seed_zero_context(conn)
|
|
|
|
found = unroutable_models(conn, CFG)
|
|
|
|
assert [d["model_id"] for d in found] == ["broken"]
|
|
assert found[0]["reason"].startswith("context(")
|
|
|
|
|
|
def test_the_probe_uses_one_token_not_zero(tmp_path):
|
|
"""context_ceilings probes with zero, which is exactly why it could not
|
|
see this: a zero-token window passes `0 < 0` and is counted a candidate,
|
|
then contributes 0 to a max()."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
_seed_zero_context(conn)
|
|
|
|
# The broken row IS in the candidate set the ceiling is taken over...
|
|
ceilings = context_ceilings(conn, CFG)
|
|
assert any(b["count"] for b in ceilings.values())
|
|
# ...and yet contributes nothing, so no ceiling moves and nothing warns.
|
|
assert max(b["ceiling"] for b in ceilings.values()) == 192500
|
|
|
|
assert unroutable_models(conn, CFG)
|
|
|
|
|
|
def test_a_healthy_catalog_reports_nothing(tmp_path):
|
|
"""The check must be quiet on a normal catalog or it is worthless."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
|
|
assert unroutable_models(conn, CFG) == []
|
|
assert selection_coverage_warnings(selection_coverage(conn, CFG)) == []
|
|
|
|
|
|
def test_an_admin_deprecation_is_not_reported_as_a_defect(tmp_path):
|
|
"""Switching a model off is a choice. Reporting it would make this warn
|
|
about the exact thing it exists to distinguish from."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
_seed_zero_context(conn)
|
|
conn.executescript(_ADMIN_TABLE_SQL)
|
|
conn.execute(
|
|
"INSERT INTO admin_model_overrides "
|
|
"(model_id, provider, availability, updated_at) "
|
|
"VALUES ('broken', 'neuralwatt', 'deprecated', ?)",
|
|
(_now().isoformat(),),
|
|
)
|
|
conn.commit()
|
|
|
|
assert unroutable_models(conn, CFG) == []
|
|
|
|
|
|
def test_a_restricted_category_row_is_not_called_unroutable(tmp_path):
|
|
"""eligible_categories is a restrict-only gate -- a narrowing, not a
|
|
defect. The probe asks with a category the row itself admits."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
conn.execute(
|
|
"UPDATE models SET eligible_categories = 'file_summarization,"
|
|
"diff_checking' WHERE model_id = 'tiny'"
|
|
)
|
|
conn.commit()
|
|
|
|
assert unroutable_models(conn, CFG) == []
|
|
|
|
|
|
def test_one_warning_per_gate_not_per_model(tmp_path):
|
|
"""The live incident produced 11 rows failing the identical gate. Eleven
|
|
near-identical warnings is a wall nobody reads, and they share one root
|
|
cause and one fix."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
for i in range(11):
|
|
_seed_zero_context(conn, f"broken-{i}")
|
|
|
|
warnings = selection_coverage_warnings(selection_coverage(conn, CFG))
|
|
|
|
assert len(warnings) == 1
|
|
assert "11 active models can never be selected" in warnings[0]
|
|
# Named enough to recognize, summarized past that.
|
|
assert "+7 more" in warnings[0]
|
|
|
|
|
|
def test_never_selected_is_data_and_never_a_warning(tmp_path):
|
|
"""22 of 38 routable models were unselected on the live DB. Enumerating
|
|
those as a warning would fire permanently and mean 'you have more models
|
|
than winners', which is normal."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
|
|
coverage = selection_coverage(conn, CFG)
|
|
|
|
assert len(coverage["never_selected"]) == 3
|
|
assert coverage["selected_models"] == 0
|
|
assert selection_coverage_warnings(coverage) == []
|
|
|
|
|
|
def test_selection_coverage_counts_only_inside_the_window(tmp_path):
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
for model_id, age_hours in (("cheap", 1), ("dear", 24 * 30)):
|
|
conn.execute(
|
|
"INSERT INTO route_decisions (observed_at, kind, selected_model,"
|
|
" selected_provider) VALUES (?, 'chat', ?, 'neuralwatt')",
|
|
((_now() - timedelta(hours=age_hours)).isoformat(), model_id),
|
|
)
|
|
conn.commit()
|
|
|
|
coverage = selection_coverage(conn, CFG)
|
|
|
|
# 'dear' was picked, but a month ago -- outside the 168h window.
|
|
assert coverage["selected_models"] == 1
|
|
assert coverage["by_provider"]["neuralwatt"] == {"routable": 3, "selected": 1}
|
|
assert {r["model_id"] for r in coverage["never_selected"]} == {"dear", "tiny"}
|
|
|
|
|
|
def test_an_access_gated_row_is_not_counted_routable(tmp_path):
|
|
"""A canary or preview row can never be selected under this config BY
|
|
DESIGN, so counting it routable-but-unselected pads never_selected and
|
|
makes the two routable_models figures in one /metrics payload disagree.
|
|
|
|
Live after the detector shipped: selection said 51 while scoring_coverage
|
|
said 46 -- exactly the catalog's 1 canary + 4 preview rows.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
conn.execute(
|
|
"UPDATE models SET access_level = 'canary' WHERE model_id = 'tiny'"
|
|
)
|
|
conn.commit()
|
|
|
|
coverage = selection_coverage(conn, CFG)
|
|
|
|
assert coverage["routable_models"] == 2
|
|
assert "tiny" not in {r["model_id"] for r in coverage["never_selected"]}
|
|
# The two figures in one payload must agree.
|
|
assert (
|
|
coverage["routable_models"]
|
|
== scoring_coverage(conn, CFG)["routable_models"]
|
|
)
|
|
|
|
|
|
# =============================================================================
|
|
# cache_rate_series / cache_rate_warnings (Wave 1 item 1.4)
|
|
# =============================================================================
|
|
|
|
|
|
def _seed_cache_rows(
|
|
conn: sqlite3.Connection,
|
|
*,
|
|
model_id: str = "cheap",
|
|
provider: str = "neuralwatt",
|
|
n: int,
|
|
prompt_tokens: int,
|
|
cached: object,
|
|
source: str = "reported",
|
|
task_category: str = "coding_general",
|
|
age_hours: float = 1.0,
|
|
) -> None:
|
|
"""Insert *n* energy_observations rows shaped for the cache-rate query."""
|
|
at = (_now() - timedelta(hours=age_hours)).isoformat()
|
|
for _ in range(n):
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO energy_observations (
|
|
model_id, provider, task_category, prompt_tokens,
|
|
completion_tokens, energy_kwh, cached_prompt_tokens,
|
|
cached_tokens_source, observed_at
|
|
) VALUES (?, ?, ?, ?, 100, 0.0, ?, ?, ?)
|
|
""",
|
|
(model_id, provider, task_category, prompt_tokens, cached, source, at),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _cfg_with(**objective_overrides):
|
|
cfg = CFG.model_copy(deep=True)
|
|
for key, value in objective_overrides.items():
|
|
setattr(cfg.objective, key, value)
|
|
return cfg
|
|
|
|
|
|
def test_cache_rate_series_is_token_weighted_not_a_mean_of_rates(tmp_path):
|
|
"""sum(cached)/sum(prompt), not the average of per-request rates.
|
|
|
|
The bill is denominated in tokens, so one 100k-token turn is not one
|
|
observation's worth of evidence against a 1k one. A mean of per-request
|
|
rates here reads 0.667; the token-weighted answer is 0.505.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=1, prompt_tokens=1000, cached=500)
|
|
_seed_cache_rows(conn, n=1, prompt_tokens=1000, cached=1000)
|
|
# One huge mostly-uncached turn, which a mean of rates would let vanish.
|
|
_seed_cache_rows(conn, n=1, prompt_tokens=100000, cached=50000)
|
|
|
|
series = cache_rate_series(conn, CFG)
|
|
|
|
assert series["prompt_tokens"] == 102000
|
|
assert series["cached_prompt_tokens"] == 51500
|
|
assert series["cache_rate"] == pytest.approx(51500 / 102000)
|
|
assert series["observations"] == 3
|
|
|
|
|
|
def test_cache_rate_series_counts_only_reported_rows(tmp_path):
|
|
"""A row saying nothing must not enter the denominator.
|
|
|
|
A 'details_no_count' or 'no_details' row carries a real prompt_tokens
|
|
figure, so including it would divide a genuine cached total by a
|
|
denominator full of prompts nobody measured -- understating the rate
|
|
exactly where coverage is thinnest.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=2, prompt_tokens=1000, cached=900)
|
|
_seed_cache_rows(
|
|
conn, n=5, prompt_tokens=1000, cached=None, source="details_no_count"
|
|
)
|
|
_seed_cache_rows(
|
|
conn, n=5, prompt_tokens=1000, cached=None, source="no_details"
|
|
)
|
|
|
|
series = cache_rate_series(conn, CFG)
|
|
|
|
assert series["observations"] == 2
|
|
assert series["prompt_tokens"] == 2000
|
|
assert series["cache_rate"] == pytest.approx(0.9)
|
|
|
|
|
|
def test_cache_rate_series_counts_an_explicit_zero(tmp_path):
|
|
"""A reported 0 IS a measured full miss and must drag the rate down.
|
|
|
|
This is what af18009 exists for: "reported, and the number was zero" is
|
|
evidence, while NULL is an absence. Dropping the zeros would leave every
|
|
rate conditioned on a hit having occurred.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=1, prompt_tokens=1000, cached=1000)
|
|
_seed_cache_rows(conn, n=1, prompt_tokens=1000, cached=0)
|
|
|
|
series = cache_rate_series(conn, CFG)
|
|
|
|
assert series["observations"] == 2
|
|
assert series["cache_rate"] == pytest.approx(0.5)
|
|
|
|
|
|
def test_cache_rate_series_excludes_seed_reference_sweeps(tmp_path):
|
|
"""seed_energy.py's sweep is not user traffic.
|
|
|
|
Those rows were the entire reason NeuralWatt's cached-token coverage read
|
|
12.8% while real dispatch traffic reads ~98%.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=2, prompt_tokens=1000, cached=900)
|
|
_seed_cache_rows(
|
|
conn, n=40, prompt_tokens=400, cached=0, task_category="seed_reference"
|
|
)
|
|
|
|
series = cache_rate_series(conn, CFG)
|
|
|
|
assert series["observations"] == 2
|
|
assert series["cache_rate"] == pytest.approx(0.9)
|
|
|
|
|
|
def test_cache_rate_series_is_bounded_by_the_trailing_window(tmp_path):
|
|
"""Rows older than the window are out, so the pre-capture era ages away by
|
|
itself rather than needing a hardcoded start date."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=2, prompt_tokens=1000, cached=900, age_hours=1)
|
|
_seed_cache_rows(conn, n=9, prompt_tokens=1000, cached=0, age_hours=200)
|
|
|
|
series = cache_rate_series(conn, CFG)
|
|
|
|
assert series["observations"] == 2
|
|
assert series["cache_rate"] == pytest.approx(0.9)
|
|
assert series["window_hours"] == 168
|
|
|
|
|
|
def test_cache_rate_series_groups_on_provider_and_model(tmp_path):
|
|
"""Two providers serving the same base model are two groups, not one."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, model_id="dv4", provider="neuralwatt", n=3,
|
|
prompt_tokens=1000, cached=900)
|
|
_seed_cache_rows(conn, model_id="dv4", provider="openrouter", n=3,
|
|
prompt_tokens=1000, cached=400)
|
|
|
|
series = cache_rate_series(conn, CFG)
|
|
|
|
by_key = {(r["provider"], r["model_id"]): r for r in series["by_model"]}
|
|
assert set(by_key) == {("neuralwatt", "dv4"), ("openrouter", "dv4")}
|
|
assert by_key[("neuralwatt", "dv4")]["cache_rate"] == pytest.approx(0.9)
|
|
assert by_key[("openrouter", "dv4")]["cache_rate"] == pytest.approx(0.4)
|
|
|
|
|
|
def test_cache_rate_series_survives_a_pre_migration_schema(tmp_path):
|
|
"""cached_tokens_source arrives by ALTER at dispatcher start-up, and
|
|
admin.py can open a database that has not been through it. A missing
|
|
column must degrade this series, not 500 the whole /metrics payload."""
|
|
conn = _make_db(tmp_path)
|
|
conn.execute("DROP TABLE energy_observations")
|
|
conn.execute(
|
|
"CREATE TABLE energy_observations ("
|
|
" id INTEGER PRIMARY KEY, model_id TEXT, provider TEXT, observed_at TEXT)"
|
|
)
|
|
conn.commit()
|
|
|
|
series = cache_rate_series(conn, CFG)
|
|
|
|
assert series["cache_rate"] is None
|
|
assert series["by_model"] == []
|
|
assert cache_rate_warnings(conn, CFG, series) == []
|
|
|
|
|
|
def test_cache_rate_warning_silent_below_the_observation_floor(tmp_path):
|
|
"""Never mere presence: a divergence over a handful of requests is one
|
|
session's luck, and a warning that flaps on low traffic goes unread."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=5, prompt_tokens=1000, cached=100)
|
|
|
|
assert cache_rate_warnings(conn, CFG) == []
|
|
|
|
|
|
def test_cache_rate_warning_silent_inside_the_margin(tmp_path):
|
|
"""0.900 against an assumed 0.917 is ordinary session-mix drift."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=900)
|
|
|
|
assert cache_rate_warnings(conn, CFG) == []
|
|
|
|
|
|
def test_cache_rate_warning_fires_when_the_premise_expires(tmp_path):
|
|
"""The aggregate class names the constant it is expiring, not a floor."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=400)
|
|
|
|
warnings = cache_rate_warnings(conn, CFG)
|
|
|
|
premise = [w for w in warnings if w.startswith("cache rate: measured")]
|
|
assert len(premise) == 1
|
|
assert "0.400" in premise[0]
|
|
assert "assumed_cache_rate" in premise[0]
|
|
assert "below" in premise[0]
|
|
|
|
|
|
def test_cache_rate_warning_fires_in_the_other_direction_too(tmp_path):
|
|
"""A rate ABOVE the assumption mis-prices just as surely -- it overstates
|
|
the cache and underprices every candidate. Shown against a lower assumed
|
|
rate because at the shipped 0.917 a 0.10 margin cannot be exceeded
|
|
upward: the measured rate is bounded by 1.0."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=900)
|
|
|
|
warnings = cache_rate_warnings(conn, _cfg_with(assumed_cache_rate=0.5))
|
|
|
|
premise = [w for w in warnings if w.startswith("cache rate: measured")]
|
|
assert len(premise) == 1
|
|
assert "above" in premise[0]
|
|
|
|
|
|
def test_cache_rate_outlier_fires_while_the_aggregate_stays_quiet(tmp_path):
|
|
"""The per-group half is the one that catches a single bad route.
|
|
|
|
The aggregate sits inside the margin; the openrouter group at 0.400 does
|
|
not. Only the outlier class fires, and it names the structured
|
|
(provider, model) pair rather than a formatted label.
|
|
"""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, model_id="hot", provider="neuralwatt", n=300,
|
|
prompt_tokens=1000, cached=920)
|
|
_seed_cache_rows(conn, model_id="cold", provider="openrouter", n=30,
|
|
prompt_tokens=1000, cached=400)
|
|
|
|
series = cache_rate_series(conn, CFG)
|
|
warnings = cache_rate_warnings(conn, CFG, series)
|
|
|
|
assert series["cache_rate"] == pytest.approx(
|
|
(300 * 920 + 30 * 400) / (330 * 1000)
|
|
)
|
|
assert not [w for w in warnings if w.startswith("cache rate: measured")]
|
|
outliers = [w for w in warnings if w.startswith("cache rate outlier:")]
|
|
assert len(outliers) == 1
|
|
assert "cold" in outliers[0] and "openrouter" in outliers[0]
|
|
|
|
|
|
def test_cache_rate_outlier_respects_the_per_group_observation_floor(tmp_path):
|
|
"""A four-request group that looks catastrophic is four requests."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, model_id="hot", provider="neuralwatt", n=300,
|
|
prompt_tokens=1000, cached=920)
|
|
_seed_cache_rows(conn, model_id="cold", provider="openrouter", n=4,
|
|
prompt_tokens=1000, cached=100)
|
|
|
|
warnings = cache_rate_warnings(conn, CFG)
|
|
|
|
assert not [w for w in warnings if w.startswith("cache rate outlier:")]
|
|
|
|
|
|
def test_cache_rate_warning_silent_without_an_assumption_to_expire(tmp_path):
|
|
"""Nothing to diverge from means nothing to say."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=100)
|
|
|
|
assert cache_rate_warnings(conn, _cfg_with(assumed_cache_rate=None)) == []
|
|
|
|
|
|
def test_scoring_coverage_carries_the_cache_series_and_its_warnings(tmp_path):
|
|
"""The series and the warning are one reading, not two queries that can
|
|
disagree about the number on screen."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
_seed_cache_rows(conn, n=30, prompt_tokens=1000, cached=400)
|
|
|
|
coverage = scoring_coverage(conn, CFG)
|
|
|
|
assert coverage["cache"]["cache_rate"] == pytest.approx(0.4)
|
|
assert any(w.startswith("cache rate: measured") for w in coverage["warnings"])
|