"""Two report-only /metrics series: cost-estimator calibration, and latency. Both are MEASUREMENT SURFACES. Neither is read by routing, and the tests that matter most here are the ones pinning that: a live experiment (``objective.incumbent_cache_pricing``) is running while these landed, and a series that quietly reached the ranker would confound it. Every test seeds a throwaway SQLite DB from ``config/schema.sql`` and asserts on queried aggregates, never on mock calls — the convention ``test_metrics.py`` established here. """ from __future__ import annotations import inspect import sqlite3 from datetime import datetime, timedelta, timezone from pathlib import Path from types import SimpleNamespace import pytest import metrics from metrics import cost_estimate_calibration, latency_series ROOT = Path(__file__).resolve().parent.parent SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text() CFG = SimpleNamespace( objective=SimpleNamespace( cost_calibration_window_hours=168, cost_calibration_min_observations=3, latency_window_hours=168, latency_min_observations=3, ) ) def _cfg(**overrides) -> SimpleNamespace: base = dict( cost_calibration_window_hours=168, cost_calibration_min_observations=3, latency_window_hours=168, latency_min_observations=3, ) base.update(overrides) return SimpleNamespace(objective=SimpleNamespace(**base)) @pytest.fixture() def conn(tmp_path): c = sqlite3.connect(tmp_path / "m.db") c.executescript(SCHEMA_SQL) c.row_factory = sqlite3.Row yield c c.close() def _ago(hours: float) -> str: return (datetime.now(timezone.utc) - timedelta(hours=hours)).isoformat() def _pair( conn, request_id, model_id, provider, est, billed, *, age_hours=1.0, task_category="coding_general", eo_model_id=None, eo_provider=None, ): """One decision plus the energy row its completion produced.""" at = _ago(age_hours) conn.execute( "INSERT INTO route_decisions " "(kind, task_category, selected_model, selected_provider, " " est_cost_usd, request_id, observed_at) " "VALUES ('chat', ?, ?, ?, ?, ?, ?)", (task_category, model_id, provider, est, request_id, at), ) conn.execute( "INSERT INTO energy_observations " "(model_id, provider, request_id, task_category, cost_usd, observed_at) " "VALUES (?, ?, ?, ?, ?, ?)", ( eo_model_id or model_id, eo_provider or provider, request_id, task_category, billed, at, ), ) def _latency_row( conn, model_id, provider, wall, ttft, *, age_hours=1.0, task_category="coding_general", ): conn.execute( "INSERT INTO energy_observations " "(model_id, provider, task_category, router_wall_seconds, " " router_ttft_seconds, observed_at) " "VALUES (?, ?, ?, ?, ?, ?)", (model_id, provider, task_category, wall, ttft, _ago(age_hours)), ) # --------------------------------------------------------------------------- # Item 1 — cost-estimator calibration # --------------------------------------------------------------------------- def test_correction_factor_is_billed_over_estimate(conn): """The two ratios are reciprocals, and each points the direction it says.""" for i in range(4): _pair(conn, f"r{i}", "m", "neuralwatt", est=4.0, billed=1.0) conn.commit() series = cost_estimate_calibration(conn, _cfg()) assert series["observations"] == 4 assert series["est_cost_usd"] == pytest.approx(16.0) assert series["billed_cost_usd"] == pytest.approx(4.0) # est/billed: the estimator charged the ranker 4x what the account paid. assert series["overestimate_ratio"] == pytest.approx(4.0) # billed/est: what you would multiply INTO the estimate. Nothing does. assert series["correction_factor"] == pytest.approx(0.25) entry = series["by_model"][0] assert (entry["provider"], entry["model_id"]) == ("neuralwatt", "m") assert entry["est_per_request_usd"] == pytest.approx(4.0) assert entry["billed_per_request_usd"] == pytest.approx(1.0) assert entry["overestimate_ratio"] == pytest.approx(4.0) assert entry["correction_factor"] == pytest.approx(0.25) def test_spread_is_the_reported_headline_not_the_scale(conn): """A uniform scale error reorders nothing; the spread is what does. This is the whole reason the series exists rather than a single global factor, so it is pinned with the live shape: two models whose estimates rank one way and whose bills rank the other. """ # cheap-by-estimate, expensive-by-bill for i in range(4): _pair(conn, f"a{i}", "under", "openrouter", est=3.0, billed=0.66) # expensive-by-estimate, cheap-by-bill for i in range(4): _pair(conn, f"b{i}", "over", "neuralwatt", est=4.1, billed=0.51) conn.commit() series = cost_estimate_calibration(conn, _cfg()) by_key = {(e["provider"], e["model_id"]): e for e in series["by_model"]} under = by_key[("openrouter", "under")] over = by_key[("neuralwatt", "over")] # The estimator orders them one way... assert under["est_per_request_usd"] < over["est_per_request_usd"] # ...and the bill orders them the other. That inversion is the finding. assert under["billed_per_request_usd"] > over["billed_per_request_usd"] assert series["spread"] == pytest.approx( over["overestimate_ratio"] / under["overestimate_ratio"] ) assert series["spread"] > 1.5 assert series["spread_low"]["model_id"] == "under" assert series["spread_high"]["model_id"] == "over" def test_thin_groups_are_listed_but_excluded_from_spread(conn): """A ratio over one request is information, never the headline.""" for i in range(4): _pair(conn, f"a{i}", "thick", "neuralwatt", est=4.0, billed=1.0) _pair(conn, "solo", "thin", "neuralwatt", est=100.0, billed=1.0) conn.commit() series = cost_estimate_calibration(conn, _cfg()) by_key = {e["model_id"]: e for e in series["by_model"]} assert by_key["thin"]["sufficient"] is False assert by_key["thin"]["overestimate_ratio"] == pytest.approx(100.0) assert by_key["thick"]["sufficient"] is True # One qualifying group -> spread 1.0, and the 100x outlier does not set it. assert series["spread"] == pytest.approx(1.0) assert series["spread_high"]["model_id"] == "thick" def test_seed_reference_is_excluded_from_both_sides(conn): """Sweep traffic is a fixed 400/400 shape no real request resembles.""" for i in range(4): _pair(conn, f"real{i}", "m", "neuralwatt", est=4.0, billed=1.0) for i in range(20): _pair( conn, f"seed{i}", "m", "neuralwatt", est=1.0, billed=1.0, task_category="seed_reference", ) conn.commit() series = cost_estimate_calibration(conn, _cfg()) assert series["observations"] == 4 assert series["overestimate_ratio"] == pytest.approx(4.0) def test_a_failover_row_does_not_pair_one_models_estimate_with_anothers_bill(conn): """The request id alone is not enough to join on. Live on router.db: 5 of 4,143 joined rows carry a decision that selected `qwen3.6-35b` (neuralwatt) against a completion billed by `qwen/qwen3.6-35b-a3b` (openrouter) — a cross-provider failover. Joining on request_id alone would price one model's estimate against another model's bill, which is precisely the contamination this series cannot afford. """ for i in range(4): _pair(conn, f"ok{i}", "picked", "neuralwatt", est=4.0, billed=1.0) _pair( conn, "failover", "picked", "neuralwatt", est=4.0, billed=99.0, eo_model_id="answered", eo_provider="openrouter", ) conn.commit() series = cost_estimate_calibration(conn, _cfg()) assert series["observations"] == 4 assert {e["model_id"] for e in series["by_model"]} == {"picked"} assert series["billed_cost_usd"] == pytest.approx(4.0) def test_unbilled_and_unestimated_rows_are_dropped(conn): """A zero or NULL bill is a row the provider did not price, not a free one.""" for i in range(4): _pair(conn, f"ok{i}", "m", "neuralwatt", est=4.0, billed=1.0) _pair(conn, "zero", "m", "neuralwatt", est=4.0, billed=0.0) _pair(conn, "null-bill", "m", "neuralwatt", est=4.0, billed=None) _pair(conn, "null-est", "m", "neuralwatt", est=None, billed=1.0) conn.commit() series = cost_estimate_calibration(conn, _cfg()) assert series["observations"] == 4 def test_rows_outside_the_window_are_dropped(conn): for i in range(4): _pair(conn, f"new{i}", "m", "neuralwatt", est=4.0, billed=1.0) for i in range(10): _pair( conn, f"old{i}", "m", "neuralwatt", est=1.0, billed=1.0, age_hours=200 ) conn.commit() series = cost_estimate_calibration(conn, _cfg(cost_calibration_window_hours=168)) assert series["observations"] == 4 def test_empty_database_returns_a_shaped_series(conn): """No data is not an error, and every key a consumer reads still exists.""" series = cost_estimate_calibration(conn, _cfg()) assert series["observations"] == 0 assert series["by_model"] == [] for key in ("overestimate_ratio", "correction_factor", "spread"): assert series[key] is None def test_config_defaults_are_used_when_the_knobs_are_absent(conn): """An older config file, or a SimpleNamespace in a test, must not crash.""" bare = SimpleNamespace(objective=SimpleNamespace()) series = cost_estimate_calibration(conn, bare) assert series["window_hours"] == metrics._COST_CALIBRATION_WINDOW_HOURS assert series["min_observations"] == metrics._COST_CALIBRATION_MIN_OBSERVATIONS # --------------------------------------------------------------------------- # Item 2 — latency # --------------------------------------------------------------------------- def test_percentiles_per_model(conn): for wall, ttft in [(1.0, 0.5), (2.0, 1.0), (3.0, 1.5), (100.0, 50.0)]: _latency_row(conn, "slow", "openrouter", wall, ttft) conn.commit() series = latency_series(conn, _cfg()) entry = series["by_model"][0] assert entry["wall_observations"] == 4 assert entry["ttft_observations"] == 4 assert entry["wall_p50"] == pytest.approx(2.5) assert entry["wall_p95"] == pytest.approx(85.45) assert entry["ttft_p50"] == pytest.approx(1.25) assert entry["wall_sufficient"] is True assert entry["ttft_sufficient"] is True # The aggregate is over the same rows when there is only one model. assert series["wall_p50"] == pytest.approx(2.5) def test_wall_and_ttft_counts_are_independent(conn): """TTFT is streaming-only, so a group can be thick on wall and thin on TTFT. Reporting one count for both would let a handful of streamed requests borrow the credibility of a large buffered sample. """ for i in range(6): _latency_row(conn, "m", "neuralwatt", 2.0 + i, None) _latency_row(conn, "m", "neuralwatt", 2.0, 1.0) conn.commit() series = latency_series(conn, _cfg(latency_min_observations=3)) entry = series["by_model"][0] assert entry["wall_observations"] == 7 assert entry["ttft_observations"] == 1 assert entry["wall_sufficient"] is True assert entry["ttft_sufficient"] is False assert entry["ttft_p50"] == pytest.approx(1.0) assert series["wall_observations"] == 7 assert series["ttft_observations"] == 1 def test_a_group_with_no_ttft_at_all_reports_none_not_zero(conn): """No measurement is not a measurement of zero.""" for i in range(4): _latency_row(conn, "buffered", "neuralwatt", 3.0, None) conn.commit() entry = latency_series(conn, _cfg())["by_model"][0] assert entry["ttft_observations"] == 0 assert entry["ttft_p50"] is None assert entry["ttft_p95"] is None assert entry["wall_p50"] == pytest.approx(3.0) def test_seed_reference_is_excluded_from_latency(conn): """Redundant today (the sweep writes no wall clock) and kept anyway.""" _latency_row(conn, "m", "neuralwatt", 2.0, 1.0) for i in range(20): _latency_row( conn, "m", "neuralwatt", 999.0, 999.0, task_category="seed_reference" ) conn.commit() series = latency_series(conn, _cfg()) assert series["wall_observations"] == 1 assert series["wall_p50"] == pytest.approx(2.0) def test_rows_outside_the_latency_window_are_dropped(conn): _latency_row(conn, "m", "neuralwatt", 2.0, 1.0) _latency_row(conn, "m", "neuralwatt", 999.0, 999.0, age_hours=200) conn.commit() series = latency_series(conn, _cfg(latency_window_hours=168)) assert series["wall_observations"] == 1 def test_latency_degrades_rather_than_raising_without_the_columns(tmp_path): """Both columns arrive by ALTER, and `admin` may open a pre-migration DB. A missing column must degrade this series, never 500 the whole /metrics payload — the contract `cache_rate_series` already keeps. """ schema = SCHEMA_SQL.replace(" router_wall_seconds REAL,\n", "").replace( " router_ttft_seconds REAL,\n", "" ) c = sqlite3.connect(tmp_path / "old.db") c.executescript(schema) c.row_factory = sqlite3.Row try: # Assert on the DB, not on the SQL text: both names also appear in # schema.sql's prose comments, so a substring check on the source # would pass while the columns were still there. cols = {r[1] for r in c.execute("PRAGMA table_info(energy_observations)")} assert "router_wall_seconds" not in cols assert "router_ttft_seconds" not in cols series = latency_series(c, _cfg()) finally: c.close() assert series["wall_observations"] == 0 assert series["wall_p50"] is None assert series["by_model"] == [] def test_latency_config_defaults_when_the_knobs_are_absent(conn): bare = SimpleNamespace(objective=SimpleNamespace()) series = latency_series(conn, bare) assert series["window_hours"] == metrics._LATENCY_WINDOW_HOURS assert series["min_observations"] == metrics._LATENCY_MIN_OBSERVATIONS # --------------------------------------------------------------------------- # The constraint both series are under # --------------------------------------------------------------------------- def test_neither_series_is_read_by_the_routing_path(): """Measurement only. A live experiment is running; a series that reached the ranker would confound it. Asserted against the module source rather than trusted to review: the two names must not appear in routing.py or on dispatcher's dispatch path at all. ``dispatcher`` calls them exactly once each, from the /metrics endpoint. """ import dispatcher import routing routing_src = inspect.getsource(routing) for name in ("cost_estimate_calibration", "latency_series"): assert name not in routing_src, ( f"{name} reached routing.py. Both series are report-only; " "applying one changes which model gets picked." ) dispatcher_src = inspect.getsource(dispatcher) for name in ("cost_estimate_calibration", "latency_series"): # Once in the import block, once in the /metrics payload. assert dispatcher_src.count(name) == 2, ( f"{name} is referenced {dispatcher_src.count(name)} times in " "dispatcher.py; it should appear only in the import and in the " "/metrics payload." ) def test_metrics_still_imports_without_dispatcher(): """metrics.py must never import dispatcher — the circular-import break.""" src = inspect.getsource(metrics) assert "import dispatcher" not in src