Two measurement surfaces on /metrics. Neither is read by routing, and a test asserts that against the module source: a live incumbent-cache-pricing experiment is running, and a series that reached the ranker would confound it. cost_calibration -- routing.estimated_cost against the provider's own bill, per (provider, model), joined on request_id. The scale error is not the finding: a uniform overestimate reorders nothing, because the ranking is a comparison and every candidate moves together. The SPREAD does reorder, so that is the headline figure. Live over 168h, 4,152 joined requests: est/billed runs 1.62x (qwen/qwen3.6-35b-a3b, openrouter) to 12.86x (qwen3.6-35b-fast, neuralwatt) -- a 7.9x spread, wider than the 5x the brief was written against. The factors are REPORTED, not applied; whether they are stable enough to trust is the question this exists to answer, and this project has already mistaken one moment of a moving per-model quantity for a constant. The join needs the model as well as the request id. On the live database 5 of 4,143 rows pair a decision that selected qwen3.6-35b (neuralwatt) with a completion billed by qwen/qwen3.6-35b-a3b (openrouter) -- a cross-provider failover, and one model's estimate against another's bill. latency -- p50/p95 of router_wall_seconds and router_ttft_seconds, which landed recently and nothing read. These are the router's own clock, not the provider's duration_seconds, which is why OpenRouter is visible here at all: it reports no duration. z-ai/glm-5.3-flash, a model with "flash" in its name, measures p50 9.60s / p95 40.08s to first token against 1.58s / 3.09s for deepseek-v4-flash on NeuralWatt. Reported, never scored. The wall and TTFT sample counts are kept independent because TTFT is streaming-only by nature. No new warning class, deliberately. Every group is out of band on both series today, so a divergence warning would fire on all of them from the first run -- bare presence, which is the anti-pattern rejection_warnings exists to avoid -- and a warning is a form of trust these factors have not yet earned. Four knobs under objective, all report-only, all recorded in DELIBERATELY_NOT_IN_ADMIN with the reason: they change what /metrics shows, and not even what it warns about. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
455 lines
16 KiB
Python
455 lines
16 KiB
Python
"""Two report-only /metrics series: cost-estimator calibration, and latency.
|
|
|
|
Both are MEASUREMENT SURFACES. Neither is read by routing, and the tests that
|
|
matter most here are the ones pinning that: a live experiment
|
|
(``objective.incumbent_cache_pricing``) is running while these landed, and a
|
|
series that quietly reached the ranker would confound it.
|
|
|
|
Every test seeds a throwaway SQLite DB from ``config/schema.sql`` and asserts
|
|
on queried aggregates, never on mock calls — the convention ``test_metrics.py``
|
|
established here.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import inspect
|
|
import sqlite3
|
|
from datetime import datetime, timedelta, timezone
|
|
from pathlib import Path
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
|
|
import metrics
|
|
from metrics import cost_estimate_calibration, latency_series
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
|
|
|
CFG = SimpleNamespace(
|
|
objective=SimpleNamespace(
|
|
cost_calibration_window_hours=168,
|
|
cost_calibration_min_observations=3,
|
|
latency_window_hours=168,
|
|
latency_min_observations=3,
|
|
)
|
|
)
|
|
|
|
|
|
def _cfg(**overrides) -> SimpleNamespace:
|
|
base = dict(
|
|
cost_calibration_window_hours=168,
|
|
cost_calibration_min_observations=3,
|
|
latency_window_hours=168,
|
|
latency_min_observations=3,
|
|
)
|
|
base.update(overrides)
|
|
return SimpleNamespace(objective=SimpleNamespace(**base))
|
|
|
|
|
|
@pytest.fixture()
|
|
def conn(tmp_path):
|
|
c = sqlite3.connect(tmp_path / "m.db")
|
|
c.executescript(SCHEMA_SQL)
|
|
c.row_factory = sqlite3.Row
|
|
yield c
|
|
c.close()
|
|
|
|
|
|
def _ago(hours: float) -> str:
|
|
return (datetime.now(timezone.utc) - timedelta(hours=hours)).isoformat()
|
|
|
|
|
|
def _pair(
|
|
conn,
|
|
request_id,
|
|
model_id,
|
|
provider,
|
|
est,
|
|
billed,
|
|
*,
|
|
age_hours=1.0,
|
|
task_category="coding_general",
|
|
eo_model_id=None,
|
|
eo_provider=None,
|
|
):
|
|
"""One decision plus the energy row its completion produced."""
|
|
at = _ago(age_hours)
|
|
conn.execute(
|
|
"INSERT INTO route_decisions "
|
|
"(kind, task_category, selected_model, selected_provider, "
|
|
" est_cost_usd, request_id, observed_at) "
|
|
"VALUES ('chat', ?, ?, ?, ?, ?, ?)",
|
|
(task_category, model_id, provider, est, request_id, at),
|
|
)
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, request_id, task_category, cost_usd, observed_at) "
|
|
"VALUES (?, ?, ?, ?, ?, ?)",
|
|
(
|
|
eo_model_id or model_id,
|
|
eo_provider or provider,
|
|
request_id,
|
|
task_category,
|
|
billed,
|
|
at,
|
|
),
|
|
)
|
|
|
|
|
|
def _latency_row(
|
|
conn,
|
|
model_id,
|
|
provider,
|
|
wall,
|
|
ttft,
|
|
*,
|
|
age_hours=1.0,
|
|
task_category="coding_general",
|
|
):
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, task_category, router_wall_seconds, "
|
|
" router_ttft_seconds, observed_at) "
|
|
"VALUES (?, ?, ?, ?, ?, ?)",
|
|
(model_id, provider, task_category, wall, ttft, _ago(age_hours)),
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Item 1 — cost-estimator calibration
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_correction_factor_is_billed_over_estimate(conn):
|
|
"""The two ratios are reciprocals, and each points the direction it says."""
|
|
for i in range(4):
|
|
_pair(conn, f"r{i}", "m", "neuralwatt", est=4.0, billed=1.0)
|
|
conn.commit()
|
|
|
|
series = cost_estimate_calibration(conn, _cfg())
|
|
|
|
assert series["observations"] == 4
|
|
assert series["est_cost_usd"] == pytest.approx(16.0)
|
|
assert series["billed_cost_usd"] == pytest.approx(4.0)
|
|
# est/billed: the estimator charged the ranker 4x what the account paid.
|
|
assert series["overestimate_ratio"] == pytest.approx(4.0)
|
|
# billed/est: what you would multiply INTO the estimate. Nothing does.
|
|
assert series["correction_factor"] == pytest.approx(0.25)
|
|
|
|
entry = series["by_model"][0]
|
|
assert (entry["provider"], entry["model_id"]) == ("neuralwatt", "m")
|
|
assert entry["est_per_request_usd"] == pytest.approx(4.0)
|
|
assert entry["billed_per_request_usd"] == pytest.approx(1.0)
|
|
assert entry["overestimate_ratio"] == pytest.approx(4.0)
|
|
assert entry["correction_factor"] == pytest.approx(0.25)
|
|
|
|
|
|
def test_spread_is_the_reported_headline_not_the_scale(conn):
|
|
"""A uniform scale error reorders nothing; the spread is what does.
|
|
|
|
This is the whole reason the series exists rather than a single global
|
|
factor, so it is pinned with the live shape: two models whose estimates
|
|
rank one way and whose bills rank the other.
|
|
"""
|
|
# cheap-by-estimate, expensive-by-bill
|
|
for i in range(4):
|
|
_pair(conn, f"a{i}", "under", "openrouter", est=3.0, billed=0.66)
|
|
# expensive-by-estimate, cheap-by-bill
|
|
for i in range(4):
|
|
_pair(conn, f"b{i}", "over", "neuralwatt", est=4.1, billed=0.51)
|
|
conn.commit()
|
|
|
|
series = cost_estimate_calibration(conn, _cfg())
|
|
|
|
by_key = {(e["provider"], e["model_id"]): e for e in series["by_model"]}
|
|
under = by_key[("openrouter", "under")]
|
|
over = by_key[("neuralwatt", "over")]
|
|
|
|
# The estimator orders them one way...
|
|
assert under["est_per_request_usd"] < over["est_per_request_usd"]
|
|
# ...and the bill orders them the other. That inversion is the finding.
|
|
assert under["billed_per_request_usd"] > over["billed_per_request_usd"]
|
|
|
|
assert series["spread"] == pytest.approx(
|
|
over["overestimate_ratio"] / under["overestimate_ratio"]
|
|
)
|
|
assert series["spread"] > 1.5
|
|
assert series["spread_low"]["model_id"] == "under"
|
|
assert series["spread_high"]["model_id"] == "over"
|
|
|
|
|
|
def test_thin_groups_are_listed_but_excluded_from_spread(conn):
|
|
"""A ratio over one request is information, never the headline."""
|
|
for i in range(4):
|
|
_pair(conn, f"a{i}", "thick", "neuralwatt", est=4.0, billed=1.0)
|
|
_pair(conn, "solo", "thin", "neuralwatt", est=100.0, billed=1.0)
|
|
conn.commit()
|
|
|
|
series = cost_estimate_calibration(conn, _cfg())
|
|
|
|
by_key = {e["model_id"]: e for e in series["by_model"]}
|
|
assert by_key["thin"]["sufficient"] is False
|
|
assert by_key["thin"]["overestimate_ratio"] == pytest.approx(100.0)
|
|
assert by_key["thick"]["sufficient"] is True
|
|
|
|
# One qualifying group -> spread 1.0, and the 100x outlier does not set it.
|
|
assert series["spread"] == pytest.approx(1.0)
|
|
assert series["spread_high"]["model_id"] == "thick"
|
|
|
|
|
|
def test_seed_reference_is_excluded_from_both_sides(conn):
|
|
"""Sweep traffic is a fixed 400/400 shape no real request resembles."""
|
|
for i in range(4):
|
|
_pair(conn, f"real{i}", "m", "neuralwatt", est=4.0, billed=1.0)
|
|
for i in range(20):
|
|
_pair(
|
|
conn,
|
|
f"seed{i}",
|
|
"m",
|
|
"neuralwatt",
|
|
est=1.0,
|
|
billed=1.0,
|
|
task_category="seed_reference",
|
|
)
|
|
conn.commit()
|
|
|
|
series = cost_estimate_calibration(conn, _cfg())
|
|
|
|
assert series["observations"] == 4
|
|
assert series["overestimate_ratio"] == pytest.approx(4.0)
|
|
|
|
|
|
def test_a_failover_row_does_not_pair_one_models_estimate_with_anothers_bill(conn):
|
|
"""The request id alone is not enough to join on.
|
|
|
|
Live on router.db: 5 of 4,143 joined rows carry a decision that selected
|
|
`qwen3.6-35b` (neuralwatt) against a completion billed by
|
|
`qwen/qwen3.6-35b-a3b` (openrouter) — a cross-provider failover. Joining
|
|
on request_id alone would price one model's estimate against another
|
|
model's bill, which is precisely the contamination this series cannot
|
|
afford.
|
|
"""
|
|
for i in range(4):
|
|
_pair(conn, f"ok{i}", "picked", "neuralwatt", est=4.0, billed=1.0)
|
|
_pair(
|
|
conn,
|
|
"failover",
|
|
"picked",
|
|
"neuralwatt",
|
|
est=4.0,
|
|
billed=99.0,
|
|
eo_model_id="answered",
|
|
eo_provider="openrouter",
|
|
)
|
|
conn.commit()
|
|
|
|
series = cost_estimate_calibration(conn, _cfg())
|
|
|
|
assert series["observations"] == 4
|
|
assert {e["model_id"] for e in series["by_model"]} == {"picked"}
|
|
assert series["billed_cost_usd"] == pytest.approx(4.0)
|
|
|
|
|
|
def test_unbilled_and_unestimated_rows_are_dropped(conn):
|
|
"""A zero or NULL bill is a row the provider did not price, not a free one."""
|
|
for i in range(4):
|
|
_pair(conn, f"ok{i}", "m", "neuralwatt", est=4.0, billed=1.0)
|
|
_pair(conn, "zero", "m", "neuralwatt", est=4.0, billed=0.0)
|
|
_pair(conn, "null-bill", "m", "neuralwatt", est=4.0, billed=None)
|
|
_pair(conn, "null-est", "m", "neuralwatt", est=None, billed=1.0)
|
|
conn.commit()
|
|
|
|
series = cost_estimate_calibration(conn, _cfg())
|
|
assert series["observations"] == 4
|
|
|
|
|
|
def test_rows_outside_the_window_are_dropped(conn):
|
|
for i in range(4):
|
|
_pair(conn, f"new{i}", "m", "neuralwatt", est=4.0, billed=1.0)
|
|
for i in range(10):
|
|
_pair(
|
|
conn, f"old{i}", "m", "neuralwatt", est=1.0, billed=1.0, age_hours=200
|
|
)
|
|
conn.commit()
|
|
|
|
series = cost_estimate_calibration(conn, _cfg(cost_calibration_window_hours=168))
|
|
assert series["observations"] == 4
|
|
|
|
|
|
def test_empty_database_returns_a_shaped_series(conn):
|
|
"""No data is not an error, and every key a consumer reads still exists."""
|
|
series = cost_estimate_calibration(conn, _cfg())
|
|
assert series["observations"] == 0
|
|
assert series["by_model"] == []
|
|
for key in ("overestimate_ratio", "correction_factor", "spread"):
|
|
assert series[key] is None
|
|
|
|
|
|
def test_config_defaults_are_used_when_the_knobs_are_absent(conn):
|
|
"""An older config file, or a SimpleNamespace in a test, must not crash."""
|
|
bare = SimpleNamespace(objective=SimpleNamespace())
|
|
series = cost_estimate_calibration(conn, bare)
|
|
assert series["window_hours"] == metrics._COST_CALIBRATION_WINDOW_HOURS
|
|
assert series["min_observations"] == metrics._COST_CALIBRATION_MIN_OBSERVATIONS
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Item 2 — latency
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_percentiles_per_model(conn):
|
|
for wall, ttft in [(1.0, 0.5), (2.0, 1.0), (3.0, 1.5), (100.0, 50.0)]:
|
|
_latency_row(conn, "slow", "openrouter", wall, ttft)
|
|
conn.commit()
|
|
|
|
series = latency_series(conn, _cfg())
|
|
|
|
entry = series["by_model"][0]
|
|
assert entry["wall_observations"] == 4
|
|
assert entry["ttft_observations"] == 4
|
|
assert entry["wall_p50"] == pytest.approx(2.5)
|
|
assert entry["wall_p95"] == pytest.approx(85.45)
|
|
assert entry["ttft_p50"] == pytest.approx(1.25)
|
|
assert entry["wall_sufficient"] is True
|
|
assert entry["ttft_sufficient"] is True
|
|
# The aggregate is over the same rows when there is only one model.
|
|
assert series["wall_p50"] == pytest.approx(2.5)
|
|
|
|
|
|
def test_wall_and_ttft_counts_are_independent(conn):
|
|
"""TTFT is streaming-only, so a group can be thick on wall and thin on TTFT.
|
|
|
|
Reporting one count for both would let a handful of streamed requests
|
|
borrow the credibility of a large buffered sample.
|
|
"""
|
|
for i in range(6):
|
|
_latency_row(conn, "m", "neuralwatt", 2.0 + i, None)
|
|
_latency_row(conn, "m", "neuralwatt", 2.0, 1.0)
|
|
conn.commit()
|
|
|
|
series = latency_series(conn, _cfg(latency_min_observations=3))
|
|
entry = series["by_model"][0]
|
|
|
|
assert entry["wall_observations"] == 7
|
|
assert entry["ttft_observations"] == 1
|
|
assert entry["wall_sufficient"] is True
|
|
assert entry["ttft_sufficient"] is False
|
|
assert entry["ttft_p50"] == pytest.approx(1.0)
|
|
assert series["wall_observations"] == 7
|
|
assert series["ttft_observations"] == 1
|
|
|
|
|
|
def test_a_group_with_no_ttft_at_all_reports_none_not_zero(conn):
|
|
"""No measurement is not a measurement of zero."""
|
|
for i in range(4):
|
|
_latency_row(conn, "buffered", "neuralwatt", 3.0, None)
|
|
conn.commit()
|
|
|
|
entry = latency_series(conn, _cfg())["by_model"][0]
|
|
assert entry["ttft_observations"] == 0
|
|
assert entry["ttft_p50"] is None
|
|
assert entry["ttft_p95"] is None
|
|
assert entry["wall_p50"] == pytest.approx(3.0)
|
|
|
|
|
|
def test_seed_reference_is_excluded_from_latency(conn):
|
|
"""Redundant today (the sweep writes no wall clock) and kept anyway."""
|
|
_latency_row(conn, "m", "neuralwatt", 2.0, 1.0)
|
|
for i in range(20):
|
|
_latency_row(
|
|
conn, "m", "neuralwatt", 999.0, 999.0, task_category="seed_reference"
|
|
)
|
|
conn.commit()
|
|
|
|
series = latency_series(conn, _cfg())
|
|
assert series["wall_observations"] == 1
|
|
assert series["wall_p50"] == pytest.approx(2.0)
|
|
|
|
|
|
def test_rows_outside_the_latency_window_are_dropped(conn):
|
|
_latency_row(conn, "m", "neuralwatt", 2.0, 1.0)
|
|
_latency_row(conn, "m", "neuralwatt", 999.0, 999.0, age_hours=200)
|
|
conn.commit()
|
|
|
|
series = latency_series(conn, _cfg(latency_window_hours=168))
|
|
assert series["wall_observations"] == 1
|
|
|
|
|
|
def test_latency_degrades_rather_than_raising_without_the_columns(tmp_path):
|
|
"""Both columns arrive by ALTER, and `admin` may open a pre-migration DB.
|
|
|
|
A missing column must degrade this series, never 500 the whole /metrics
|
|
payload — the contract `cache_rate_series` already keeps.
|
|
"""
|
|
schema = SCHEMA_SQL.replace(" router_wall_seconds REAL,\n", "").replace(
|
|
" router_ttft_seconds REAL,\n", ""
|
|
)
|
|
|
|
c = sqlite3.connect(tmp_path / "old.db")
|
|
c.executescript(schema)
|
|
c.row_factory = sqlite3.Row
|
|
try:
|
|
# Assert on the DB, not on the SQL text: both names also appear in
|
|
# schema.sql's prose comments, so a substring check on the source
|
|
# would pass while the columns were still there.
|
|
cols = {r[1] for r in c.execute("PRAGMA table_info(energy_observations)")}
|
|
assert "router_wall_seconds" not in cols
|
|
assert "router_ttft_seconds" not in cols
|
|
|
|
series = latency_series(c, _cfg())
|
|
finally:
|
|
c.close()
|
|
|
|
assert series["wall_observations"] == 0
|
|
assert series["wall_p50"] is None
|
|
assert series["by_model"] == []
|
|
|
|
|
|
def test_latency_config_defaults_when_the_knobs_are_absent(conn):
|
|
bare = SimpleNamespace(objective=SimpleNamespace())
|
|
series = latency_series(conn, bare)
|
|
assert series["window_hours"] == metrics._LATENCY_WINDOW_HOURS
|
|
assert series["min_observations"] == metrics._LATENCY_MIN_OBSERVATIONS
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# The constraint both series are under
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_neither_series_is_read_by_the_routing_path():
|
|
"""Measurement only. A live experiment is running; a series that reached
|
|
the ranker would confound it.
|
|
|
|
Asserted against the module source rather than trusted to review: the two
|
|
names must not appear in routing.py or on dispatcher's dispatch path at
|
|
all. ``dispatcher`` calls them exactly once each, from the /metrics
|
|
endpoint.
|
|
"""
|
|
import dispatcher
|
|
import routing
|
|
|
|
routing_src = inspect.getsource(routing)
|
|
for name in ("cost_estimate_calibration", "latency_series"):
|
|
assert name not in routing_src, (
|
|
f"{name} reached routing.py. Both series are report-only; "
|
|
"applying one changes which model gets picked."
|
|
)
|
|
|
|
dispatcher_src = inspect.getsource(dispatcher)
|
|
for name in ("cost_estimate_calibration", "latency_series"):
|
|
# Once in the import block, once in the /metrics payload.
|
|
assert dispatcher_src.count(name) == 2, (
|
|
f"{name} is referenced {dispatcher_src.count(name)} times in "
|
|
"dispatcher.py; it should appear only in the import and in the "
|
|
"/metrics payload."
|
|
)
|
|
|
|
|
|
def test_metrics_still_imports_without_dispatcher():
|
|
"""metrics.py must never import dispatcher — the circular-import break."""
|
|
src = inspect.getsource(metrics)
|
|
assert "import dispatcher" not in src
|