Pinch is the router's only lever on the ~92.6% of cost that is prompt tokens, and until now there was no way to ask whether it earned its place. Instrumentation: - route_decisions gains pinch_original_tokens / pinch_final_tokens, in config/schema.sql and the code-side ensure_route_decisions migration, so live databases predating the columns pick them up on next start. - metrics.pinch_summary aggregates 30-day savings: share pruned, total and median tokens saved, and estimated dollars saved. The dollar figure is priced at the SAME blended prompt rate routing.estimated_cost uses (src/routing.py:363-374), including the fallback when the cached price is NULL, and reads cfg.objective.assumed_cache_rate so the dashboard and the router's own cost model cannot disagree. - Wired into /metrics, /admin/api/snapshot, the TUI model, and an admin dashboard card. Returns None when pinch is disabled, matching local_energy_summary's contract; all consumers guard for it. - The pinch log moves from debug to info, and only when pruning occurred. Accounting fix: - prune_context and _relevance_order_for both take extra_fixed_tokens, and the dispatcher passes the tool-definition overhead. opencode sends ~32k prompt tokens of tool definitions on a trivial request, and pinch was excluding all of it from the budget comparison. Threading it through _relevance_order_for as well matters: without it, relevance ordering would silently decline to run on exactly the requests that newly need pruning. Persistence covers the routed success, routed rejection, and passthrough paths. On passthrough the prune is hoisted above the persist so pinch stats are recorded rather than NULL -- the persist deliberately stays above _check_pinned_capabilities, which raises 422, so a rejected pin still writes the decision row that explains it. 860 tests pass, serially and under pytest-xdist. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
212 lines
7.2 KiB
Python
212 lines
7.2 KiB
Python
"""Tests for the /admin/api health and snapshot endpoints.
|
|
|
|
``admin.py`` is a standalone FastAPI router that mirrors ``metrics.py``'s
|
|
contract — never import ``dispatcher``, take ``(conn, cfg)`` explicitly — and
|
|
is mounted onto the dispatcher app under the ``/admin`` prefix. These tests
|
|
drive a real TestClient GET against the seeded temp DB (never a mock-call
|
|
assertion), mirroring ``test_metrics_endpoint.py``.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import sqlite3
|
|
import subprocess
|
|
import sys
|
|
from datetime import datetime, timedelta, timezone
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
from starlette.testclient import TestClient
|
|
|
|
import dispatcher
|
|
from config import load_config
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
|
CFG = load_config(str(ROOT / "config" / "config.yaml"))
|
|
|
|
|
|
def _now() -> datetime:
|
|
return datetime.now(timezone.utc)
|
|
|
|
|
|
def _make_db(tmp_path: Path) -> sqlite3.Connection:
|
|
conn = sqlite3.connect(str(tmp_path / "test.db"))
|
|
conn.row_factory = sqlite3.Row
|
|
conn.executescript(SCHEMA_SQL)
|
|
return conn
|
|
|
|
|
|
def _seed_models(conn: sqlite3.Connection) -> None:
|
|
for model_id, tier, context, cost, vision in (
|
|
("cheap", 2, 262128, 0.30, 1),
|
|
("dear", 2, 262128, 9.00, 0),
|
|
("tiny", 1, 131072, 0.10, 1),
|
|
):
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES (?, 'neuralwatt', ?, ?, ?, 192500, 16384, ?, ?,
|
|
?, 1, 'standard', 'default', 'full', 'public', 'active',
|
|
'2026-08-22T00:00:00+00:00')
|
|
""",
|
|
(model_id, model_id, tier, context, cost, cost / 3, vision),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _seed_decision(conn: sqlite3.Connection) -> None:
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO route_decisions (
|
|
observed_at, kind, task_category, task_tier, required_context_tokens,
|
|
confidence, classifier_ms, classification_source, latency_tolerance,
|
|
candidates_considered, selected_model, selected_provider,
|
|
runner_up_models, est_cost_usd, est_proficiency,
|
|
session_key, tools, images, json_mode, streamed,
|
|
flex_preference, flex_swapped, flex_forced
|
|
) VALUES (?, 'route', 'coding_general', 2, 100, 0.95, 200,
|
|
'classifier', 'interactive', 5, 'cheap', 'neuralwatt',
|
|
'[{"model_id":"dear","provider":"neuralwatt"}]',
|
|
0.001, 0.9, 'abc123', 0, 0, 0, 0,
|
|
'auto', 0, 1)
|
|
""",
|
|
(_now().isoformat(),),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _seed_energy(conn: sqlite3.Connection) -> None:
|
|
now = _now()
|
|
conn.execute(
|
|
"INSERT INTO energy_observations "
|
|
"(model_id, provider, task_category, completion_tokens, energy_kwh, "
|
|
"cost_usd, carbon_g_co2eq, attribution_ratio, observed_at) "
|
|
"VALUES ('cheap', 'neuralwatt', 'coding_general', 100, 5.0e-05, 0.001, "
|
|
"2.4e-03, 0.25, ?)",
|
|
((now - timedelta(days=2)).isoformat(),),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _seed_verification(conn: sqlite3.Connection) -> None:
|
|
conn.execute(
|
|
"INSERT INTO verifications (model_id, provider, kind, verdict, observed_at) "
|
|
"VALUES ('cheap', 'neuralwatt', 'structural', 'ok', ?)",
|
|
(_now().isoformat(),),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def _seed_proficiency(conn: sqlite3.Connection) -> None:
|
|
conn.execute(
|
|
"INSERT INTO proficiency (model_id, provider, category, blended_score, "
|
|
"source, last_updated) "
|
|
"VALUES ('cheap', 'neuralwatt', 'coding_general', 0.9, "
|
|
"'self_eval_thin', '2026-01-01T00:00:00+00:00')",
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
@pytest.fixture
|
|
def seeded_client(tmp_path, monkeypatch):
|
|
"""A TestClient wired to a seeded temp DB, at /admin."""
|
|
conn = _make_db(tmp_path)
|
|
_seed_models(conn)
|
|
for _ in range(3):
|
|
_seed_decision(conn)
|
|
_seed_energy(conn)
|
|
_seed_verification(conn)
|
|
_seed_proficiency(conn)
|
|
conn.close()
|
|
|
|
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
|
|
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
|
|
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
|
|
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
|
|
|
|
with TestClient(dispatcher.app) as client:
|
|
yield client
|
|
|
|
|
|
def test_admin_health_returns_ok(seeded_client):
|
|
"""GET /admin/api/health returns 200 with {"status": "ok"}."""
|
|
resp = seeded_client.get("/admin/api/health")
|
|
assert resp.status_code == 200
|
|
assert resp.json() == {"status": "ok"}
|
|
|
|
|
|
def test_admin_snapshot_has_all_top_level_keys(seeded_client):
|
|
"""GET /admin/api/snapshot returns 200 with every required key."""
|
|
resp = seeded_client.get("/admin/api/snapshot")
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
for key in (
|
|
"quota",
|
|
"coverage",
|
|
"recent_decisions",
|
|
"per_model",
|
|
"verdict_mix",
|
|
"top_proficiency",
|
|
"local_energy",
|
|
"health",
|
|
"pinch",
|
|
"generated_at",
|
|
):
|
|
assert key in data, f"missing top-level key {key!r}"
|
|
|
|
|
|
def test_admin_snapshot_health_has_expected_shape(seeded_client):
|
|
"""Snapshot's health block carries the same shape as /health."""
|
|
data = seeded_client.get("/admin/api/snapshot").json()
|
|
health = data["health"]
|
|
assert "status" in health
|
|
assert "counts" in health
|
|
assert "classifier_reachable" in health and isinstance(
|
|
health["classifier_reachable"], bool
|
|
)
|
|
assert "tiers" in health
|
|
assert "providers" in health
|
|
assert "api_keys_present" in health
|
|
|
|
|
|
def test_admin_snapshot_returns_seeded_data(seeded_client):
|
|
"""quota / per_model / verdict_mix / top_proficiency reflect the seed."""
|
|
data = seeded_client.get("/admin/api/snapshot").json()
|
|
assert data["quota"] is not None
|
|
assert isinstance(data["per_model"], list)
|
|
assert any(r["model_id"] == "cheap" for r in data["per_model"])
|
|
assert data["verdict_mix"]["ok"] == 1
|
|
assert isinstance(data["top_proficiency"], list)
|
|
assert data["top_proficiency"][0]["model_id"] == "cheap"
|
|
assert data["health"]["counts"]["models"] == 3
|
|
|
|
|
|
def test_admin_snapshot_contains_no_session_dir(seeded_client):
|
|
"""The JSON must never name session_dir or expose conversation text."""
|
|
body = seeded_client.get("/admin/api/snapshot").text
|
|
assert "session_dir" not in body
|
|
|
|
|
|
def test_admin_does_not_import_dispatcher():
|
|
"""Importing admin.py must not pull dispatcher into sys.modules."""
|
|
probe = (
|
|
"import admin; import sys; "
|
|
"assert 'dispatcher' not in sys.modules, "
|
|
"'import admin transitively imported dispatcher'"
|
|
)
|
|
subprocess.run(
|
|
[sys.executable, "-c", probe],
|
|
check=True,
|
|
cwd=str(ROOT),
|
|
capture_output=True,
|
|
env={**os.environ, "PYTHONPATH": str(ROOT / "src")},
|
|
)
|