Files
6krrt/tests/test_admin_models.py
adlee-was-taken 99a6083516 fix(admin): show a profile's models, and pick them from a two-pane list
The profiles page could tell you a profile admitted 40 models and not
which 40. The only surface for the list was a title= tooltip truncated to
eight names, even though /admin/api/profiles returns the whole array. The
detail pane now renders it.

The allowed_model_ids control had a worse version of the same problem. It
was one <select multiple> holding the entire 446-row catalog and showing
six rows, so a profile pinned to kimi-k3 put its first selected option at
index 72 and the browser's scroll restore landed several hundred pixels
short of it. The selection was correct and invisible, which reads as no
selection at all -- and Duplicate did not scroll at all, so nothing looked
selected there under any circumstances.

It is now two panes, Available and Selected. What is chosen is the only
thing in the Selected pane, so it cannot go off screen. Filtering narrows
Available while Selected stays put, and "Add all" takes the whole filtered
set, so "filter to kimi, add all shown" is one gesture instead of a
ctrl-click hunt.

That also closes a silent data loss. Selection used to be read back out of
selectedOptions, so a stored id with no matching <option> -- a model
deprecated, renamed, or pruned from the provider allowlist since the
profile was written -- was simply not selected, and the next save posted
the selection without it. Opening Edit and pressing Save with no changes
turned a profile pinned to one model into one that admits everything, with
no warning. Selection is now a Set on the picker root, so an id the
catalog no longer offers stays in Selected flagged "not routable" and
round-trips.

GET /admin/api/models defaults to routable rows only, because every model
*picker* in the portal reads it and offering a row the router will never
choose is worse than offering nothing. For an allowlisted provider that
filter IS the allowlist: the poller deprecates every OpenRouter row not on
provider_model_allowlist, so the picker offers exactly the 30 allowlisted
models rather than all 422. A model becomes selectable by being
allowlisted and then polled, which is the order an operator expects.
include_unroutable=true returns everything for models.html, whose whole
job is the availability table and whose "Show deprecated / stale" toggle
is the intended way to see them.

Also: the list summary reads "2 models" instead of "1 filter", and the
helper prose under both pickers is gone -- the pane headers and the
"none = all allowed" empty state say it where it is needed.

Findings and what is still open: plans/admin-portal-functional-sweep.md

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
2026-09-08 10:59:04 -04:00

495 lines
19 KiB
Python

"""Tests for the /admin/api/models read surfaces.
``admin.py`` mirrors ``metrics.py``'s contract — never import dispatcher, take
``(conn, cfg)`` explicitly — and is mounted under the ``/admin`` prefix. These
tests drive a real TestClient GET against the seeded temp DB, mirroring
``test_admin_health.py``'s ``seeded_client`` fixture.
The routes under test:
- GET /admin/api/models -> list of all models + proficiency
- GET /admin/api/models/{model_id}/{provider}-> single model detail (404 if absent)
- POST /admin/api/models/{model_id}/{provider}/availability -> upsert override
- DELETE /admin/api/models/{model_id}/{provider}/availability -> remove override
"""
from __future__ import annotations
import sqlite3
from datetime import datetime, timezone
from pathlib import Path
import pytest
from starlette.testclient import TestClient
import dispatcher
from config import load_config
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
CFG = load_config(str(ROOT / "config" / "config.yaml"))
_ADMIN_TABLE_SQL = """
CREATE TABLE IF NOT EXISTS admin_model_overrides (
model_id TEXT NOT NULL,
provider TEXT NOT NULL,
availability TEXT NOT NULL,
reason TEXT,
updated_at TEXT NOT NULL,
PRIMARY KEY (model_id, provider)
);
CREATE INDEX IF NOT EXISTS idx_admin_model_overrides_availability
ON admin_model_overrides (availability);
"""
def _make_db(tmp_path: Path) -> sqlite3.Connection:
conn = sqlite3.connect(str(tmp_path / "test.db"))
conn.row_factory = sqlite3.Row
conn.executescript(SCHEMA_SQL)
conn.executescript(_ADMIN_TABLE_SQL)
return conn
def _seed_models(conn: sqlite3.Connection, model_ids: tuple[str, ...]) -> None:
for model_id in model_ids:
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, display_name, tier,
context_window, effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_tools, supports_json_mode, supports_vision,
supports_reasoning, reasoning_default_enabled,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, ?, ?, ?, 192500, 16384,
?, ?, 1, 1, 1, 1, 1,
'standard', 'default', 'full', 'public', 'active',
'2026-08-22T00:00:00+00:00')
""",
(
model_id,
model_id,
model_id,
2,
262128,
0.30,
0.10,
),
)
conn.commit()
def _seed_proficiency(
conn: sqlite3.Connection,
spec: tuple[tuple[str, str, float], ...],
) -> None:
"""Insert proficiency rows as (model_id, category, blended_score)."""
for model_id, category, blended in spec:
conn.execute(
"INSERT INTO proficiency (model_id, provider, category, "
"blended_score, source, last_updated) "
"VALUES (?, 'neuralwatt', ?, ?, 'self_eval_thin', "
"'2026-01-01T00:00:00+00:00')",
(model_id, category, blended),
)
conn.commit()
@pytest.fixture
def seeded_client(tmp_path, monkeypatch):
"""A TestClient wired to a seeded temp DB, at /admin."""
conn = _make_db(tmp_path)
_seed_models(conn, ("cheap", "dear", "tiny"))
_seed_proficiency(
conn,
(
("cheap", "coding_general", 0.9),
("cheap", "debugging", 0.85),
("dear", "coding_general", 0.95),
),
)
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
with TestClient(dispatcher.app) as client:
yield client
@pytest.fixture
def empty_client(tmp_path, monkeypatch):
"""A TestClient over an empty DB (no models rows) at /admin."""
conn = _make_db(tmp_path)
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
with TestClient(dispatcher.app) as client:
yield client
def test_admin_models_returns_all_models_with_proficiency(seeded_client):
"""GET /admin/api/models returns one object per seeded model, each with a
category -> blended_score proficiency map."""
resp = seeded_client.get("/admin/api/models")
assert resp.status_code == 200
rows = resp.json()
assert isinstance(rows, list)
assert len(rows) == 3
by_id = {r["model_id"]: r for r in rows}
assert set(by_id) == {"cheap", "dear", "tiny"}
cheap = by_id["cheap"]
assert cheap["proficiency"] == {"coding_general": 0.9, "debugging": 0.85}
assert by_id["dear"]["proficiency"] == {"coding_general": 0.95}
assert by_id["tiny"]["proficiency"] == {}
def test_admin_models_row_shape(seeded_client):
"""Each model object carries every required scalar field."""
row = seeded_client.get("/admin/api/models").json()[0]
for key in (
"model_id",
"provider",
"base_model_id",
"display_name",
"availability",
"tier",
"context_window",
"effective_context_window",
"latency_class",
"reasoning_mode",
"context_variant",
"access_level",
"supports_tools",
"supports_json_mode",
"supports_vision",
"supports_reasoning",
"reasoning_default_enabled",
"cost_per_1m_prompt",
"cost_per_1m_completion",
"proficiency",
):
assert key in row, f"missing field {key!r}"
assert row["provider"] == "neuralwatt"
assert row["availability"] == "active"
assert row["tier"] == 2
assert row["context_window"] == 262128
assert row["supports_tools"] is True
assert row["supports_json_mode"] is True
assert row["supports_vision"] is True
assert row["supports_reasoning"] is True
assert row["reasoning_default_enabled"] is True
def test_admin_model_detail_returns_single_object(seeded_client):
"""GET /admin/api/models/{model_id}/{provider} returns one object, not a list."""
resp = seeded_client.get("/admin/api/models/cheap/neuralwatt")
assert resp.status_code == 200
data = resp.json()
assert isinstance(data, dict)
assert data["model_id"] == "cheap"
assert data["provider"] == "neuralwatt"
assert data["proficiency"] == {"coding_general": 0.9, "debugging": 0.85}
def test_admin_model_detail_unknown_model_returns_404(seeded_client):
"""A model_id that does not exist -> 404."""
resp = seeded_client.get("/admin/api/models/nope/neuralwatt")
assert resp.status_code == 404
def test_admin_model_detail_unknown_provider_returns_404(seeded_client):
"""A provider that does not exist for a known model -> 404."""
resp = seeded_client.get("/admin/api/models/cheap/nope")
assert resp.status_code == 404
def test_admin_models_empty_table_returns_empty_list(empty_client):
"""GET /admin/api/models with zero rows -> 200 + empty list."""
resp = empty_client.get("/admin/api/models")
assert resp.status_code == 200
assert resp.json() == []
def test_admin_models_never_expose_session_dir(seeded_client):
"""No response object may name session_dir or carry a prompt/conversation key."""
rows = seeded_client.get("/admin/api/models").json()
for row in rows:
for key in ("session_dir", "prompt", "conversation"):
assert key not in row, f"leaked {key!r} in model row"
def test_admin_model_detail_never_expose_session_dir(seeded_client):
"""The single-model JSON must never name session_dir or conversation keys."""
data = seeded_client.get("/admin/api/models/cheap/neuralwatt").json()
for key in ("session_dir", "prompt", "conversation"):
assert key not in data, f"leaked {key!r} in model detail"
# --- admin model overrides --------------------------------------------------
def test_post_override_sets_effective_availability_deprecated(seeded_client):
"""POST override -> effective_availability flips to deprecated, is_overridden
is True, raw availability stays 'active'."""
resp = seeded_client.post(
"/admin/api/models/cheap/neuralwatt/availability",
json={"availability": "deprecated", "reason": "failing verification"},
)
assert resp.status_code == 200
data = resp.json()
assert data["is_overridden"] is True
assert data["effective_availability"] == "deprecated"
assert data["availability"] == "active"
def test_post_override_sets_effective_availability_stale(seeded_client):
"""POST override with stale availability."""
resp = seeded_client.post(
"/admin/api/models/cheap/neuralwatt/availability",
json={"availability": "stale", "reason": "last seen long ago"},
)
assert resp.status_code == 200
data = resp.json()
assert data["is_overridden"] is True
assert data["effective_availability"] == "stale"
def test_post_override_invalid_availability_returns_422(seeded_client):
"""Bad availability value -> 422 validation error."""
resp = seeded_client.post(
"/admin/api/models/cheap/neuralwatt/availability",
json={"availability": "retired", "reason": "nope"},
)
assert resp.status_code == 422
def test_post_override_nonexistent_model_returns_404(seeded_client):
"""Trying to override a model that doesn't exist -> 404."""
resp = seeded_client.post(
"/admin/api/models/nonexistent/neuralwatt/availability",
json={"availability": "deprecated", "reason": "test"},
)
assert resp.status_code == 404
def test_delete_override_reverts_to_db_value(seeded_client):
"""POST then DELETE -> is_overridden becomes False, effective_availability
reverts to the DB value ('active')."""
# Set override
seeded_client.post(
"/admin/api/models/cheap/neuralwatt/availability",
json={"availability": "deprecated", "reason": "test"},
)
# Delete it
resp = seeded_client.delete(
"/admin/api/models/cheap/neuralwatt/availability"
)
assert resp.status_code == 200
data = resp.json()
assert data["is_overridden"] is False
assert data["effective_availability"] == "active"
assert data["availability"] == "active"
# --- vendor/model-shaped ids (OpenRouter's native id format) ---------------
#
# Every OpenRouter model id contains a slash (e.g. "google/lyria-3-pro-preview"),
# and the browser dutifully encodeURIComponent()s it -- but Starlette decodes
# %2F back into a literal '/' before matching a plain `str` path parameter
# against the route, so the request splits into extra path segments and 404s.
# Confirmed live: every admin availability-override click on an OpenRouter
# model failed this way until model_id was declared `:path`.
@pytest.fixture
def slash_id_client(tmp_path, monkeypatch):
"""A TestClient seeded with one vendor/model-shaped id."""
conn = _make_db(tmp_path)
_seed_models(conn, ("google/lyria-3-pro-preview",))
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
with TestClient(dispatcher.app) as client:
yield client
def test_model_detail_resolves_slash_containing_model_id(slash_id_client):
"""GET /admin/api/models/{model_id}/{provider} works for a vendor/model id,
sent the way a browser actually sends it: %2F for the embedded slash."""
resp = slash_id_client.get(
"/admin/api/models/google%2Flyria-3-pro-preview/neuralwatt"
)
assert resp.status_code == 200
assert resp.json()["model_id"] == "google/lyria-3-pro-preview"
def test_post_override_resolves_slash_containing_model_id(slash_id_client):
"""POST .../availability works for a vendor/model id (%2F-encoded)."""
resp = slash_id_client.post(
"/admin/api/models/google%2Flyria-3-pro-preview/neuralwatt/availability",
json={"availability": "deprecated", "reason": "non-chat model"},
)
assert resp.status_code == 200
data = resp.json()
assert data["model_id"] == "google/lyria-3-pro-preview"
assert data["effective_availability"] == "deprecated"
def test_delete_override_resolves_slash_containing_model_id(slash_id_client):
"""DELETE .../availability works for a vendor/model id (%2F-encoded)."""
slash_id_client.post(
"/admin/api/models/google%2Flyria-3-pro-preview/neuralwatt/availability",
json={"availability": "deprecated", "reason": "test"},
)
resp = slash_id_client.delete(
"/admin/api/models/google%2Flyria-3-pro-preview/neuralwatt/availability"
)
assert resp.status_code == 200
assert resp.json()["effective_availability"] == "active"
def test_model_list_reflects_effective_availability(seeded_client):
"""GET /admin/api/models includes effective_availability and is_overridden."""
seeded_client.post(
"/admin/api/models/cheap/neuralwatt/availability",
json={"availability": "deprecated", "reason": "test"},
)
# include_unroutable: the deprecated row is exactly what this assertion is
# about, and the default response no longer carries it.
rows = seeded_client.get("/admin/api/models?include_unroutable=true").json()
by_id = {r["model_id"]: r for r in rows}
cheap = by_id["cheap"]
assert "effective_availability" in cheap
assert "is_overridden" in cheap
# dear and tiny should not be overridden
assert cheap["is_overridden"] is True
assert cheap["effective_availability"] == "deprecated"
assert by_id["dear"]["is_overridden"] is False
def test_model_list_defaults_to_routable_only(seeded_client):
"""The default response is the pickers' contract: routable rows only.
Every model *picker* in the portal reads this endpoint, and for an
allowlisted provider the availability filter IS the allowlist -- the
poller deprecates any row that is not allowlisted. Offering a deprecated
row would let an operator build a profile that admits nothing, with
nothing on screen saying why.
"""
seeded_client.post(
"/admin/api/models/cheap/neuralwatt/availability",
json={"availability": "deprecated", "reason": "test"},
)
default_ids = {r["model_id"] for r in seeded_client.get("/admin/api/models").json()}
assert "cheap" not in default_ids
assert "dear" in default_ids
full = seeded_client.get("/admin/api/models?include_unroutable=true").json()
assert "cheap" in {r["model_id"] for r in full}
assert all(r["effective_availability"] == "active" for r in
seeded_client.get("/admin/api/models").json())
def test_model_list_default_excludes_stale(seeded_client):
"""A 'stale' override is not routable either, so it is not offered."""
seeded_client.post(
"/admin/api/models/cheap/neuralwatt/availability",
json={"availability": "stale", "reason": "test"},
)
default_ids = {r["model_id"] for r in seeded_client.get("/admin/api/models").json()}
assert "cheap" not in default_ids
full_ids = {
r["model_id"]
for r in seeded_client.get("/admin/api/models?include_unroutable=true").json()
}
assert "cheap" in full_ids
def _seed_tier3_ceiling_models(conn: sqlite3.Connection) -> None:
for model_id, eff_ctx in (("large", 200000), ("small", 100000)):
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, display_name, tier,
context_window, effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_tools, supports_json_mode, supports_vision,
supports_reasoning, reasoning_default_enabled,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, ?, 3, ?, ?, 16384,
0.30, 0.10, 1, 1, 1, 1, 1,
'standard', 'default', 'full', 'public', 'active',
'2026-08-22T00:00:00+00:00')
""",
(model_id, model_id, model_id, eff_ctx, eff_ctx),
)
conn.commit()
def _seed_tier3_chat_decision(conn: sqlite3.Connection, tokens: int) -> None:
conn.execute(
"""
INSERT INTO route_decisions (
observed_at, kind, task_category, task_tier, required_context_tokens,
confidence, classifier_ms, classification_source, latency_tolerance,
candidates_considered, selected_model, selected_provider,
runner_up_models, est_cost_usd, est_proficiency,
session_key, tools, images, json_mode, streamed,
flex_preference, flex_swapped, flex_forced
) VALUES (?, 'chat', 'coding_general', 3, ?, 0.95, 200,
'classifier', 'interactive', 2, 'large', 'neuralwatt',
NULL, 0.001, 0.9, 'abc123', 0, 0, 0, 0,
'auto', 0, 1)
""",
(datetime.now(timezone.utc).isoformat(), tokens),
)
conn.commit()
def test_set_availability_returns_warning_when_override_drops_ceiling(
tmp_path, monkeypatch
):
"""Deprecating a tier-3 model via admin override drops the ceiling to the
remaining smaller model. Observed demand above that ceiling is returned in
the response ``warnings`` list.
"""
conn = _make_db(tmp_path)
_seed_tier3_ceiling_models(conn)
_seed_tier3_chat_decision(conn, 120000)
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
with TestClient(dispatcher.app) as client:
resp = client.post(
"/admin/api/models/large/neuralwatt/availability",
json={"availability": "deprecated", "reason": "drops ceiling"},
)
assert resp.status_code == 200
data = resp.json()
assert "warnings" in data, data
assert any(
"tier 3 context ceiling (100000) is below observed max demand (120000)" in w
for w in data["warnings"]
)