Sweep findings #3 and #4 (plans/admin-portal-functional-sweep.md). #3. "stale" was inert. The models page offers active / deprecated / stale, and only "deprecated" was ever read. The reasoning in the old docstring was wrong rather than conservative: it said stale "already has its own filter", meaning freshness.exclude_stale -- but that filter reads models.availability, the catalog column, and an admin override lives in admin_model_overrides and is never merged into those rows. Selecting "stale" wrote a row and changed nothing. Measured on /route before the fix: 27 candidates with a stale override set, 27 with it cleared, against 28 when a deprecated override was cleared. Both off-states now exclude, unconditionally rather than gated on freshness.exclude_stale / exclude_deprecated. Those settings are a policy about catalog age; an override is an operator naming one model, and an explicit instruction should not need a second flag to take effect. _admin_deprecated_models is renamed _admin_excluded_models in all three modules, since "deprecated" no longer describes what it returns. #4. The override bound /route but nothing else. /v1/models reads models.availability directly, so a switched-off model stayed advertised; _resolve_pinned_provider had no check, so a client that took it from that list was served by it as normal. The override applied to routed traffic only, which is the opposite of what "deprecated" reads as in the portal. Both are now closed, and /v1/models grows the same guard it already applies under gaming mode -- with the comment there giving the reason verbatim: do not advertise a model whose pin is about to be refused. The new tests parametrize over both off-states rather than testing "deprecated" and trusting "stale" to follow, since trusting that is exactly how stale stayed inert. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
291 lines
10 KiB
Python
291 lines
10 KiB
Python
"""Integration test: admin_model_overrides wire into routing.
|
|
|
|
Verifies that POST /admin/api/models/{model_id}/{provider}/availability
|
|
actually removes the model from /route candidates. The wired exclude set
|
|
(_admin_excluded_models) must intersect with select_candidates'
|
|
exclude_models filter so the overridden model never appears in the
|
|
route response.
|
|
|
|
Uses the pattern from test_route_decisions.py: temp DB, seeded models with
|
|
energy + proficiency so a specific model WOULD win, then override + /route
|
|
assertion.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import sqlite3
|
|
import time
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import pytest
|
|
from openai import OpenAI
|
|
from starlette.testclient import TestClient
|
|
|
|
import dispatcher
|
|
from config import load_config
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
|
CFG = load_config(str(ROOT / "config" / "config.yaml"))
|
|
|
|
ADMIN_TABLE_SQL = """
|
|
CREATE TABLE IF NOT EXISTS admin_model_overrides (
|
|
model_id TEXT NOT NULL,
|
|
provider TEXT NOT NULL,
|
|
availability TEXT NOT NULL,
|
|
reason TEXT,
|
|
updated_at TEXT NOT NULL,
|
|
PRIMARY KEY (model_id, provider)
|
|
);
|
|
CREATE INDEX IF NOT EXISTS idx_admin_model_overrides_availability
|
|
ON admin_model_overrides (availability);
|
|
"""
|
|
|
|
# A model we will seed with best proficiency + energy so it WOULD be picked.
|
|
WINNER_MODEL = "premium"
|
|
# A second model that would be runner-up.
|
|
RUNNER_MODEL = "mid"
|
|
|
|
|
|
def _completion(model_id: str) -> dict:
|
|
"""An OpenAI-compatible completion body for a *fake* provider response."""
|
|
return {
|
|
"id": f"chatcmpl-{model_id}",
|
|
"object": "chat.completion",
|
|
"model": model_id,
|
|
"created": int(time.time()),
|
|
"choices": [{"index": 0, "finish_reason": "stop", "message": {"role": "assistant", "content": "done"}}],
|
|
}
|
|
|
|
|
|
class FakeResponse:
|
|
"""Minimal fake for requests.post() and httpx.Response."""
|
|
|
|
def __init__(
|
|
self,
|
|
body: dict | None = None,
|
|
lines: list[str] | None = None,
|
|
headers: dict | None = None,
|
|
) -> None:
|
|
self.body = body or _completion("dummy")
|
|
self.lines = lines or []
|
|
self.headers = headers or {"content-type": "application/json"}
|
|
self.status_code = 200
|
|
|
|
@property
|
|
def text(self) -> str:
|
|
return json.dumps(self.body)
|
|
|
|
@property
|
|
def content(self) -> bytes:
|
|
return json.dumps(self.body).encode()
|
|
|
|
def json(self) -> dict:
|
|
return self.body
|
|
|
|
|
|
@pytest.fixture
|
|
def admin_override_router(tmp_path: Path, monkeypatch) -> tuple[TestClient, Path]:
|
|
"""A TestClient with a temp DB that has admin_model_overrides table,
|
|
seeded with two models (WINNER_MODEL and RUNNER_MODEL) where WINNER
|
|
has best proficiency + energy."""
|
|
|
|
import admin
|
|
|
|
db_path = tmp_path / "test.db"
|
|
conn = sqlite3.connect(str(db_path))
|
|
conn.executescript(SCHEMA_SQL)
|
|
conn.executescript(ADMIN_TABLE_SQL)
|
|
dispatcher.ensure_route_decisions(conn)
|
|
|
|
# Seed two models: WINNER (tier 2, best) and RUNNER (tier 2).
|
|
for mid, cost_prompt, cost_compl, prof_score in [
|
|
(WINNER_MODEL, 0.50, 0.30, 0.95),
|
|
(RUNNER_MODEL, 0.30, 0.15, 0.60),
|
|
]:
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, display_name, tier,
|
|
context_window, effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion,
|
|
supports_tools, supports_json_mode, supports_vision,
|
|
supports_reasoning, reasoning_default_enabled,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, last_updated
|
|
) VALUES (?, 'neuralwatt', ?, ?, ?, 262128, 192500, 16384,
|
|
?, ?, 1, 1, 1, 1, 1,
|
|
'standard', 'default', 'full', 'public', 'active',
|
|
'2026-08-22T00:00:00+00:00')
|
|
""",
|
|
(
|
|
mid,
|
|
mid,
|
|
mid,
|
|
2,
|
|
cost_prompt,
|
|
cost_compl,
|
|
),
|
|
)
|
|
|
|
# Seed proficiency: WINNER has the best score for coding_general.
|
|
conn.execute(
|
|
"INSERT INTO proficiency (model_id, provider, category, "
|
|
"blended_score, source, last_updated) "
|
|
"VALUES (?, 'neuralwatt', 'coding_general', ?, 'self_eval_thin', "
|
|
"'2026-01-01T00:00:00+00:00')",
|
|
(WINNER_MODEL, 0.95),
|
|
)
|
|
conn.execute(
|
|
"INSERT INTO proficiency (model_id, provider, category, "
|
|
"blended_score, source, last_updated) "
|
|
"VALUES (?, 'neuralwatt', 'coding_general', ?, 'self_eval_thin', "
|
|
"'2026-01-01T00:00:00+00:00')",
|
|
(RUNNER_MODEL, 0.60),
|
|
)
|
|
|
|
# Seed one energy observation so scoring works.
|
|
now = "2026-08-22T00:00:00+00:00"
|
|
for mid in (WINNER_MODEL, RUNNER_MODEL):
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO energy_observations (
|
|
model_id, provider, prompt_tokens, completion_tokens,
|
|
energy_kwh, carbon_g_co2eq, cost_usd, observed_at
|
|
) VALUES (?, 'neuralwatt', 200, 500, 0.005, 2.0, 0.10, ?)
|
|
""",
|
|
(mid, now),
|
|
)
|
|
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
# Redirect the dispatcher to the temp DB.
|
|
monkeypatch.setattr(dispatcher.cfg.database, "path", str(db_path))
|
|
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
|
|
monkeypatch.setattr(dispatcher.cfg.freshness, "exclude_stale", True)
|
|
monkeypatch.setattr(dispatcher.cfg.freshness, "exclude_deprecated", True)
|
|
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
|
|
monkeypatch.setattr(dispatcher.cfg.logging, "log_route_decisions", True)
|
|
# Exploration defaults to enabled; this fixture asserts a deterministic
|
|
# winner on tier-2 routes, so pin epsilon to 0. Leave the enabled flag as
|
|
# configured so the production default is still exercised structurally.
|
|
monkeypatch.setattr(dispatcher.cfg.exploration, "epsilon", 0.0)
|
|
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
|
|
|
|
# Ensure admin tables are wired into the startup migration.
|
|
from admin import ensure_admin_tables
|
|
|
|
temp_conn = sqlite3.connect(str(db_path))
|
|
temp_conn.row_factory = sqlite3.Row
|
|
try:
|
|
ensure_admin_tables(conn=temp_conn)
|
|
except Exception:
|
|
pass # may already exist
|
|
finally:
|
|
temp_conn.close()
|
|
|
|
# Stub the classifier — always returns coding_general, tier 2.
|
|
monkeypatch.setattr(
|
|
dispatcher,
|
|
"classify",
|
|
lambda task, context: dispatcher.Classification(
|
|
task_category="coding_general",
|
|
task_tier=2,
|
|
required_context_tokens=100,
|
|
confidence=0.9,
|
|
),
|
|
)
|
|
|
|
# Stub the provider call so the /route endpoint doesn't actually call
|
|
# NeuralWatt. It just returns a minimal response.
|
|
def fake_post(url: str, headers: Any = None, json: Any = None,
|
|
stream: bool = False, timeout: int = 600):
|
|
if stream:
|
|
return FakeResponse(lines=["data: ..."])
|
|
return FakeResponse(_completion(json["model"] if json else "dummy"))
|
|
|
|
monkeypatch.setattr(dispatcher.requests, "post", fake_post)
|
|
|
|
app = dispatcher.app
|
|
|
|
with TestClient(app) as client:
|
|
yield client, db_path
|
|
|
|
|
|
def test_admin_override_excludes_model_from_route(admin_override_router):
|
|
"""When an admin override marks WINNER_MODEL as deprecated, POST /route
|
|
must NOT pick it. The selected model should be RUNNER_MODEL instead,
|
|
and candidates_considered should be 1 (not 2).
|
|
|
|
This is the CRITICAL integration test: override → router exclusion.
|
|
"""
|
|
client, db_path = admin_override_router
|
|
|
|
# First, verify that WITHOUT an override, the router picks WINNER.
|
|
resp = client.post("/route", json={"task": "write a python function"})
|
|
assert resp.status_code == 200
|
|
body = resp.json()
|
|
assert body["selected"]["model_id"] == WINNER_MODEL
|
|
assert body["candidates_considered"] == 2
|
|
|
|
# Now mark WINNER as deprecated via admin override.
|
|
override_resp = client.post(
|
|
f"/admin/api/models/{WINNER_MODEL}/neuralwatt/availability",
|
|
json={"availability": "deprecated", "reason": "failing verification"},
|
|
)
|
|
assert override_resp.status_code == 200
|
|
override_data = override_resp.json()
|
|
assert override_data["is_overridden"] is True
|
|
assert override_data["effective_availability"] == "deprecated"
|
|
|
|
# POST /route again: the router must NOT pick the overridden model.
|
|
resp = client.post("/route", json={"task": "write a python function"})
|
|
assert resp.status_code == 200
|
|
body = resp.json()
|
|
assert body["selected"]["model_id"] == RUNNER_MODEL
|
|
assert body["candidates_considered"] == 1
|
|
# Verify the override model is in the excluded set via the decision log.
|
|
conn = sqlite3.connect(str(db_path))
|
|
conn.row_factory = sqlite3.Row
|
|
decision_row = conn.execute(
|
|
"SELECT * FROM route_decisions ORDER BY id DESC LIMIT 1"
|
|
).fetchone()
|
|
conn.close()
|
|
assert decision_row["selected_model"] == RUNNER_MODEL
|
|
|
|
|
|
def test_admin_override_revert_includes_model_again(admin_override_router):
|
|
"""After DELETE on the admin override, the model must become routable
|
|
again and be picked if it still has best scores."""
|
|
client, db_path = admin_override_router
|
|
|
|
# Mark as deprecated.
|
|
client.post(
|
|
f"/admin/api/models/{WINNER_MODEL}/neuralwatt/availability",
|
|
json={"availability": "deprecated", "reason": "test"},
|
|
)
|
|
|
|
# Verify route picks RUNNER.
|
|
resp = client.post("/route", json={"task": "write a python function"})
|
|
assert resp.json()["selected"]["model_id"] == RUNNER_MODEL
|
|
|
|
# Delete override.
|
|
delete_resp = client.delete(
|
|
f"/admin/api/models/{WINNER_MODEL}/neuralwatt/availability"
|
|
)
|
|
assert delete_resp.status_code == 200
|
|
del_data = delete_resp.json()
|
|
assert del_data["is_overridden"] is False
|
|
|
|
# Route again: WINNER should be back on the radar.
|
|
resp = client.post("/route", json={"task": "write a python function"})
|
|
assert resp.status_code == 200
|
|
body = resp.json()
|
|
assert body["selected"]["model_id"] == WINNER_MODEL
|
|
assert body["candidates_considered"] == 2
|
|
|
|
|