Files
6krrt/tests/test_admin_proficiency.py

308 lines
12 KiB
Python

"""GET /admin/api/proficiency and the page that renders it.
``proficiency_score`` is the only category-dependent term in the ranking, so
this table decides routing — and the portal's entire surface for it was a
top-N list for one category on the dashboard. An operator could see which
model won a decision and not why.
The single load-bearing requirement, restated because it is easy to lose in a
refactor: a score that traffic EARNED must be distinguishable from one that
was COPIED IN. ``outcome_prior`` and ``self_eval_thin`` must not look like
``outcome_blended``. This project already made that mistake once, on ``-flex``
rows, where an inherited score froze at its first copy while the row it came
from moved from 0.85 to 0.973.
Fixtures are seeded with known answers rather than read off the live DB, so a
failure here means the code changed and not that the router routed something.
"""
from __future__ import annotations
import sqlite3
from pathlib import Path
import pytest
from fastapi import FastAPI
from starlette.testclient import TestClient
import metrics
from admin import build_router
from config import load_config
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
# (model_id, category, blended, self_eval_n, outcome_n, source, inherited_from)
SEED_PROFICIENCY = [
# Measured here from real traffic: the shape a score should have.
("kimi-k3", "coding_general", 0.94, 14, 31, "outcome_blended", None),
# Inherited: a variant carrying its family's number, never measured.
("kimi-k3-flex", "coding_general", 0.94, 14, 31, "outcome_prior", "kimi-k3"),
# Benchmark only, and thin — two samples is not a measurement of a ceiling.
("gemma-4-31b", "coding_general", 1.00, 2, 0, "self_eval_thin", None),
# Benchmark past the floor.
("gemma-4-31b", "docs_writing", 0.79, 14, 0, "self_eval", None),
]
SEED_OUTCOMES = [
# (model, category, verdict, attributable, applied_at)
("kimi-k3", "coding_general", "succeeded", 1, "2026-09-04T10:00:00+00:00"),
("kimi-k3", "coding_general", "failed", 1, None),
# The one an operator could not see: recorded, then excluded from folding
# because the classification that produced it was degraded.
("gemma-4-31b", "general_chat", "failed", 0, None),
]
def _seed(path: Path) -> None:
conn = sqlite3.connect(str(path))
conn.executescript(SCHEMA_SQL)
conn.executemany(
"""
INSERT INTO models
(model_id, provider, availability, access_level, tier,
context_window, effective_context_window, latency_class,
last_updated)
VALUES (?, 'neuralwatt', 'active', 'public', 2, 128000, 128000,
'standard', '2026-09-01T00:00:00+00:00')
""",
[(m,) for m in sorted({r[0] for r in SEED_PROFICIENCY})],
)
conn.executemany(
"""
INSERT INTO proficiency
(model_id, provider, category, blended_score, self_eval_score,
self_eval_samples, outcome_score, outcome_samples, source,
inherited_from, last_updated)
VALUES (?, 'neuralwatt', ?, ?, ?, ?, ?, ?, ?, ?,
'2026-09-01T00:00:00+00:00')
""",
[
(m, c, b, b, sn, b, on, src, inh)
for (m, c, b, sn, on, src, inh) in SEED_PROFICIENCY
],
)
conn.executemany(
"""
INSERT INTO verifications
(model_id, provider, task_category, kind, verdict, detail,
observed_at, applied_at, model_attributable)
VALUES (?, 'neuralwatt', ?, 'client_outcome', ?, 'seeded',
'2026-09-04T09:00:00+00:00', ?, ?)
""",
[(m, c, v, applied, attr) for (m, c, v, attr, applied) in SEED_OUTCOMES],
)
# A non-outcome verification, which must NOT appear in the outcomes list.
conn.execute(
"""
INSERT INTO verifications
(model_id, provider, task_category, kind, verdict, detail,
observed_at, model_attributable)
VALUES ('kimi-k3', 'neuralwatt', 'coding_general', 'structural',
'malformed', 'not a client report',
'2026-09-04T09:30:00+00:00', 1)
"""
)
conn.commit()
conn.close()
@pytest.fixture
def client(tmp_path):
db_path = tmp_path / "prof.db"
_seed(db_path)
cfg = load_config(str(ROOT / "config" / "config.yaml"))
def _db():
conn = sqlite3.connect(str(db_path))
conn.row_factory = sqlite3.Row
return conn
app = FastAPI()
app.include_router(build_router(cfg, _db, base_dir=str(ROOT)), prefix="/admin")
yield TestClient(app), cfg
# ---------------------------------------------------------------------------
# The endpoint
# ---------------------------------------------------------------------------
def test_endpoint_returns_every_row_with_its_provenance(client):
c, cfg = client
body = c.get("/admin/api/proficiency").json()
assert len(body["rows"]) == len(SEED_PROFICIENCY)
assert body["categories"] == list(cfg.proficiency.categories)
assert body["self_eval_min_samples"] == cfg.proficiency.self_eval_min_samples
for row in body["rows"]:
# All four columns whose absence makes blended_score incomparable.
assert "source" in row
assert "self_eval_samples" in row
assert "outcome_samples" in row
assert "inherited_from" in row
assert "last_updated" in row
def test_inherited_rows_are_marked_as_such(client):
"""An inherited prior is a COPY, and routing weights it like a measurement.
Without this flag the two are the same number in the same column.
"""
c, _cfg = client
rows = {(r["model_id"], r["category"]): r for r in c.get("/admin/api/proficiency").json()["rows"]}
inherited = rows[("kimi-k3-flex", "coding_general")]
measured = rows[("kimi-k3", "coding_general")]
assert inherited["blended_score"] == measured["blended_score"], (
"the fixture is only interesting because the NUMBERS agree"
)
assert inherited["inherited"] is True
assert inherited["inherited_from"] == "kimi-k3"
assert inherited["source"] == "outcome_prior"
assert measured["inherited"] is False
assert measured["inherited_from"] is None
assert measured["source"] == "outcome_blended"
def test_thin_is_computed_against_the_configured_floor(client):
"""A 1.00 at n=2 and a 0.79 at n=14 are not the same claim.
docs_writing read 0.70-1.00 at n=2 with a model at the ceiling, and the
router paid for that ceiling until six more passes spread it 0.66-0.97.
"""
c, cfg = client
rows = {(r["model_id"], r["category"]): r for r in c.get("/admin/api/proficiency").json()["rows"]}
thin = rows[("gemma-4-31b", "coding_general")]
thick = rows[("gemma-4-31b", "docs_writing")]
assert cfg.proficiency.self_eval_min_samples > 2, "fixture assumes a floor above 2"
assert thin["total_samples"] == 2
assert thin["thin"] is True
assert thick["thin"] is False
def test_source_counts_summarize_the_table(client):
c, _cfg = client
counts = c.get("/admin/api/proficiency").json()["source_counts"]
assert counts == {
"outcome_blended": 1,
"outcome_prior": 1,
"self_eval_thin": 1,
"self_eval": 1,
}
def test_there_is_no_write_path_for_proficiency(client):
"""Read-only, deliberately. A hand-edited score is a fabricated measurement.
Same failure as the empty leaderboards.yaml and the provider's
static_fallback carbon constant this project already excludes.
"""
c, _cfg = client
for method in ("POST", "PUT", "PATCH", "DELETE"):
resp = c.request(method, "/admin/api/proficiency")
assert resp.status_code == 405, f"{method} /api/proficiency is routable"
def test_metrics_does_not_import_dispatcher():
"""The separation that keeps these queries usable from /health and the TUI."""
text = (ROOT / "src" / "metrics.py").read_text()
assert "import dispatcher" not in text
# ---------------------------------------------------------------------------
# Client outcomes: the exclusion must not be silent
# ---------------------------------------------------------------------------
def test_outcomes_are_listed_with_their_attribution(client):
c, _cfg = client
outcomes = c.get("/admin/api/proficiency").json()["outcomes"]
assert len(outcomes) == len(SEED_OUTCOMES)
excluded = [o for o in outcomes if not o["model_attributable"]]
assert len(excluded) == 1, (
"a model_attributable = 0 outcome is recorded and then excluded from "
"folding; if the portal cannot show that, the exclusion is invisible"
)
assert excluded[0]["model_id"] == "gemma-4-31b"
def test_outcomes_report_whether_they_have_been_folded_in(client):
c, _cfg = client
outcomes = {
(o["model_id"], o["verdict"]): o
for o in c.get("/admin/api/proficiency").json()["outcomes"]
}
assert outcomes[("kimi-k3", "succeeded")]["applied_at"] is not None
assert outcomes[("kimi-k3", "failed")]["applied_at"] is None
def test_only_client_outcomes_are_listed(client):
"""Structural and local_llm verdicts are diagnostics, not ground truth."""
c, _cfg = client
outcomes = c.get("/admin/api/proficiency").json()["outcomes"]
assert all(o["verdict"] != "malformed" for o in outcomes)
def test_recent_client_outcomes_respects_its_limit(tmp_path):
db_path = tmp_path / "prof.db"
_seed(db_path)
conn = sqlite3.connect(str(db_path))
conn.row_factory = sqlite3.Row
try:
assert len(metrics.recent_client_outcomes(conn, limit=1)) == 1
finally:
conn.close()
# ---------------------------------------------------------------------------
# The page
# ---------------------------------------------------------------------------
def test_page_is_served_and_reachable_from_every_other_page(client):
# Reachability moved client-side with the nav: each page ships the
# empty #nav-links mount and navbar.js renders the list into it, so
# the every-page-links-everywhere invariant is owned by
# tests/test_admin_nav_is_complete.py (navbar.js's NAV_LINKS). Here
# the page contract is just that the mount exists.
c, _cfg = client
resp = c.get("/admin/proficiency")
assert resp.status_code == 200
assert resp.headers["content-type"].startswith("text/html")
for name in ("index", "models", "profiles", "decisions", "controls"):
page = (ROOT / "admin" / "frontend" / f"{name}.html").read_text()
assert 'id="nav-links"' in page, f"no nav-links mount on {name}.html"
def test_page_distinguishes_measured_from_inherited(client):
"""The single most important requirement on this page.
Provenance drives the cell class, and `inherited` gets its own marker on
top of it, because a row can be inherited without being thin and thin
without being inherited.
"""
c, _cfg = client
html = c.get("/admin/proficiency").text
for cls in ("cell-measured", "cell-prior", "cell-thin", "cell-inherited"):
assert f".{cls}" in html, f"{cls} has no style rule"
assert "outcome_blended: 'cell-measured'" in html
assert "outcome_prior: 'cell-prior'" in html
assert "self_eval_thin: 'cell-thin'" in html
# Inheritance OUTRANKS source. A -flex row carries its family's number
# verbatim, source column included, so a row can read outcome_blended and
# still be a copy — colouring that green would paint a copy as a
# measurement, which is the exact failure this page exists to prevent.
assert "row.inherited ? 'cell-prior'" in html
def test_page_offers_no_editing_controls(client):
"""No write path in the backend, and no affordance in the page either."""
c, _cfg = client
html = c.get("/admin/proficiency").text
assert "method: 'POST'" not in html
assert "api/proficiency`, {" not in html