308 lines
12 KiB
Python
308 lines
12 KiB
Python
"""GET /admin/api/proficiency and the page that renders it.
|
|
|
|
``proficiency_score`` is the only category-dependent term in the ranking, so
|
|
this table decides routing — and the portal's entire surface for it was a
|
|
top-N list for one category on the dashboard. An operator could see which
|
|
model won a decision and not why.
|
|
|
|
The single load-bearing requirement, restated because it is easy to lose in a
|
|
refactor: a score that traffic EARNED must be distinguishable from one that
|
|
was COPIED IN. ``outcome_prior`` and ``self_eval_thin`` must not look like
|
|
``outcome_blended``. This project already made that mistake once, on ``-flex``
|
|
rows, where an inherited score froze at its first copy while the row it came
|
|
from moved from 0.85 to 0.973.
|
|
|
|
Fixtures are seeded with known answers rather than read off the live DB, so a
|
|
failure here means the code changed and not that the router routed something.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import sqlite3
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
from fastapi import FastAPI
|
|
from starlette.testclient import TestClient
|
|
|
|
import metrics
|
|
from admin import build_router
|
|
from config import load_config
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
|
|
|
# (model_id, category, blended, self_eval_n, outcome_n, source, inherited_from)
|
|
SEED_PROFICIENCY = [
|
|
# Measured here from real traffic: the shape a score should have.
|
|
("kimi-k3", "coding_general", 0.94, 14, 31, "outcome_blended", None),
|
|
# Inherited: a variant carrying its family's number, never measured.
|
|
("kimi-k3-flex", "coding_general", 0.94, 14, 31, "outcome_prior", "kimi-k3"),
|
|
# Benchmark only, and thin — two samples is not a measurement of a ceiling.
|
|
("gemma-4-31b", "coding_general", 1.00, 2, 0, "self_eval_thin", None),
|
|
# Benchmark past the floor.
|
|
("gemma-4-31b", "docs_writing", 0.79, 14, 0, "self_eval", None),
|
|
]
|
|
|
|
SEED_OUTCOMES = [
|
|
# (model, category, verdict, attributable, applied_at)
|
|
("kimi-k3", "coding_general", "succeeded", 1, "2026-09-04T10:00:00+00:00"),
|
|
("kimi-k3", "coding_general", "failed", 1, None),
|
|
# The one an operator could not see: recorded, then excluded from folding
|
|
# because the classification that produced it was degraded.
|
|
("gemma-4-31b", "general_chat", "failed", 0, None),
|
|
]
|
|
|
|
|
|
def _seed(path: Path) -> None:
|
|
conn = sqlite3.connect(str(path))
|
|
conn.executescript(SCHEMA_SQL)
|
|
conn.executemany(
|
|
"""
|
|
INSERT INTO models
|
|
(model_id, provider, availability, access_level, tier,
|
|
context_window, effective_context_window, latency_class,
|
|
last_updated)
|
|
VALUES (?, 'neuralwatt', 'active', 'public', 2, 128000, 128000,
|
|
'standard', '2026-09-01T00:00:00+00:00')
|
|
""",
|
|
[(m,) for m in sorted({r[0] for r in SEED_PROFICIENCY})],
|
|
)
|
|
conn.executemany(
|
|
"""
|
|
INSERT INTO proficiency
|
|
(model_id, provider, category, blended_score, self_eval_score,
|
|
self_eval_samples, outcome_score, outcome_samples, source,
|
|
inherited_from, last_updated)
|
|
VALUES (?, 'neuralwatt', ?, ?, ?, ?, ?, ?, ?, ?,
|
|
'2026-09-01T00:00:00+00:00')
|
|
""",
|
|
[
|
|
(m, c, b, b, sn, b, on, src, inh)
|
|
for (m, c, b, sn, on, src, inh) in SEED_PROFICIENCY
|
|
],
|
|
)
|
|
conn.executemany(
|
|
"""
|
|
INSERT INTO verifications
|
|
(model_id, provider, task_category, kind, verdict, detail,
|
|
observed_at, applied_at, model_attributable)
|
|
VALUES (?, 'neuralwatt', ?, 'client_outcome', ?, 'seeded',
|
|
'2026-09-04T09:00:00+00:00', ?, ?)
|
|
""",
|
|
[(m, c, v, applied, attr) for (m, c, v, attr, applied) in SEED_OUTCOMES],
|
|
)
|
|
# A non-outcome verification, which must NOT appear in the outcomes list.
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO verifications
|
|
(model_id, provider, task_category, kind, verdict, detail,
|
|
observed_at, model_attributable)
|
|
VALUES ('kimi-k3', 'neuralwatt', 'coding_general', 'structural',
|
|
'malformed', 'not a client report',
|
|
'2026-09-04T09:30:00+00:00', 1)
|
|
"""
|
|
)
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
|
|
@pytest.fixture
|
|
def client(tmp_path):
|
|
db_path = tmp_path / "prof.db"
|
|
_seed(db_path)
|
|
cfg = load_config(str(ROOT / "config" / "config.yaml"))
|
|
|
|
def _db():
|
|
conn = sqlite3.connect(str(db_path))
|
|
conn.row_factory = sqlite3.Row
|
|
return conn
|
|
|
|
app = FastAPI()
|
|
app.include_router(build_router(cfg, _db, base_dir=str(ROOT)), prefix="/admin")
|
|
yield TestClient(app), cfg
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# The endpoint
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_endpoint_returns_every_row_with_its_provenance(client):
|
|
c, cfg = client
|
|
body = c.get("/admin/api/proficiency").json()
|
|
|
|
assert len(body["rows"]) == len(SEED_PROFICIENCY)
|
|
assert body["categories"] == list(cfg.proficiency.categories)
|
|
assert body["self_eval_min_samples"] == cfg.proficiency.self_eval_min_samples
|
|
for row in body["rows"]:
|
|
# All four columns whose absence makes blended_score incomparable.
|
|
assert "source" in row
|
|
assert "self_eval_samples" in row
|
|
assert "outcome_samples" in row
|
|
assert "inherited_from" in row
|
|
assert "last_updated" in row
|
|
|
|
|
|
def test_inherited_rows_are_marked_as_such(client):
|
|
"""An inherited prior is a COPY, and routing weights it like a measurement.
|
|
|
|
Without this flag the two are the same number in the same column.
|
|
"""
|
|
c, _cfg = client
|
|
rows = {(r["model_id"], r["category"]): r for r in c.get("/admin/api/proficiency").json()["rows"]}
|
|
|
|
inherited = rows[("kimi-k3-flex", "coding_general")]
|
|
measured = rows[("kimi-k3", "coding_general")]
|
|
|
|
assert inherited["blended_score"] == measured["blended_score"], (
|
|
"the fixture is only interesting because the NUMBERS agree"
|
|
)
|
|
assert inherited["inherited"] is True
|
|
assert inherited["inherited_from"] == "kimi-k3"
|
|
assert inherited["source"] == "outcome_prior"
|
|
assert measured["inherited"] is False
|
|
assert measured["inherited_from"] is None
|
|
assert measured["source"] == "outcome_blended"
|
|
|
|
|
|
def test_thin_is_computed_against_the_configured_floor(client):
|
|
"""A 1.00 at n=2 and a 0.79 at n=14 are not the same claim.
|
|
|
|
docs_writing read 0.70-1.00 at n=2 with a model at the ceiling, and the
|
|
router paid for that ceiling until six more passes spread it 0.66-0.97.
|
|
"""
|
|
c, cfg = client
|
|
rows = {(r["model_id"], r["category"]): r for r in c.get("/admin/api/proficiency").json()["rows"]}
|
|
|
|
thin = rows[("gemma-4-31b", "coding_general")]
|
|
thick = rows[("gemma-4-31b", "docs_writing")]
|
|
|
|
assert cfg.proficiency.self_eval_min_samples > 2, "fixture assumes a floor above 2"
|
|
assert thin["total_samples"] == 2
|
|
assert thin["thin"] is True
|
|
assert thick["thin"] is False
|
|
|
|
|
|
def test_source_counts_summarize_the_table(client):
|
|
c, _cfg = client
|
|
counts = c.get("/admin/api/proficiency").json()["source_counts"]
|
|
assert counts == {
|
|
"outcome_blended": 1,
|
|
"outcome_prior": 1,
|
|
"self_eval_thin": 1,
|
|
"self_eval": 1,
|
|
}
|
|
|
|
|
|
def test_there_is_no_write_path_for_proficiency(client):
|
|
"""Read-only, deliberately. A hand-edited score is a fabricated measurement.
|
|
|
|
Same failure as the empty leaderboards.yaml and the provider's
|
|
static_fallback carbon constant this project already excludes.
|
|
"""
|
|
c, _cfg = client
|
|
for method in ("POST", "PUT", "PATCH", "DELETE"):
|
|
resp = c.request(method, "/admin/api/proficiency")
|
|
assert resp.status_code == 405, f"{method} /api/proficiency is routable"
|
|
|
|
|
|
def test_metrics_does_not_import_dispatcher():
|
|
"""The separation that keeps these queries usable from /health and the TUI."""
|
|
text = (ROOT / "src" / "metrics.py").read_text()
|
|
assert "import dispatcher" not in text
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Client outcomes: the exclusion must not be silent
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_outcomes_are_listed_with_their_attribution(client):
|
|
c, _cfg = client
|
|
outcomes = c.get("/admin/api/proficiency").json()["outcomes"]
|
|
|
|
assert len(outcomes) == len(SEED_OUTCOMES)
|
|
excluded = [o for o in outcomes if not o["model_attributable"]]
|
|
assert len(excluded) == 1, (
|
|
"a model_attributable = 0 outcome is recorded and then excluded from "
|
|
"folding; if the portal cannot show that, the exclusion is invisible"
|
|
)
|
|
assert excluded[0]["model_id"] == "gemma-4-31b"
|
|
|
|
|
|
def test_outcomes_report_whether_they_have_been_folded_in(client):
|
|
c, _cfg = client
|
|
outcomes = {
|
|
(o["model_id"], o["verdict"]): o
|
|
for o in c.get("/admin/api/proficiency").json()["outcomes"]
|
|
}
|
|
assert outcomes[("kimi-k3", "succeeded")]["applied_at"] is not None
|
|
assert outcomes[("kimi-k3", "failed")]["applied_at"] is None
|
|
|
|
|
|
def test_only_client_outcomes_are_listed(client):
|
|
"""Structural and local_llm verdicts are diagnostics, not ground truth."""
|
|
c, _cfg = client
|
|
outcomes = c.get("/admin/api/proficiency").json()["outcomes"]
|
|
assert all(o["verdict"] != "malformed" for o in outcomes)
|
|
|
|
|
|
def test_recent_client_outcomes_respects_its_limit(tmp_path):
|
|
db_path = tmp_path / "prof.db"
|
|
_seed(db_path)
|
|
conn = sqlite3.connect(str(db_path))
|
|
conn.row_factory = sqlite3.Row
|
|
try:
|
|
assert len(metrics.recent_client_outcomes(conn, limit=1)) == 1
|
|
finally:
|
|
conn.close()
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# The page
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_page_is_served_and_reachable_from_every_other_page(client):
|
|
# Reachability moved client-side with the nav: each page ships the
|
|
# empty #nav-links mount and navbar.js renders the list into it, so
|
|
# the every-page-links-everywhere invariant is owned by
|
|
# tests/test_admin_nav_is_complete.py (navbar.js's NAV_LINKS). Here
|
|
# the page contract is just that the mount exists.
|
|
c, _cfg = client
|
|
resp = c.get("/admin/proficiency")
|
|
assert resp.status_code == 200
|
|
assert resp.headers["content-type"].startswith("text/html")
|
|
for name in ("index", "models", "profiles", "decisions", "controls"):
|
|
page = (ROOT / "admin" / "frontend" / f"{name}.html").read_text()
|
|
assert 'id="nav-links"' in page, f"no nav-links mount on {name}.html"
|
|
|
|
|
|
def test_page_distinguishes_measured_from_inherited(client):
|
|
"""The single most important requirement on this page.
|
|
|
|
Provenance drives the cell class, and `inherited` gets its own marker on
|
|
top of it, because a row can be inherited without being thin and thin
|
|
without being inherited.
|
|
"""
|
|
c, _cfg = client
|
|
html = c.get("/admin/proficiency").text
|
|
for cls in ("cell-measured", "cell-prior", "cell-thin", "cell-inherited"):
|
|
assert f".{cls}" in html, f"{cls} has no style rule"
|
|
assert "outcome_blended: 'cell-measured'" in html
|
|
assert "outcome_prior: 'cell-prior'" in html
|
|
assert "self_eval_thin: 'cell-thin'" in html
|
|
# Inheritance OUTRANKS source. A -flex row carries its family's number
|
|
# verbatim, source column included, so a row can read outcome_blended and
|
|
# still be a copy — colouring that green would paint a copy as a
|
|
# measurement, which is the exact failure this page exists to prevent.
|
|
assert "row.inherited ? 'cell-prior'" in html
|
|
|
|
|
|
def test_page_offers_no_editing_controls(client):
|
|
"""No write path in the backend, and no affordance in the page either."""
|
|
c, _cfg = client
|
|
html = c.get("/admin/proficiency").text
|
|
assert "method: 'POST'" not in html
|
|
assert "api/proficiency`, {" not in html
|