feat(tui): catch the dashboard up to the schema #33

Merged
alee merged 4 commits from feat/tui-overhaul into main 2026-09-05 15:36:21 +00:00
7 changed files with 1016 additions and 5 deletions

View File

@@ -877,7 +877,11 @@ def recent_decisions(
conn: sqlite3.Connection,
limit: int = 50,
) -> List[dict]:
"""Last *N* rows from route_decisions, ordered by id DESC."""
"""Last *N* rows from route_decisions, ordered by id DESC.
The row is the field set the TUI consumes; it is pinned by
tests/test_tui_schema_drift.py.
"""
return [
dict(row)
for row in conn.execute(
@@ -889,6 +893,7 @@ def recent_decisions(
runner_up_models, est_cost_usd, est_proficiency,
rejected_reason, session_key, tools, images, json_mode, streamed,
flex_preference, flex_swapped, flex_forced,
exploration, request_id,
pinch_original_tokens, pinch_final_tokens, profile
FROM route_decisions
ORDER BY id DESC

View File

@@ -24,6 +24,7 @@ from __future__ import annotations
import logging
import os
from datetime import datetime
from typing import Callable, Optional
logger = logging.getLogger(__name__)
@@ -57,6 +58,43 @@ def _fmt_usd(v: Optional[float]) -> str:
return f"{v:.6f}"
def _fmt_runway(hours: float) -> str:
"""Runway as hours below two days, days above. Money-scale, not micro."""
if hours < 48:
return f"~{hours:.1f}h"
return f"~{hours / 24:.1f}d"
def _format_quota_lead(
balance: Optional[float],
burn: Optional[float],
runway_hours: Optional[float],
low_warning: bool,
) -> str:
"""The quota panel's headline: balance, burn rate, projected runway.
Deliberately 2dp rather than ``_fmt_usd``'s 6 — this is money an operator
acts on, not per-request micro-money.
``burn`` and ``runway_hours`` are legitimately ``None`` when the sample
segment is too short to extrapolate (PR #30's guard against a wild rate
right after a top-up). That is a real answer, not missing data, so it
renders "burn n/a" and the caller shows ``runway_note`` alongside. The
literal "None" must never reach the screen.
"""
if balance is None:
return "balance n/a"
parts = [f"balance ${balance:.2f} left"]
parts.append(f"burn ${burn:.2f}/h" if burn is not None else "burn n/a")
if runway_hours is not None:
runway = f"runway {_fmt_runway(float(runway_hours))}"
# Same markup idiom the legend already uses for `resets`.
parts.append(
f"[rgb(200,80,80)]{runway}[/rgb(200,80,80)]" if low_warning else runway
)
return " · ".join(parts)
# Compact bottom keys legend. The 11 BINDINGS are collapsed to these short
# (key, label) pairs — duplicate quit bindings (q/Q/ctrl+c) show once. Rendered
# as a single markup string so Textual's Static wraps rather than truncates.
@@ -95,6 +133,47 @@ def _flex_indicator(r: dict) -> str:
return str(pref)
def _short_time(value: object) -> str:
"""``HH:MM:SS`` in LOCAL time from an ISO-8601 ``observed_at``.
The decision feed is dense and almost always same-day, so the date would
crowd out ``selected`` — the column operators actually read. Local rather
than UTC because this is a live feed watched next to a wall clock.
Returns ``""`` for a missing or unparseable value rather than raising: a
malformed timestamp must not take down the whole table render.
"""
if not value:
return ""
try:
parsed = datetime.fromisoformat(str(value))
except (TypeError, ValueError):
return ""
if parsed.tzinfo is not None:
parsed = parsed.astimezone()
return parsed.strftime("%H:%M:%S")
def _exploration_flag(r: dict) -> str:
"""``E`` when epsilon-greedy picked this row instead of the ranking.
An exploratory pick is a deliberate random sample, NOT the router's
judgement. Without a visible marker an operator reads it as the ranking's
answer and draws the wrong conclusion about model quality.
"""
return "E" if r.get("exploration") else ""
def _flags_indicator(r: dict) -> str:
"""The combined flags cell: flex label and exploration marker.
One cell rather than two columns — the terminal is not wide, and the spec
caps the table at one added column plus one flag.
"""
parts = [part for part in (_flex_indicator(r), _exploration_flag(r)) if part]
return " ".join(parts)
class DashboardApp(App):
"""A terminal dashboard that reads GET /metrics and renders panels."""
@@ -155,6 +234,12 @@ class DashboardApp(App):
#quota-progress {
width: 100%;
}
#quota-lead {
text-style: bold;
}
#quota-note {
color: $text-muted;
}
#quota-legend {
color: $text-muted;
}
@@ -207,6 +292,12 @@ class DashboardApp(App):
yield Static("", id="loading-panel")
yield Static("Quota burn", classes="panel-title")
with Vertical(id="quota-panel"):
# Balance and runway lead; the plan-percentage bar is demoted
# below them. A bar against a routinely-exceeded plan implies
# a ceiling that does not exist — the credit balance is the
# real one (PR #30).
yield Static(markup=True, id="quota-lead")
yield Static(id="quota-note")
yield ProgressBar(id="quota-progress")
yield Static(markup=True, id="quota-legend")
yield Static("Recent decisions (enter = details)", classes="panel-title")
@@ -312,7 +403,16 @@ class DashboardApp(App):
model_table.add_columns("model", "calls", "cost $", "kWh", "gCO2eq")
decision_table = self.query_one("#decision-table", DataTable)
decision_table.add_columns(
"id", "kind", "category", "tier", "ctx", "selected", "est $", "flex"
"time",
"id",
"kind",
"category",
"tier",
"ctx",
"profile",
"selected",
"est $",
"flags",
)
decision_table.cursor_type = "row"
decision_table.zebra_stripes = True
@@ -509,6 +609,37 @@ class DashboardApp(App):
metered = rows.get("metered_kwh_30d")
frac = rows.get("fraction")
calls = rows.get("calls")
# Balance/runway lead. Wired OUTSIDE the plan gate on purpose: the
# credit balance is meaningful whether or not a kWh plan is set, and
# it is the figure that actually stops traffic.
lead = self.query_one("#quota-lead", Static)
note = self.query_one("#quota-note", Static)
if rows:
balance = rows.get("balance_usd")
burn = rows.get("burn_rate_usd_per_hour")
lead.update(
_format_quota_lead(
balance,
burn,
rows.get("projected_hours_remaining"),
bool(rows.get("runway_low_warning")),
)
)
# Explain an absent burn rate rather than leaving a bare "n/a".
if balance is not None and burn is None:
note.update(
str(rows.get("runway_note") or "burn estimate unavailable")
)
note.display = True
else:
note.update("")
note.display = False
else:
lead.update("")
note.update("")
note.display = False
if plan is not None and float(plan) > 0:
bar.total = float(plan)
bar.progress = float(metered or 0)
@@ -590,18 +721,23 @@ class DashboardApp(App):
dt.clear()
for r in self._last_model["recent_decisions"]:
dt.add_row(
_short_time(r.get("observed_at")),
str(r.get("id")),
str(r.get("kind")),
str(r.get("category")),
str(r.get("tier")),
str(r.get("required_context_tokens")),
# `or ""` so an absent profile is blank, never the literal "None".
str(r.get("profile") or ""),
str(r.get("selected")),
_fmt_usd(r.get("est_cost_usd")),
_flex_indicator(r),
_flags_indicator(r),
key=str(r.get("id")),
)
if not self._last_model["recent_decisions"]:
dt.add_row("(no decisions)", "", "", "", "", "", "", "", key="_placeholder")
dt.add_row(
"(no decisions)", "", "", "", "", "", "", "", "", "", key="_placeholder"
)
self._restore_cursor(dt, saved_decision_key)
def _render_category_table(self) -> None:

View File

@@ -143,6 +143,12 @@ def decision_row(r: dict) -> dict:
"flex_preference": r.get("flex_preference"),
"flex_swapped": r.get("flex_swapped"),
"flex_forced": r.get("flex_forced"),
"profile": r.get("profile"),
"exploration": r.get("exploration"),
"pinch_original_tokens": r.get("pinch_original_tokens"),
"pinch_final_tokens": r.get("pinch_final_tokens"),
"request_id": r.get("request_id"),
"session_key": r.get("session_key"),
}

View File

@@ -59,6 +59,9 @@ ROUTE_DECISIONS_COLUMNS = [
"flex_preference",
"flex_swapped",
"flex_forced",
# Enforcement for this list lives in tests/test_tui_schema_drift.py —
# keep new route_decisions columns registered there, not just here.
"request_id",
"exploration",
"pinch_original_tokens",
"pinch_final_tokens",

View File

@@ -11,7 +11,9 @@ file that may. No non-tui module imports textual.
from __future__ import annotations
import asyncio
import copy
import time
from datetime import datetime
import pytest
import requests
@@ -20,7 +22,7 @@ from textual.widgets import ProgressBar, Static
import tui
import tui_model
from tui_model import build_model, fetch_metrics
from tui_screens import VerdictMixScreen
from tui_screens import DecisionDetailScreen, VerdictMixScreen
def _fixture() -> dict:
@@ -144,6 +146,46 @@ def _fixture() -> dict:
}
def _enriched_payload() -> dict:
"""A deep copy of _fixture() whose recent_decisions rows carry the six
T2 data-layer fields, mirroring what /metrics returns after the drift
catch-up. Row 42 is the popup test's target row (exploration off);
row 41 is the exploratory pick; row 40 stays a rejection row."""
data = copy.deepcopy(_fixture())
enrichments = [
{
"profile": "default",
"exploration": 0,
"request_id": "chatcmpl-fixture-42",
"pinch_original_tokens": 120000,
"pinch_final_tokens": 96122,
"session_key": "sess-42",
"observed_at": "2026-08-23T10:00:00+00:00",
},
{
"profile": "default",
"exploration": 1,
"request_id": "chatcmpl-fixture-41",
"pinch_original_tokens": 90000,
"pinch_final_tokens": 87500,
"session_key": "sess-41",
"observed_at": "2026-08-23T09:59:00+00:00",
},
{
"profile": "default",
"exploration": 0,
"request_id": None,
"pinch_original_tokens": None,
"pinch_final_tokens": None,
"session_key": "sess-40",
"observed_at": "2026-08-23T09:58:00+00:00",
},
]
for row, extra in zip(data["recent_decisions"], enrichments):
row.update(extra)
return data
# --------------------------------------------------------------------------
# Direct unit tests of the data layer (no TUI running).
# --------------------------------------------------------------------------
@@ -1056,6 +1098,78 @@ def test_show_decision_detail_pushes_modal_with_full_row():
asyncio.run(_go())
def test_detail_popup_carries_forensics_fields():
"""The e-key popup carries the six T2 forensics fields — profile,
exploration, pinch tokens, request_id, session_key — both on the live
decision dict and in the rendered JSON for a metrics-shaped full row."""
stub = _StubFetcher()
stub.payload = _enriched_payload()
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
async def _go():
async with app.run_test() as pilot:
await pilot.pause()
app.query_one("#decision-table").focus()
await pilot.pause()
await pilot.press("e")
await pilot.pause()
from textual.screen import ModalScreen
assert isinstance(app.screen, ModalScreen)
decision = app.screen.decision
assert decision["request_id"] == "chatcmpl-fixture-42"
assert decision["session_key"] == "sess-42"
assert decision["pinch_original_tokens"] == 120000
assert decision["pinch_final_tokens"] == 96122
assert decision["profile"] == "default"
assert decision["exploration"] == 0
asyncio.run(_go())
metrics_row = {
"id": 43,
"observed_at": "2026-09-05T12:34:56+00:00",
"kind": "chat",
"task_category": "coding_general",
"task_tier": 2,
"required_context_tokens": 123456,
"confidence": 0.91,
"classifier_ms": 1800,
"classification_source": "classifier",
"latency_tolerance": "interactive",
"candidates_considered": 7,
"selected_model": "fixture-model",
"selected_provider": "neuralwatt",
"runner_up_models": None,
"est_cost_usd": 0.001,
"est_proficiency": 0.9,
"rejected_reason": None,
"session_key": "sess-43",
"tools": 0,
"images": 0,
"json_mode": 0,
"streamed": 1,
"flex_preference": "auto",
"flex_swapped": 0,
"flex_forced": 0,
"request_id": "chatcmpl-fixture-43",
"exploration": 1,
"pinch_original_tokens": 120000,
"pinch_final_tokens": 96122,
"profile": "default",
}
text = DecisionDetailScreen(tui_model.decision_row(metrics_row))._render_text()
for key in (
"profile",
"exploration",
"pinch_original_tokens",
"pinch_final_tokens",
"request_id",
"session_key",
):
assert key in text
def test_app_run_test_ctrl_v_opens_verdict_popup():
"""Pressing ``ctrl+v`` pushes a VerdictMixScreen modal showing the verdict
mix rows from the current model, and ``escape`` dismisses it back to the
@@ -1161,6 +1275,46 @@ def test_live_decision_inserts_row_at_front_and_rerenders():
asyncio.run(_go())
def test_live_decision_carries_new_fields():
"""A live SSE decision keeps the six T2 fields through the shared
decision_row projection, so a prompt `e` on a fresh decision shows the
same forensics as a /metrics-sourced row."""
stub = _StubFetcher()
stub.payload = _fixture()
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
async def _go():
async with app.run_test() as pilot:
await pilot.pause()
app._handle_live_decision(
{
"id": 1001,
"kind": "chat",
"task_category": "coding_general",
"task_tier": 2,
"selected_model": "deepseek-v4-flash",
"est_cost_usd": 0.0002,
"profile": "default",
"exploration": 1,
"request_id": "chatcmpl-live-1001",
"pinch_original_tokens": 120000,
"pinch_final_tokens": 96122,
"session_key": "sess-live-1001",
}
)
await pilot.pause()
row = app._last_model["recent_decisions"][0]
assert row["id"] == 1001
assert row["profile"] == "default"
assert row["exploration"] == 1
assert row["request_id"] == "chatcmpl-live-1001"
assert row["pinch_original_tokens"] == 120000
assert row["pinch_final_tokens"] == 96122
assert row["session_key"] == "sess-live-1001"
asyncio.run(_go())
def test_live_decision_caps_recent_decisions_at_fifty():
"""The live feed never grows the in-memory list past the /metrics cap, so
the dashboard's view stays consistent with a /metrics refresh."""
@@ -1404,3 +1558,219 @@ def test_format_quota_legend_shows_next_reset_date():
assert "2026-07-26" in legend # window start still shown
assert "2026-09-06" in legend # billing reset
assert "resets" in legend.lower()
# --------------------------------------------------------------------------
# T3 — decision table display: time column, profile column, exploration flag.
# --------------------------------------------------------------------------
def test_decision_table_columns_and_order():
"""The table declares exactly ten columns with `time` first.
Time leads because it is the natural scan axis for a live feed; `flex`
became `flags` because exploration now shares that cell.
"""
stub = _StubFetcher()
stub.payload = _enriched_payload()
app = tui.DashboardApp(fetcher=stub)
def _assert(a):
dt = a.query_one("#decision-table")
labels = [str(c.label) for c in dt.columns.values()]
assert labels == [
"time",
"id",
"kind",
"category",
"tier",
"ctx",
"profile",
"selected",
"est $",
"flags",
]
_run_app(app, _assert)
def test_short_time_renders_hhmmss_and_degrades_quietly():
"""`_short_time` is HH:MM:SS local, and never raises on bad input.
A malformed timestamp must not take down the whole table render, so the
unparseable cases return "" rather than propagating ValueError.
"""
rendered = tui._short_time("2026-08-23T10:00:00+00:00")
assert len(rendered) == 8 and rendered.count(":") == 2
# Local conversion, computed the same way the helper does it.
expected = (
datetime.fromisoformat("2026-08-23T10:00:00+00:00")
.astimezone()
.strftime("%H:%M:%S")
)
assert rendered == expected
# No date component leaks into the cell.
assert "2026" not in rendered
for bad in (None, "", "not-a-timestamp", 12345):
assert tui._short_time(bad) == ""
def test_app_decision_table_profile_column_blank_not_none():
"""`profile` renders its value, and renders BLANK when absent.
A literal "None" in a dense table reads as a real profile name.
"""
stub = _StubFetcher()
data = _enriched_payload()
data["recent_decisions"][1]["profile"] = None
stub.payload = data
app = tui.DashboardApp(fetcher=stub)
def _assert(a):
dt = a.query_one("#decision-table")
labels = [str(c.label) for c in dt.columns.values()]
profile_at = labels.index("profile")
assert str(dt.get_row_at(0)[profile_at]) == "default"
# Absent profile is an empty cell, never the literal "None".
assert str(dt.get_row_at(1)[profile_at]) == ""
_run_app(app, _assert)
def test_exploration_flag_and_flags_indicator_compose():
"""`E` marks an exploratory pick and composes with the flex label.
An epsilon-greedy pick is a random sample, not the ranking's judgement;
without the marker an operator reads it as the router's answer.
"""
assert tui._exploration_flag({"exploration": 1}) == "E"
assert tui._exploration_flag({"exploration": 0}) == ""
assert tui._exploration_flag({}) == ""
# Composition: flex label first, then the marker, space separated.
both = tui._flags_indicator(
{"flex_preference": "auto", "flex_swapped": 0, "exploration": 1}
)
assert both == "auto E"
# Either alone survives; neither yields an empty cell.
assert tui._flags_indicator({"exploration": 1}) == "E"
assert tui._flags_indicator({"flex_preference": "auto"}) == "auto"
assert tui._flags_indicator({}) == ""
def test_app_decision_table_exploration_flag_renders():
"""Row 41 is the exploratory pick in the fixture; its flags cell shows E."""
stub = _StubFetcher()
stub.payload = _enriched_payload()
app = tui.DashboardApp(fetcher=stub)
def _assert(a):
dt = a.query_one("#decision-table")
exploratory = [str(c) for c in dt.get_row_at(1)]
assert any("E" == cell or cell.endswith(" E") for cell in exploratory)
non_exploratory = [str(c) for c in dt.get_row_at(0)]
assert not any(cell == "E" or cell.endswith(" E") for cell in non_exploratory)
_run_app(app, _assert)
def test_placeholder_row_matches_column_arity():
"""The empty-state placeholder must have exactly ten cells.
A short placeholder row raises inside Textual at render time — an arity
mismatch is a crash, not a cosmetic issue.
"""
stub = _StubFetcher()
data = _enriched_payload()
data["recent_decisions"] = []
stub.payload = data
app = tui.DashboardApp(fetcher=stub)
def _assert(a):
dt = a.query_one("#decision-table")
assert len(dt.get_row_at(0)) == len(dt.columns) == 10
_run_app(app, _assert)
# --------------------------------------------------------------------------
# T4 — quota panel leads with balance and runway.
# --------------------------------------------------------------------------
def test_format_quota_lead_states():
"""All four lead states, including the one this deployment is actually in.
burn/runway are legitimately None when the sample segment is too short
(PR #30). That is deliberate, not missing data, so the None path must
still say something an operator can read.
"""
full = tui._format_quota_lead(8.50, 1.0, 8.5, False)
assert "balance $8.50 left" in full
assert "burn $1.00/h" in full
assert "runway ~8.5h" in full
none_burn = tui._format_quota_lead(8.50, None, None, False)
assert "burn n/a" in none_burn
assert "None" not in none_burn
assert tui._format_quota_lead(None, 1.0, 8.5, False) == "balance n/a"
assert "runway ~20.8d" in tui._format_quota_lead(100.0, 1.0, 500.0, False)
low = tui._format_quota_lead(8.50, 4.0, 1.5, True)
assert "[rgb(200,80,80)]runway ~1.5h" in low
def test_quota_panel_burn_unavailable_renders_note_verbatim():
"""When burn is None the note renders VERBATIM — never blank, 0, or None.
This is the state the live deployment is in right now, so it is the first
thing an operator sees, not an edge case.
"""
stub = _StubFetcher()
data = _fixture()
note = "burn estimate unavailable: segment after last balance increase spans only 9 minutes"
for block in (data["quota"], data["coverage"]["quota"]):
block["balance_usd"] = 7.5
block["burn_rate_usd_per_hour"] = None
block["projected_hours_remaining"] = None
block["runway_low_warning"] = False
block["runway_note"] = note
stub.payload = data
app = tui.DashboardApp(fetcher=stub)
def _assert(a):
lead_text = str(a.query_one("#quota-lead", Static).content)
note_text = str(a.query_one("#quota-note", Static).content)
assert "balance $7.50 left" in lead_text
assert "burn n/a" in lead_text
assert "None" not in lead_text
# The note is passed through unchanged, not summarised.
assert note_text == note
assert note_text.strip() != ""
assert note_text.strip() != "0"
assert note_text.strip() != "None"
_run_app(app, _assert)
def test_quota_panel_note_hidden_when_burn_is_available():
"""With a real burn rate there is nothing to explain, so no note line."""
stub = _StubFetcher()
data = _fixture()
for block in (data["quota"], data["coverage"]["quota"]):
block["balance_usd"] = 12.0
block["burn_rate_usd_per_hour"] = 0.5
block["projected_hours_remaining"] = 24.0
block["runway_low_warning"] = False
block["runway_note"] = None
stub.payload = data
app = tui.DashboardApp(fetcher=stub)
def _assert(a):
lead_text = str(a.query_one("#quota-lead", Static).content)
assert "balance $12.00 left" in lead_text
assert "burn $0.50/h" in lead_text
assert "runway ~24.0h" in lead_text
assert a.query_one("#quota-note", Static).display is False
_run_app(app, _assert)

View File

@@ -0,0 +1,200 @@
"""Schema-drift guard: every route_decisions column reaches the TUI on purpose.
Five-plus columns shipped to ``route_decisions`` across recent merges without
any of them reaching the dashboard, and ``tests/test_route_decisions.py``'s own
hand-maintained column list had ALREADY drifted (missing ``request_id``) — the
drift recurs, so this file is the enforcement, not the catch-up.
Two registries, not a magic diff:
- ``SCHEMA_TO_MODEL_KEYS`` — schema column -> the ``tui_model.decision_row``
key that surfaces it. This is the recorded DECISION about surfacing. Three
keys are intentionally renamed (``task_category`` -> ``category``,
``task_tier`` -> ``tier``, ``selected_model`` -> ``selected``), which is
exactly why the mapping is explicit and never derived by string identity.
- ``UNSURFACED_COLUMNS`` — columns deliberately NOT surfaced, each with a
one-line reason. Empty today.
Every new schema column must land in one of the two registries or
``test_schema_columns_all_have_surfacing_decisions`` fails naming the column;
every registry entry must actually round-trip through ``decision_row`` and the
``/metrics`` SELECT or the other two tests fail naming the key.
Offline: temp SQLite created from ``config/schema.sql``, never the live
router.db (pattern borrowed from tests/test_route_decisions.py).
"""
import sqlite3
from pathlib import Path
import metrics
import tui_model
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
# route_decisions column -> the tui_model.decision_row key that surfaces it.
# Schema order (config/schema.sql route_decisions declaration). Add a new
# schema column here when the TUI should surface it, or to
# UNSURFACED_COLUMNS when it deliberately should not.
SCHEMA_TO_MODEL_KEYS = {
"id": "id",
"observed_at": "observed_at",
"kind": "kind",
"task_category": "category",
"task_tier": "tier",
"required_context_tokens": "required_context_tokens",
"confidence": "confidence",
"classifier_ms": "classifier_ms",
"classification_source": "classification_source",
"latency_tolerance": "latency_tolerance",
"candidates_considered": "candidates_considered",
"selected_model": "selected",
"selected_provider": "selected_provider",
"runner_up_models": "runner_up_models",
"est_cost_usd": "est_cost_usd",
"est_proficiency": "est_proficiency",
"rejected_reason": "rejected_reason",
"session_key": "session_key",
"tools": "tools",
"images": "images",
"json_mode": "json_mode",
"streamed": "streamed",
"flex_preference": "flex_preference",
"flex_swapped": "flex_swapped",
"flex_forced": "flex_forced",
"request_id": "request_id",
"exploration": "exploration",
"pinch_original_tokens": "pinch_original_tokens",
"pinch_final_tokens": "pinch_final_tokens",
"profile": "profile",
}
# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a
# one-line reason (callers must keep the comment). Empty today: every column is
# surfaced via decision_row. A new schema column that lands in NEITHER registry
# fails the drift test.
UNSURFACED_COLUMNS: set[str] = set()
# Every column seeded non-NULL: the single source of truth for the INSERT and
# the round-trip assertions, so the tests prove the column flows, not that
# None flows.
FULL_ROW = {
"id": 1,
"observed_at": "2026-09-05T12:34:56+00:00",
"kind": "chat",
"task_category": "coding_general",
"task_tier": 2,
"required_context_tokens": 123456,
"confidence": 0.91,
"classifier_ms": 1800,
"classification_source": "classifier",
"latency_tolerance": "interactive",
"candidates_considered": 7,
"selected_model": "fixture-model",
"selected_provider": "neuralwatt",
"runner_up_models": '[{"model_id": "runner", "provider": "neuralwatt"}]',
"est_cost_usd": 0.00123,
"est_proficiency": 0.88,
# Schema-legal even for a selected row (no CHECK constraint); non-NULL so the
# round-trip assertion proves the column flows, not that None flows.
"rejected_reason": "fixture rejected reason",
"session_key": "fixture-session-key",
"tools": 0,
"images": 0,
"json_mode": 1,
"streamed": 1,
"flex_preference": "auto",
"flex_swapped": 0,
"flex_forced": 0,
"request_id": "chatcmpl-fixture-42",
"exploration": 1,
"pinch_original_tokens": 120000,
"pinch_final_tokens": 96122,
"profile": "default",
}
def _db_from_schema(tmp_path, row_factory=None) -> sqlite3.Connection:
"""A throwaway DB from config/schema.sql — never the live router.db.
``row_factory`` defaults to None (plain tuples) like the PRAGMA test
needs; the metrics round-trip test passes ``sqlite3.Row`` because
``metrics.recent_decisions`` builds its rows via ``dict(row)``.
"""
conn = sqlite3.connect(tmp_path / "schema-drift.db")
conn.executescript(SCHEMA_SQL)
conn.row_factory = row_factory
return conn
def test_schema_columns_all_have_surfacing_decisions(tmp_path):
"""Every route_decisions column is either surfaced or consciously not."""
conn = _db_from_schema(tmp_path)
schema_cols = {
row[1] for row in conn.execute("PRAGMA table_info(route_decisions)")
}
conn.close()
registered = set(SCHEMA_TO_MODEL_KEYS) | UNSURFACED_COLUMNS
missing = schema_cols - set(SCHEMA_TO_MODEL_KEYS) - UNSURFACED_COLUMNS
extra = registered - schema_cols
assert not missing, (
f"route_decisions column(s) added without a surfacing decision: "
f"{sorted(missing)}. Add each to SCHEMA_TO_MODEL_KEYS if the TUI "
f"should surface it (project it in tui_model.decision_row), or to "
f"UNSURFACED_COLUMNS with a reason comment."
)
assert not extra, (
f"registry registers column(s) that no longer exist in "
f"route_decisions: {sorted(extra)}."
)
def test_decision_row_projects_every_tracked_column():
"""decision_row surfaces every registered column's value faithfully."""
projected = tui_model.decision_row(dict(FULL_ROW))
expected_keys = set(SCHEMA_TO_MODEL_KEYS.values())
assert set(projected) == expected_keys, (
"tui_model.decision_row no longer matches the surfacing registry "
f"(symmetric difference: {sorted(expected_keys ^ set(projected))}).\n"
f" registry values: {sorted(expected_keys)}\n"
f" decision_row keys: {sorted(projected)}"
)
for col, key in SCHEMA_TO_MODEL_KEYS.items():
assert projected[key] == FULL_ROW[col], (
f"route_decisions column {col!r} must surface as decision_row "
f"key {key!r}: got {projected[key]!r}, want {FULL_ROW[col]!r}"
)
def test_metrics_recent_decisions_selects_every_tracked_column(tmp_path):
"""The /metrics path carries every registered column end-to-end.
``request_id`` reaches the TUI ONLY through this SELECT (SSE publishes
before the backfill), so a narrowed SELECT silently erases forensics
fields a few seconds after a live decision arrives.
"""
conn = _db_from_schema(tmp_path, row_factory=sqlite3.Row)
cols = ",".join(FULL_ROW.keys())
placeholders = ",".join("?" * len(FULL_ROW))
conn.execute(
f"INSERT INTO route_decisions ({cols}) VALUES ({placeholders})",
tuple(FULL_ROW.values()),
)
conn.commit()
rows = metrics.recent_decisions(conn, limit=5)
conn.close()
assert rows, "the seeded FULL_ROW must come back from recent_decisions"
projected = tui_model.decision_row(rows[0])
unreachable = []
for col, key in SCHEMA_TO_MODEL_KEYS.items():
if projected.get(key) != FULL_ROW[col]:
unreachable.append(
f"route_decisions column {col!r} does not reach the TUI "
f"through /metrics — is it selected by "
f"metrics.recent_decisions()?"
)
assert not unreachable, "\n".join(unreachable)

291
tests/test_tui_warnings.py Normal file
View File

@@ -0,0 +1,291 @@
"""Tripwire: every warning class /metrics can emit must reach the panel.
The 2026-09-04 vision-ceiling incident was invisible for ~19 hours because a
whole warning family was computed correctly and never displayed. The fix there
was surfacing, not computing — so the guard has to be a test that fails when a
warning class exists but nothing renders it.
Two directions, both load-bearing:
* every REGISTERED class actually fires against a fixture that provokes all of
them — catches the fixture rotting or a message shape changing underneath;
* every EMITTED warning is registered — catches a NEW class being added to
``metrics.scoring_coverage`` without anyone deciding to surface it. This one
fails with the raw warning text, because the whole point is that nobody knew
the class existed.
Deliberately no assertions on warning COUNT: the numbers shift as fixture
semantics evolve, and item 4 of the spec asks for class coverage, not arity.
"""
from __future__ import annotations
import asyncio
import copy
import re
import sqlite3
from datetime import datetime, timedelta, timezone
from pathlib import Path
from types import SimpleNamespace
import pytest
from textual.widgets import Static
import metrics
import tui
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
# The registry. A class here without a matching emitted warning means the
# fixture rotted; an emitted warning without a class here means someone added
# a warning nobody decided to display.
WARN_CLASS_MATCHERS = {
"quota-over-plan": r"metered usage is \d+% of the",
"models-missing-energy": r"\d+/\d+ routable models have no reference-workload",
"models-missing-proficiency": r"\d+/\d+ routable models have no proficiency",
"catalog-stale": r"catalog last polled",
# Fixed-width lookbehind so this cannot swallow the capability warnings,
# which share the "tier N context ceiling" tail.
"demand-ceiling": r"(?<!capable )tier \d+ context ceiling",
"escalation-hazard": r"escalation hazard: tier",
"zero-eligible-models": r"tier \d+ has 0 eligible models",
"capability-subceiling": r"(vision|json_mode)-capable tier",
"rejection-new-pattern": r"new rejection pattern:",
"rejection-rate": r"rejection rate:",
}
CFG = SimpleNamespace(
routing=SimpleNamespace(
allowed_access_levels=["public"],
# scoring_coverage reads `.value` off this — it is an enum in real
# config, so a bare string is not a faithful stand-in.
default_flex_preference=SimpleNamespace(value="auto"),
),
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=None),
escalation=SimpleNamespace(enabled=True),
)
def _model_row(model_id, tier, window, *, vision=0, availability="active"):
return (
model_id,
"neuralwatt",
model_id,
tier,
window,
window,
4096,
0.1,
0.3,
vision,
1,
"standard",
"default",
"full",
"public",
availability,
)
def _seed_chernobyl(conn: sqlite3.Connection) -> None:
"""Seed a catalog and traffic that provoke ALL ten warning classes at once.
Named for what it is: a deliberately maximally-broken deployment. Every
timestamp is relative to now so the rejection windows (1h alert, 24h
baseline) stay valid whenever the suite runs.
"""
now = datetime.now(timezone.utc)
stale = (now - timedelta(days=3)).isoformat()
rows = [
# Covered model: has proficiency AND energy, so the "missing" counts
# come out as 2/3 rather than 3/3.
_model_row("cov-t1", 1, 20000),
# Uncovered: drives both missing- classes; its 5000 window is also the
# tier-2 ceiling behind the escalation hazard.
_model_row("gap-t2", 2, 5000),
# Vision-capable but small: the capability sub-ceiling.
_model_row("vis-t1", 1, 5000, vision=1),
# Deprecated: excluded from routable, so tier 3 has zero eligible.
_model_row("dead-t3", 3, 1000, availability="deprecated"),
]
for row in rows:
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
""",
(*row, stale),
)
conn.execute(
"INSERT INTO proficiency "
"(model_id, provider, category, blended_score, last_updated) "
"VALUES ('cov-t1','neuralwatt','coding_general',0.9,?)",
(now.isoformat(),),
)
# 6.0 kWh against a 6.25 plan = 96% > 80%, firing the quota class. The
# 'seed_reference' task_category is also what marks a model as having
# reference-workload data, so this one row serves both purposes.
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, task_category, observed_at) "
"VALUES ('cov-t1','neuralwatt',6.0,'seed_reference',?)",
((now - timedelta(days=1)).isoformat(),),
)
def _decision(**kw):
cols = {
"kind": "chat",
"task_category": "coding_general",
"task_tier": 1,
"required_context_tokens": 100,
"latency_tolerance": "interactive",
"selected_model": "cov-t1",
"selected_provider": "neuralwatt",
"observed_at": now.isoformat(),
"images": 0,
"json_mode": 0,
"rejected_reason": None,
}
cols.update(kw)
conn.execute(
f"INSERT INTO route_decisions ({','.join(cols)}) "
f"VALUES ({','.join('?' * len(cols))})",
tuple(cols.values()),
)
# One huge vision request: demand-ceiling (20000 < 999999), capability
# sub-ceiling (vision 5000 < 999999), and the escalation hazard all at once.
_decision(required_context_tokens=999999, images=1)
# Two rejections of a pattern with no baseline -> "new rejection pattern:".
for _ in range(2):
_decision(
selected_model=None,
selected_provider=None,
rejected_reason="context >= 99999 tokens; interactive",
)
# A familiar group: one in the baseline window, six in the alert window,
# crossing the min-count threshold -> "rejection rate:".
familiar = "tier >= 2; context >= 30000 tokens; interactive"
_decision(
task_tier=2,
selected_model=None,
selected_provider=None,
rejected_reason=familiar,
observed_at=(now - timedelta(hours=2)).isoformat(),
)
for _ in range(6):
_decision(
task_tier=2,
selected_model=None,
selected_provider=None,
rejected_reason=familiar,
)
conn.commit()
@pytest.fixture()
def chernobyl(tmp_path):
conn = sqlite3.connect(tmp_path / "warn.db")
conn.executescript(SCHEMA_SQL)
conn.row_factory = sqlite3.Row
_seed_chernobyl(conn)
yield conn
conn.close()
@pytest.fixture(autouse=True)
def _no_real_network(monkeypatch):
"""Even if a fetcher is mis-wired, never reach a real router."""
def _guard(base_url):
raise AssertionError(f"real fetch_metrics called with {base_url!r}")
monkeypatch.setattr(tui, "fetch_metrics", _guard)
def _run_app(app: tui.DashboardApp, body) -> None:
async def _go():
async with app.run_test() as pilot:
await pilot.pause()
body(app)
asyncio.run(_go())
def test_every_registered_warning_class_fires(chernobyl):
"""Each registered class is provoked by the fixture.
A failure here means either the fixture rotted or a warning's message
shape changed under the matcher — both silently disable the tripwire.
"""
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
for name, pattern in WARN_CLASS_MATCHERS.items():
assert any(re.search(pattern, w) for w in warnings), (
f"registered warning class {name!r} did not fire.\n"
f"Either the chernobyl fixture no longer provokes it, or the "
f"message shape changed and the matcher needs updating.\n"
f"Pattern: {pattern}\nEmitted:\n " + "\n ".join(warnings)
)
def test_every_emitted_warning_belongs_to_a_registered_class(chernobyl):
"""No warning class exists without a decision about surfacing it.
This is the direction that catches the vision-ceiling failure mode: a new
class added to scoring_coverage that nothing displays. It fails with the
RAW text because the whole point is that nobody knew it existed.
"""
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
unmatched = [
w
for w in warnings
if not any(re.search(p, w) for p in WARN_CLASS_MATCHERS.values())
]
assert not unmatched, (
"a warning class was added to metrics.scoring_coverage without a "
"surfacing decision. Register a matcher in WARN_CLASS_MATCHERS (and a "
"chernobyl seed if it needs one).\nUNMATCHED:\n "
+ "\n ".join(unmatched)
)
def test_registered_warnings_render_in_the_panel(chernobyl):
"""The panel actually shows them — computing is not surfacing.
Drives the real app with the chernobyl warnings so the assertion covers
the render path, not just the metrics call.
"""
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
from tests.test_tui import _fixture # noqa: PLC0415 — shared payload shape
payload = copy.deepcopy(_fixture())
payload["coverage"]["warnings"] = warnings
class _Stub:
def __call__(self, base_url):
return payload
app = tui.DashboardApp(fetcher=_Stub())
def _assert(a):
panel = str(a.query_one("#warnings-panel", Static).content)
for name, pattern in WARN_CLASS_MATCHERS.items():
assert re.search(pattern, panel), (
f"warning class {name!r} is emitted by /metrics but does not "
f"render in #warnings-panel — computed but not surfaced, which "
f"is exactly the 2026-09-04 failure.\nPanel:\n{panel}"
)
_run_app(app, _assert)