feat(tui): catch the dashboard up to the schema #33
@@ -877,7 +877,11 @@ def recent_decisions(
|
|||||||
conn: sqlite3.Connection,
|
conn: sqlite3.Connection,
|
||||||
limit: int = 50,
|
limit: int = 50,
|
||||||
) -> List[dict]:
|
) -> List[dict]:
|
||||||
"""Last *N* rows from route_decisions, ordered by id DESC."""
|
"""Last *N* rows from route_decisions, ordered by id DESC.
|
||||||
|
|
||||||
|
The row is the field set the TUI consumes; it is pinned by
|
||||||
|
tests/test_tui_schema_drift.py.
|
||||||
|
"""
|
||||||
return [
|
return [
|
||||||
dict(row)
|
dict(row)
|
||||||
for row in conn.execute(
|
for row in conn.execute(
|
||||||
@@ -889,6 +893,7 @@ def recent_decisions(
|
|||||||
runner_up_models, est_cost_usd, est_proficiency,
|
runner_up_models, est_cost_usd, est_proficiency,
|
||||||
rejected_reason, session_key, tools, images, json_mode, streamed,
|
rejected_reason, session_key, tools, images, json_mode, streamed,
|
||||||
flex_preference, flex_swapped, flex_forced,
|
flex_preference, flex_swapped, flex_forced,
|
||||||
|
exploration, request_id,
|
||||||
pinch_original_tokens, pinch_final_tokens, profile
|
pinch_original_tokens, pinch_final_tokens, profile
|
||||||
FROM route_decisions
|
FROM route_decisions
|
||||||
ORDER BY id DESC
|
ORDER BY id DESC
|
||||||
|
|||||||
142
src/tui.py
142
src/tui.py
@@ -24,6 +24,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
from datetime import datetime
|
||||||
from typing import Callable, Optional
|
from typing import Callable, Optional
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
@@ -57,6 +58,43 @@ def _fmt_usd(v: Optional[float]) -> str:
|
|||||||
return f"{v:.6f}"
|
return f"{v:.6f}"
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_runway(hours: float) -> str:
|
||||||
|
"""Runway as hours below two days, days above. Money-scale, not micro."""
|
||||||
|
if hours < 48:
|
||||||
|
return f"~{hours:.1f}h"
|
||||||
|
return f"~{hours / 24:.1f}d"
|
||||||
|
|
||||||
|
|
||||||
|
def _format_quota_lead(
|
||||||
|
balance: Optional[float],
|
||||||
|
burn: Optional[float],
|
||||||
|
runway_hours: Optional[float],
|
||||||
|
low_warning: bool,
|
||||||
|
) -> str:
|
||||||
|
"""The quota panel's headline: balance, burn rate, projected runway.
|
||||||
|
|
||||||
|
Deliberately 2dp rather than ``_fmt_usd``'s 6 — this is money an operator
|
||||||
|
acts on, not per-request micro-money.
|
||||||
|
|
||||||
|
``burn`` and ``runway_hours`` are legitimately ``None`` when the sample
|
||||||
|
segment is too short to extrapolate (PR #30's guard against a wild rate
|
||||||
|
right after a top-up). That is a real answer, not missing data, so it
|
||||||
|
renders "burn n/a" and the caller shows ``runway_note`` alongside. The
|
||||||
|
literal "None" must never reach the screen.
|
||||||
|
"""
|
||||||
|
if balance is None:
|
||||||
|
return "balance n/a"
|
||||||
|
parts = [f"balance ${balance:.2f} left"]
|
||||||
|
parts.append(f"burn ${burn:.2f}/h" if burn is not None else "burn n/a")
|
||||||
|
if runway_hours is not None:
|
||||||
|
runway = f"runway {_fmt_runway(float(runway_hours))}"
|
||||||
|
# Same markup idiom the legend already uses for `resets`.
|
||||||
|
parts.append(
|
||||||
|
f"[rgb(200,80,80)]{runway}[/rgb(200,80,80)]" if low_warning else runway
|
||||||
|
)
|
||||||
|
return " · ".join(parts)
|
||||||
|
|
||||||
|
|
||||||
# Compact bottom keys legend. The 11 BINDINGS are collapsed to these short
|
# Compact bottom keys legend. The 11 BINDINGS are collapsed to these short
|
||||||
# (key, label) pairs — duplicate quit bindings (q/Q/ctrl+c) show once. Rendered
|
# (key, label) pairs — duplicate quit bindings (q/Q/ctrl+c) show once. Rendered
|
||||||
# as a single markup string so Textual's Static wraps rather than truncates.
|
# as a single markup string so Textual's Static wraps rather than truncates.
|
||||||
@@ -95,6 +133,47 @@ def _flex_indicator(r: dict) -> str:
|
|||||||
return str(pref)
|
return str(pref)
|
||||||
|
|
||||||
|
|
||||||
|
def _short_time(value: object) -> str:
|
||||||
|
"""``HH:MM:SS`` in LOCAL time from an ISO-8601 ``observed_at``.
|
||||||
|
|
||||||
|
The decision feed is dense and almost always same-day, so the date would
|
||||||
|
crowd out ``selected`` — the column operators actually read. Local rather
|
||||||
|
than UTC because this is a live feed watched next to a wall clock.
|
||||||
|
|
||||||
|
Returns ``""`` for a missing or unparseable value rather than raising: a
|
||||||
|
malformed timestamp must not take down the whole table render.
|
||||||
|
"""
|
||||||
|
if not value:
|
||||||
|
return ""
|
||||||
|
try:
|
||||||
|
parsed = datetime.fromisoformat(str(value))
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return ""
|
||||||
|
if parsed.tzinfo is not None:
|
||||||
|
parsed = parsed.astimezone()
|
||||||
|
return parsed.strftime("%H:%M:%S")
|
||||||
|
|
||||||
|
|
||||||
|
def _exploration_flag(r: dict) -> str:
|
||||||
|
"""``E`` when epsilon-greedy picked this row instead of the ranking.
|
||||||
|
|
||||||
|
An exploratory pick is a deliberate random sample, NOT the router's
|
||||||
|
judgement. Without a visible marker an operator reads it as the ranking's
|
||||||
|
answer and draws the wrong conclusion about model quality.
|
||||||
|
"""
|
||||||
|
return "E" if r.get("exploration") else ""
|
||||||
|
|
||||||
|
|
||||||
|
def _flags_indicator(r: dict) -> str:
|
||||||
|
"""The combined flags cell: flex label and exploration marker.
|
||||||
|
|
||||||
|
One cell rather than two columns — the terminal is not wide, and the spec
|
||||||
|
caps the table at one added column plus one flag.
|
||||||
|
"""
|
||||||
|
parts = [part for part in (_flex_indicator(r), _exploration_flag(r)) if part]
|
||||||
|
return " ".join(parts)
|
||||||
|
|
||||||
|
|
||||||
class DashboardApp(App):
|
class DashboardApp(App):
|
||||||
"""A terminal dashboard that reads GET /metrics and renders panels."""
|
"""A terminal dashboard that reads GET /metrics and renders panels."""
|
||||||
|
|
||||||
@@ -155,6 +234,12 @@ class DashboardApp(App):
|
|||||||
#quota-progress {
|
#quota-progress {
|
||||||
width: 100%;
|
width: 100%;
|
||||||
}
|
}
|
||||||
|
#quota-lead {
|
||||||
|
text-style: bold;
|
||||||
|
}
|
||||||
|
#quota-note {
|
||||||
|
color: $text-muted;
|
||||||
|
}
|
||||||
#quota-legend {
|
#quota-legend {
|
||||||
color: $text-muted;
|
color: $text-muted;
|
||||||
}
|
}
|
||||||
@@ -207,6 +292,12 @@ class DashboardApp(App):
|
|||||||
yield Static("", id="loading-panel")
|
yield Static("", id="loading-panel")
|
||||||
yield Static("Quota burn", classes="panel-title")
|
yield Static("Quota burn", classes="panel-title")
|
||||||
with Vertical(id="quota-panel"):
|
with Vertical(id="quota-panel"):
|
||||||
|
# Balance and runway lead; the plan-percentage bar is demoted
|
||||||
|
# below them. A bar against a routinely-exceeded plan implies
|
||||||
|
# a ceiling that does not exist — the credit balance is the
|
||||||
|
# real one (PR #30).
|
||||||
|
yield Static(markup=True, id="quota-lead")
|
||||||
|
yield Static(id="quota-note")
|
||||||
yield ProgressBar(id="quota-progress")
|
yield ProgressBar(id="quota-progress")
|
||||||
yield Static(markup=True, id="quota-legend")
|
yield Static(markup=True, id="quota-legend")
|
||||||
yield Static("Recent decisions (enter = details)", classes="panel-title")
|
yield Static("Recent decisions (enter = details)", classes="panel-title")
|
||||||
@@ -312,7 +403,16 @@ class DashboardApp(App):
|
|||||||
model_table.add_columns("model", "calls", "cost $", "kWh", "gCO2eq")
|
model_table.add_columns("model", "calls", "cost $", "kWh", "gCO2eq")
|
||||||
decision_table = self.query_one("#decision-table", DataTable)
|
decision_table = self.query_one("#decision-table", DataTable)
|
||||||
decision_table.add_columns(
|
decision_table.add_columns(
|
||||||
"id", "kind", "category", "tier", "ctx", "selected", "est $", "flex"
|
"time",
|
||||||
|
"id",
|
||||||
|
"kind",
|
||||||
|
"category",
|
||||||
|
"tier",
|
||||||
|
"ctx",
|
||||||
|
"profile",
|
||||||
|
"selected",
|
||||||
|
"est $",
|
||||||
|
"flags",
|
||||||
)
|
)
|
||||||
decision_table.cursor_type = "row"
|
decision_table.cursor_type = "row"
|
||||||
decision_table.zebra_stripes = True
|
decision_table.zebra_stripes = True
|
||||||
@@ -509,6 +609,37 @@ class DashboardApp(App):
|
|||||||
metered = rows.get("metered_kwh_30d")
|
metered = rows.get("metered_kwh_30d")
|
||||||
frac = rows.get("fraction")
|
frac = rows.get("fraction")
|
||||||
calls = rows.get("calls")
|
calls = rows.get("calls")
|
||||||
|
|
||||||
|
# Balance/runway lead. Wired OUTSIDE the plan gate on purpose: the
|
||||||
|
# credit balance is meaningful whether or not a kWh plan is set, and
|
||||||
|
# it is the figure that actually stops traffic.
|
||||||
|
lead = self.query_one("#quota-lead", Static)
|
||||||
|
note = self.query_one("#quota-note", Static)
|
||||||
|
if rows:
|
||||||
|
balance = rows.get("balance_usd")
|
||||||
|
burn = rows.get("burn_rate_usd_per_hour")
|
||||||
|
lead.update(
|
||||||
|
_format_quota_lead(
|
||||||
|
balance,
|
||||||
|
burn,
|
||||||
|
rows.get("projected_hours_remaining"),
|
||||||
|
bool(rows.get("runway_low_warning")),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
# Explain an absent burn rate rather than leaving a bare "n/a".
|
||||||
|
if balance is not None and burn is None:
|
||||||
|
note.update(
|
||||||
|
str(rows.get("runway_note") or "burn estimate unavailable")
|
||||||
|
)
|
||||||
|
note.display = True
|
||||||
|
else:
|
||||||
|
note.update("")
|
||||||
|
note.display = False
|
||||||
|
else:
|
||||||
|
lead.update("")
|
||||||
|
note.update("")
|
||||||
|
note.display = False
|
||||||
|
|
||||||
if plan is not None and float(plan) > 0:
|
if plan is not None and float(plan) > 0:
|
||||||
bar.total = float(plan)
|
bar.total = float(plan)
|
||||||
bar.progress = float(metered or 0)
|
bar.progress = float(metered or 0)
|
||||||
@@ -590,18 +721,23 @@ class DashboardApp(App):
|
|||||||
dt.clear()
|
dt.clear()
|
||||||
for r in self._last_model["recent_decisions"]:
|
for r in self._last_model["recent_decisions"]:
|
||||||
dt.add_row(
|
dt.add_row(
|
||||||
|
_short_time(r.get("observed_at")),
|
||||||
str(r.get("id")),
|
str(r.get("id")),
|
||||||
str(r.get("kind")),
|
str(r.get("kind")),
|
||||||
str(r.get("category")),
|
str(r.get("category")),
|
||||||
str(r.get("tier")),
|
str(r.get("tier")),
|
||||||
str(r.get("required_context_tokens")),
|
str(r.get("required_context_tokens")),
|
||||||
|
# `or ""` so an absent profile is blank, never the literal "None".
|
||||||
|
str(r.get("profile") or ""),
|
||||||
str(r.get("selected")),
|
str(r.get("selected")),
|
||||||
_fmt_usd(r.get("est_cost_usd")),
|
_fmt_usd(r.get("est_cost_usd")),
|
||||||
_flex_indicator(r),
|
_flags_indicator(r),
|
||||||
key=str(r.get("id")),
|
key=str(r.get("id")),
|
||||||
)
|
)
|
||||||
if not self._last_model["recent_decisions"]:
|
if not self._last_model["recent_decisions"]:
|
||||||
dt.add_row("(no decisions)", "", "", "", "", "", "", "", key="_placeholder")
|
dt.add_row(
|
||||||
|
"(no decisions)", "", "", "", "", "", "", "", "", "", key="_placeholder"
|
||||||
|
)
|
||||||
self._restore_cursor(dt, saved_decision_key)
|
self._restore_cursor(dt, saved_decision_key)
|
||||||
|
|
||||||
def _render_category_table(self) -> None:
|
def _render_category_table(self) -> None:
|
||||||
|
|||||||
@@ -143,6 +143,12 @@ def decision_row(r: dict) -> dict:
|
|||||||
"flex_preference": r.get("flex_preference"),
|
"flex_preference": r.get("flex_preference"),
|
||||||
"flex_swapped": r.get("flex_swapped"),
|
"flex_swapped": r.get("flex_swapped"),
|
||||||
"flex_forced": r.get("flex_forced"),
|
"flex_forced": r.get("flex_forced"),
|
||||||
|
"profile": r.get("profile"),
|
||||||
|
"exploration": r.get("exploration"),
|
||||||
|
"pinch_original_tokens": r.get("pinch_original_tokens"),
|
||||||
|
"pinch_final_tokens": r.get("pinch_final_tokens"),
|
||||||
|
"request_id": r.get("request_id"),
|
||||||
|
"session_key": r.get("session_key"),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -59,6 +59,9 @@ ROUTE_DECISIONS_COLUMNS = [
|
|||||||
"flex_preference",
|
"flex_preference",
|
||||||
"flex_swapped",
|
"flex_swapped",
|
||||||
"flex_forced",
|
"flex_forced",
|
||||||
|
# Enforcement for this list lives in tests/test_tui_schema_drift.py —
|
||||||
|
# keep new route_decisions columns registered there, not just here.
|
||||||
|
"request_id",
|
||||||
"exploration",
|
"exploration",
|
||||||
"pinch_original_tokens",
|
"pinch_original_tokens",
|
||||||
"pinch_final_tokens",
|
"pinch_final_tokens",
|
||||||
|
|||||||
@@ -11,7 +11,9 @@ file that may. No non-tui module imports textual.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
|
import copy
|
||||||
import time
|
import time
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
import requests
|
import requests
|
||||||
@@ -20,7 +22,7 @@ from textual.widgets import ProgressBar, Static
|
|||||||
import tui
|
import tui
|
||||||
import tui_model
|
import tui_model
|
||||||
from tui_model import build_model, fetch_metrics
|
from tui_model import build_model, fetch_metrics
|
||||||
from tui_screens import VerdictMixScreen
|
from tui_screens import DecisionDetailScreen, VerdictMixScreen
|
||||||
|
|
||||||
|
|
||||||
def _fixture() -> dict:
|
def _fixture() -> dict:
|
||||||
@@ -144,6 +146,46 @@ def _fixture() -> dict:
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _enriched_payload() -> dict:
|
||||||
|
"""A deep copy of _fixture() whose recent_decisions rows carry the six
|
||||||
|
T2 data-layer fields, mirroring what /metrics returns after the drift
|
||||||
|
catch-up. Row 42 is the popup test's target row (exploration off);
|
||||||
|
row 41 is the exploratory pick; row 40 stays a rejection row."""
|
||||||
|
data = copy.deepcopy(_fixture())
|
||||||
|
enrichments = [
|
||||||
|
{
|
||||||
|
"profile": "default",
|
||||||
|
"exploration": 0,
|
||||||
|
"request_id": "chatcmpl-fixture-42",
|
||||||
|
"pinch_original_tokens": 120000,
|
||||||
|
"pinch_final_tokens": 96122,
|
||||||
|
"session_key": "sess-42",
|
||||||
|
"observed_at": "2026-08-23T10:00:00+00:00",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"profile": "default",
|
||||||
|
"exploration": 1,
|
||||||
|
"request_id": "chatcmpl-fixture-41",
|
||||||
|
"pinch_original_tokens": 90000,
|
||||||
|
"pinch_final_tokens": 87500,
|
||||||
|
"session_key": "sess-41",
|
||||||
|
"observed_at": "2026-08-23T09:59:00+00:00",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"profile": "default",
|
||||||
|
"exploration": 0,
|
||||||
|
"request_id": None,
|
||||||
|
"pinch_original_tokens": None,
|
||||||
|
"pinch_final_tokens": None,
|
||||||
|
"session_key": "sess-40",
|
||||||
|
"observed_at": "2026-08-23T09:58:00+00:00",
|
||||||
|
},
|
||||||
|
]
|
||||||
|
for row, extra in zip(data["recent_decisions"], enrichments):
|
||||||
|
row.update(extra)
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
# --------------------------------------------------------------------------
|
# --------------------------------------------------------------------------
|
||||||
# Direct unit tests of the data layer (no TUI running).
|
# Direct unit tests of the data layer (no TUI running).
|
||||||
# --------------------------------------------------------------------------
|
# --------------------------------------------------------------------------
|
||||||
@@ -1056,6 +1098,78 @@ def test_show_decision_detail_pushes_modal_with_full_row():
|
|||||||
asyncio.run(_go())
|
asyncio.run(_go())
|
||||||
|
|
||||||
|
|
||||||
|
def test_detail_popup_carries_forensics_fields():
|
||||||
|
"""The e-key popup carries the six T2 forensics fields — profile,
|
||||||
|
exploration, pinch tokens, request_id, session_key — both on the live
|
||||||
|
decision dict and in the rendered JSON for a metrics-shaped full row."""
|
||||||
|
stub = _StubFetcher()
|
||||||
|
stub.payload = _enriched_payload()
|
||||||
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
||||||
|
|
||||||
|
async def _go():
|
||||||
|
async with app.run_test() as pilot:
|
||||||
|
await pilot.pause()
|
||||||
|
app.query_one("#decision-table").focus()
|
||||||
|
await pilot.pause()
|
||||||
|
await pilot.press("e")
|
||||||
|
await pilot.pause()
|
||||||
|
from textual.screen import ModalScreen
|
||||||
|
|
||||||
|
assert isinstance(app.screen, ModalScreen)
|
||||||
|
decision = app.screen.decision
|
||||||
|
assert decision["request_id"] == "chatcmpl-fixture-42"
|
||||||
|
assert decision["session_key"] == "sess-42"
|
||||||
|
assert decision["pinch_original_tokens"] == 120000
|
||||||
|
assert decision["pinch_final_tokens"] == 96122
|
||||||
|
assert decision["profile"] == "default"
|
||||||
|
assert decision["exploration"] == 0
|
||||||
|
|
||||||
|
asyncio.run(_go())
|
||||||
|
|
||||||
|
metrics_row = {
|
||||||
|
"id": 43,
|
||||||
|
"observed_at": "2026-09-05T12:34:56+00:00",
|
||||||
|
"kind": "chat",
|
||||||
|
"task_category": "coding_general",
|
||||||
|
"task_tier": 2,
|
||||||
|
"required_context_tokens": 123456,
|
||||||
|
"confidence": 0.91,
|
||||||
|
"classifier_ms": 1800,
|
||||||
|
"classification_source": "classifier",
|
||||||
|
"latency_tolerance": "interactive",
|
||||||
|
"candidates_considered": 7,
|
||||||
|
"selected_model": "fixture-model",
|
||||||
|
"selected_provider": "neuralwatt",
|
||||||
|
"runner_up_models": None,
|
||||||
|
"est_cost_usd": 0.001,
|
||||||
|
"est_proficiency": 0.9,
|
||||||
|
"rejected_reason": None,
|
||||||
|
"session_key": "sess-43",
|
||||||
|
"tools": 0,
|
||||||
|
"images": 0,
|
||||||
|
"json_mode": 0,
|
||||||
|
"streamed": 1,
|
||||||
|
"flex_preference": "auto",
|
||||||
|
"flex_swapped": 0,
|
||||||
|
"flex_forced": 0,
|
||||||
|
"request_id": "chatcmpl-fixture-43",
|
||||||
|
"exploration": 1,
|
||||||
|
"pinch_original_tokens": 120000,
|
||||||
|
"pinch_final_tokens": 96122,
|
||||||
|
"profile": "default",
|
||||||
|
}
|
||||||
|
text = DecisionDetailScreen(tui_model.decision_row(metrics_row))._render_text()
|
||||||
|
for key in (
|
||||||
|
"profile",
|
||||||
|
"exploration",
|
||||||
|
"pinch_original_tokens",
|
||||||
|
"pinch_final_tokens",
|
||||||
|
"request_id",
|
||||||
|
"session_key",
|
||||||
|
):
|
||||||
|
assert key in text
|
||||||
|
|
||||||
|
|
||||||
def test_app_run_test_ctrl_v_opens_verdict_popup():
|
def test_app_run_test_ctrl_v_opens_verdict_popup():
|
||||||
"""Pressing ``ctrl+v`` pushes a VerdictMixScreen modal showing the verdict
|
"""Pressing ``ctrl+v`` pushes a VerdictMixScreen modal showing the verdict
|
||||||
mix rows from the current model, and ``escape`` dismisses it back to the
|
mix rows from the current model, and ``escape`` dismisses it back to the
|
||||||
@@ -1161,6 +1275,46 @@ def test_live_decision_inserts_row_at_front_and_rerenders():
|
|||||||
asyncio.run(_go())
|
asyncio.run(_go())
|
||||||
|
|
||||||
|
|
||||||
|
def test_live_decision_carries_new_fields():
|
||||||
|
"""A live SSE decision keeps the six T2 fields through the shared
|
||||||
|
decision_row projection, so a prompt `e` on a fresh decision shows the
|
||||||
|
same forensics as a /metrics-sourced row."""
|
||||||
|
stub = _StubFetcher()
|
||||||
|
stub.payload = _fixture()
|
||||||
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
||||||
|
|
||||||
|
async def _go():
|
||||||
|
async with app.run_test() as pilot:
|
||||||
|
await pilot.pause()
|
||||||
|
app._handle_live_decision(
|
||||||
|
{
|
||||||
|
"id": 1001,
|
||||||
|
"kind": "chat",
|
||||||
|
"task_category": "coding_general",
|
||||||
|
"task_tier": 2,
|
||||||
|
"selected_model": "deepseek-v4-flash",
|
||||||
|
"est_cost_usd": 0.0002,
|
||||||
|
"profile": "default",
|
||||||
|
"exploration": 1,
|
||||||
|
"request_id": "chatcmpl-live-1001",
|
||||||
|
"pinch_original_tokens": 120000,
|
||||||
|
"pinch_final_tokens": 96122,
|
||||||
|
"session_key": "sess-live-1001",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
await pilot.pause()
|
||||||
|
row = app._last_model["recent_decisions"][0]
|
||||||
|
assert row["id"] == 1001
|
||||||
|
assert row["profile"] == "default"
|
||||||
|
assert row["exploration"] == 1
|
||||||
|
assert row["request_id"] == "chatcmpl-live-1001"
|
||||||
|
assert row["pinch_original_tokens"] == 120000
|
||||||
|
assert row["pinch_final_tokens"] == 96122
|
||||||
|
assert row["session_key"] == "sess-live-1001"
|
||||||
|
|
||||||
|
asyncio.run(_go())
|
||||||
|
|
||||||
|
|
||||||
def test_live_decision_caps_recent_decisions_at_fifty():
|
def test_live_decision_caps_recent_decisions_at_fifty():
|
||||||
"""The live feed never grows the in-memory list past the /metrics cap, so
|
"""The live feed never grows the in-memory list past the /metrics cap, so
|
||||||
the dashboard's view stays consistent with a /metrics refresh."""
|
the dashboard's view stays consistent with a /metrics refresh."""
|
||||||
@@ -1404,3 +1558,219 @@ def test_format_quota_legend_shows_next_reset_date():
|
|||||||
assert "2026-07-26" in legend # window start still shown
|
assert "2026-07-26" in legend # window start still shown
|
||||||
assert "2026-09-06" in legend # billing reset
|
assert "2026-09-06" in legend # billing reset
|
||||||
assert "resets" in legend.lower()
|
assert "resets" in legend.lower()
|
||||||
|
|
||||||
|
|
||||||
|
# --------------------------------------------------------------------------
|
||||||
|
# T3 — decision table display: time column, profile column, exploration flag.
|
||||||
|
# --------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_decision_table_columns_and_order():
|
||||||
|
"""The table declares exactly ten columns with `time` first.
|
||||||
|
|
||||||
|
Time leads because it is the natural scan axis for a live feed; `flex`
|
||||||
|
became `flags` because exploration now shares that cell.
|
||||||
|
"""
|
||||||
|
stub = _StubFetcher()
|
||||||
|
stub.payload = _enriched_payload()
|
||||||
|
app = tui.DashboardApp(fetcher=stub)
|
||||||
|
|
||||||
|
def _assert(a):
|
||||||
|
dt = a.query_one("#decision-table")
|
||||||
|
labels = [str(c.label) for c in dt.columns.values()]
|
||||||
|
assert labels == [
|
||||||
|
"time",
|
||||||
|
"id",
|
||||||
|
"kind",
|
||||||
|
"category",
|
||||||
|
"tier",
|
||||||
|
"ctx",
|
||||||
|
"profile",
|
||||||
|
"selected",
|
||||||
|
"est $",
|
||||||
|
"flags",
|
||||||
|
]
|
||||||
|
|
||||||
|
_run_app(app, _assert)
|
||||||
|
|
||||||
|
|
||||||
|
def test_short_time_renders_hhmmss_and_degrades_quietly():
|
||||||
|
"""`_short_time` is HH:MM:SS local, and never raises on bad input.
|
||||||
|
|
||||||
|
A malformed timestamp must not take down the whole table render, so the
|
||||||
|
unparseable cases return "" rather than propagating ValueError.
|
||||||
|
"""
|
||||||
|
rendered = tui._short_time("2026-08-23T10:00:00+00:00")
|
||||||
|
assert len(rendered) == 8 and rendered.count(":") == 2
|
||||||
|
# Local conversion, computed the same way the helper does it.
|
||||||
|
expected = (
|
||||||
|
datetime.fromisoformat("2026-08-23T10:00:00+00:00")
|
||||||
|
.astimezone()
|
||||||
|
.strftime("%H:%M:%S")
|
||||||
|
)
|
||||||
|
assert rendered == expected
|
||||||
|
# No date component leaks into the cell.
|
||||||
|
assert "2026" not in rendered
|
||||||
|
for bad in (None, "", "not-a-timestamp", 12345):
|
||||||
|
assert tui._short_time(bad) == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_app_decision_table_profile_column_blank_not_none():
|
||||||
|
"""`profile` renders its value, and renders BLANK when absent.
|
||||||
|
|
||||||
|
A literal "None" in a dense table reads as a real profile name.
|
||||||
|
"""
|
||||||
|
stub = _StubFetcher()
|
||||||
|
data = _enriched_payload()
|
||||||
|
data["recent_decisions"][1]["profile"] = None
|
||||||
|
stub.payload = data
|
||||||
|
app = tui.DashboardApp(fetcher=stub)
|
||||||
|
|
||||||
|
def _assert(a):
|
||||||
|
dt = a.query_one("#decision-table")
|
||||||
|
labels = [str(c.label) for c in dt.columns.values()]
|
||||||
|
profile_at = labels.index("profile")
|
||||||
|
assert str(dt.get_row_at(0)[profile_at]) == "default"
|
||||||
|
# Absent profile is an empty cell, never the literal "None".
|
||||||
|
assert str(dt.get_row_at(1)[profile_at]) == ""
|
||||||
|
|
||||||
|
_run_app(app, _assert)
|
||||||
|
|
||||||
|
|
||||||
|
def test_exploration_flag_and_flags_indicator_compose():
|
||||||
|
"""`E` marks an exploratory pick and composes with the flex label.
|
||||||
|
|
||||||
|
An epsilon-greedy pick is a random sample, not the ranking's judgement;
|
||||||
|
without the marker an operator reads it as the router's answer.
|
||||||
|
"""
|
||||||
|
assert tui._exploration_flag({"exploration": 1}) == "E"
|
||||||
|
assert tui._exploration_flag({"exploration": 0}) == ""
|
||||||
|
assert tui._exploration_flag({}) == ""
|
||||||
|
# Composition: flex label first, then the marker, space separated.
|
||||||
|
both = tui._flags_indicator(
|
||||||
|
{"flex_preference": "auto", "flex_swapped": 0, "exploration": 1}
|
||||||
|
)
|
||||||
|
assert both == "auto E"
|
||||||
|
# Either alone survives; neither yields an empty cell.
|
||||||
|
assert tui._flags_indicator({"exploration": 1}) == "E"
|
||||||
|
assert tui._flags_indicator({"flex_preference": "auto"}) == "auto"
|
||||||
|
assert tui._flags_indicator({}) == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_app_decision_table_exploration_flag_renders():
|
||||||
|
"""Row 41 is the exploratory pick in the fixture; its flags cell shows E."""
|
||||||
|
stub = _StubFetcher()
|
||||||
|
stub.payload = _enriched_payload()
|
||||||
|
app = tui.DashboardApp(fetcher=stub)
|
||||||
|
|
||||||
|
def _assert(a):
|
||||||
|
dt = a.query_one("#decision-table")
|
||||||
|
exploratory = [str(c) for c in dt.get_row_at(1)]
|
||||||
|
assert any("E" == cell or cell.endswith(" E") for cell in exploratory)
|
||||||
|
non_exploratory = [str(c) for c in dt.get_row_at(0)]
|
||||||
|
assert not any(cell == "E" or cell.endswith(" E") for cell in non_exploratory)
|
||||||
|
|
||||||
|
_run_app(app, _assert)
|
||||||
|
|
||||||
|
|
||||||
|
def test_placeholder_row_matches_column_arity():
|
||||||
|
"""The empty-state placeholder must have exactly ten cells.
|
||||||
|
|
||||||
|
A short placeholder row raises inside Textual at render time — an arity
|
||||||
|
mismatch is a crash, not a cosmetic issue.
|
||||||
|
"""
|
||||||
|
stub = _StubFetcher()
|
||||||
|
data = _enriched_payload()
|
||||||
|
data["recent_decisions"] = []
|
||||||
|
stub.payload = data
|
||||||
|
app = tui.DashboardApp(fetcher=stub)
|
||||||
|
|
||||||
|
def _assert(a):
|
||||||
|
dt = a.query_one("#decision-table")
|
||||||
|
assert len(dt.get_row_at(0)) == len(dt.columns) == 10
|
||||||
|
|
||||||
|
_run_app(app, _assert)
|
||||||
|
|
||||||
|
|
||||||
|
# --------------------------------------------------------------------------
|
||||||
|
# T4 — quota panel leads with balance and runway.
|
||||||
|
# --------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_format_quota_lead_states():
|
||||||
|
"""All four lead states, including the one this deployment is actually in.
|
||||||
|
|
||||||
|
burn/runway are legitimately None when the sample segment is too short
|
||||||
|
(PR #30). That is deliberate, not missing data, so the None path must
|
||||||
|
still say something an operator can read.
|
||||||
|
"""
|
||||||
|
full = tui._format_quota_lead(8.50, 1.0, 8.5, False)
|
||||||
|
assert "balance $8.50 left" in full
|
||||||
|
assert "burn $1.00/h" in full
|
||||||
|
assert "runway ~8.5h" in full
|
||||||
|
|
||||||
|
none_burn = tui._format_quota_lead(8.50, None, None, False)
|
||||||
|
assert "burn n/a" in none_burn
|
||||||
|
assert "None" not in none_burn
|
||||||
|
|
||||||
|
assert tui._format_quota_lead(None, 1.0, 8.5, False) == "balance n/a"
|
||||||
|
assert "runway ~20.8d" in tui._format_quota_lead(100.0, 1.0, 500.0, False)
|
||||||
|
|
||||||
|
low = tui._format_quota_lead(8.50, 4.0, 1.5, True)
|
||||||
|
assert "[rgb(200,80,80)]runway ~1.5h" in low
|
||||||
|
|
||||||
|
|
||||||
|
def test_quota_panel_burn_unavailable_renders_note_verbatim():
|
||||||
|
"""When burn is None the note renders VERBATIM — never blank, 0, or None.
|
||||||
|
|
||||||
|
This is the state the live deployment is in right now, so it is the first
|
||||||
|
thing an operator sees, not an edge case.
|
||||||
|
"""
|
||||||
|
stub = _StubFetcher()
|
||||||
|
data = _fixture()
|
||||||
|
note = "burn estimate unavailable: segment after last balance increase spans only 9 minutes"
|
||||||
|
for block in (data["quota"], data["coverage"]["quota"]):
|
||||||
|
block["balance_usd"] = 7.5
|
||||||
|
block["burn_rate_usd_per_hour"] = None
|
||||||
|
block["projected_hours_remaining"] = None
|
||||||
|
block["runway_low_warning"] = False
|
||||||
|
block["runway_note"] = note
|
||||||
|
stub.payload = data
|
||||||
|
app = tui.DashboardApp(fetcher=stub)
|
||||||
|
|
||||||
|
def _assert(a):
|
||||||
|
lead_text = str(a.query_one("#quota-lead", Static).content)
|
||||||
|
note_text = str(a.query_one("#quota-note", Static).content)
|
||||||
|
assert "balance $7.50 left" in lead_text
|
||||||
|
assert "burn n/a" in lead_text
|
||||||
|
assert "None" not in lead_text
|
||||||
|
# The note is passed through unchanged, not summarised.
|
||||||
|
assert note_text == note
|
||||||
|
assert note_text.strip() != ""
|
||||||
|
assert note_text.strip() != "0"
|
||||||
|
assert note_text.strip() != "None"
|
||||||
|
|
||||||
|
_run_app(app, _assert)
|
||||||
|
|
||||||
|
|
||||||
|
def test_quota_panel_note_hidden_when_burn_is_available():
|
||||||
|
"""With a real burn rate there is nothing to explain, so no note line."""
|
||||||
|
stub = _StubFetcher()
|
||||||
|
data = _fixture()
|
||||||
|
for block in (data["quota"], data["coverage"]["quota"]):
|
||||||
|
block["balance_usd"] = 12.0
|
||||||
|
block["burn_rate_usd_per_hour"] = 0.5
|
||||||
|
block["projected_hours_remaining"] = 24.0
|
||||||
|
block["runway_low_warning"] = False
|
||||||
|
block["runway_note"] = None
|
||||||
|
stub.payload = data
|
||||||
|
app = tui.DashboardApp(fetcher=stub)
|
||||||
|
|
||||||
|
def _assert(a):
|
||||||
|
lead_text = str(a.query_one("#quota-lead", Static).content)
|
||||||
|
assert "balance $12.00 left" in lead_text
|
||||||
|
assert "burn $0.50/h" in lead_text
|
||||||
|
assert "runway ~24.0h" in lead_text
|
||||||
|
assert a.query_one("#quota-note", Static).display is False
|
||||||
|
|
||||||
|
_run_app(app, _assert)
|
||||||
|
|||||||
200
tests/test_tui_schema_drift.py
Normal file
200
tests/test_tui_schema_drift.py
Normal file
@@ -0,0 +1,200 @@
|
|||||||
|
"""Schema-drift guard: every route_decisions column reaches the TUI on purpose.
|
||||||
|
|
||||||
|
Five-plus columns shipped to ``route_decisions`` across recent merges without
|
||||||
|
any of them reaching the dashboard, and ``tests/test_route_decisions.py``'s own
|
||||||
|
hand-maintained column list had ALREADY drifted (missing ``request_id``) — the
|
||||||
|
drift recurs, so this file is the enforcement, not the catch-up.
|
||||||
|
|
||||||
|
Two registries, not a magic diff:
|
||||||
|
|
||||||
|
- ``SCHEMA_TO_MODEL_KEYS`` — schema column -> the ``tui_model.decision_row``
|
||||||
|
key that surfaces it. This is the recorded DECISION about surfacing. Three
|
||||||
|
keys are intentionally renamed (``task_category`` -> ``category``,
|
||||||
|
``task_tier`` -> ``tier``, ``selected_model`` -> ``selected``), which is
|
||||||
|
exactly why the mapping is explicit and never derived by string identity.
|
||||||
|
- ``UNSURFACED_COLUMNS`` — columns deliberately NOT surfaced, each with a
|
||||||
|
one-line reason. Empty today.
|
||||||
|
|
||||||
|
Every new schema column must land in one of the two registries or
|
||||||
|
``test_schema_columns_all_have_surfacing_decisions`` fails naming the column;
|
||||||
|
every registry entry must actually round-trip through ``decision_row`` and the
|
||||||
|
``/metrics`` SELECT or the other two tests fail naming the key.
|
||||||
|
|
||||||
|
Offline: temp SQLite created from ``config/schema.sql``, never the live
|
||||||
|
router.db (pattern borrowed from tests/test_route_decisions.py).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import sqlite3
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import metrics
|
||||||
|
import tui_model
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
||||||
|
|
||||||
|
# route_decisions column -> the tui_model.decision_row key that surfaces it.
|
||||||
|
# Schema order (config/schema.sql route_decisions declaration). Add a new
|
||||||
|
# schema column here when the TUI should surface it, or to
|
||||||
|
# UNSURFACED_COLUMNS when it deliberately should not.
|
||||||
|
SCHEMA_TO_MODEL_KEYS = {
|
||||||
|
"id": "id",
|
||||||
|
"observed_at": "observed_at",
|
||||||
|
"kind": "kind",
|
||||||
|
"task_category": "category",
|
||||||
|
"task_tier": "tier",
|
||||||
|
"required_context_tokens": "required_context_tokens",
|
||||||
|
"confidence": "confidence",
|
||||||
|
"classifier_ms": "classifier_ms",
|
||||||
|
"classification_source": "classification_source",
|
||||||
|
"latency_tolerance": "latency_tolerance",
|
||||||
|
"candidates_considered": "candidates_considered",
|
||||||
|
"selected_model": "selected",
|
||||||
|
"selected_provider": "selected_provider",
|
||||||
|
"runner_up_models": "runner_up_models",
|
||||||
|
"est_cost_usd": "est_cost_usd",
|
||||||
|
"est_proficiency": "est_proficiency",
|
||||||
|
"rejected_reason": "rejected_reason",
|
||||||
|
"session_key": "session_key",
|
||||||
|
"tools": "tools",
|
||||||
|
"images": "images",
|
||||||
|
"json_mode": "json_mode",
|
||||||
|
"streamed": "streamed",
|
||||||
|
"flex_preference": "flex_preference",
|
||||||
|
"flex_swapped": "flex_swapped",
|
||||||
|
"flex_forced": "flex_forced",
|
||||||
|
"request_id": "request_id",
|
||||||
|
"exploration": "exploration",
|
||||||
|
"pinch_original_tokens": "pinch_original_tokens",
|
||||||
|
"pinch_final_tokens": "pinch_final_tokens",
|
||||||
|
"profile": "profile",
|
||||||
|
}
|
||||||
|
# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a
|
||||||
|
# one-line reason (callers must keep the comment). Empty today: every column is
|
||||||
|
# surfaced via decision_row. A new schema column that lands in NEITHER registry
|
||||||
|
# fails the drift test.
|
||||||
|
UNSURFACED_COLUMNS: set[str] = set()
|
||||||
|
|
||||||
|
# Every column seeded non-NULL: the single source of truth for the INSERT and
|
||||||
|
# the round-trip assertions, so the tests prove the column flows, not that
|
||||||
|
# None flows.
|
||||||
|
FULL_ROW = {
|
||||||
|
"id": 1,
|
||||||
|
"observed_at": "2026-09-05T12:34:56+00:00",
|
||||||
|
"kind": "chat",
|
||||||
|
"task_category": "coding_general",
|
||||||
|
"task_tier": 2,
|
||||||
|
"required_context_tokens": 123456,
|
||||||
|
"confidence": 0.91,
|
||||||
|
"classifier_ms": 1800,
|
||||||
|
"classification_source": "classifier",
|
||||||
|
"latency_tolerance": "interactive",
|
||||||
|
"candidates_considered": 7,
|
||||||
|
"selected_model": "fixture-model",
|
||||||
|
"selected_provider": "neuralwatt",
|
||||||
|
"runner_up_models": '[{"model_id": "runner", "provider": "neuralwatt"}]',
|
||||||
|
"est_cost_usd": 0.00123,
|
||||||
|
"est_proficiency": 0.88,
|
||||||
|
# Schema-legal even for a selected row (no CHECK constraint); non-NULL so the
|
||||||
|
# round-trip assertion proves the column flows, not that None flows.
|
||||||
|
"rejected_reason": "fixture rejected reason",
|
||||||
|
"session_key": "fixture-session-key",
|
||||||
|
"tools": 0,
|
||||||
|
"images": 0,
|
||||||
|
"json_mode": 1,
|
||||||
|
"streamed": 1,
|
||||||
|
"flex_preference": "auto",
|
||||||
|
"flex_swapped": 0,
|
||||||
|
"flex_forced": 0,
|
||||||
|
"request_id": "chatcmpl-fixture-42",
|
||||||
|
"exploration": 1,
|
||||||
|
"pinch_original_tokens": 120000,
|
||||||
|
"pinch_final_tokens": 96122,
|
||||||
|
"profile": "default",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _db_from_schema(tmp_path, row_factory=None) -> sqlite3.Connection:
|
||||||
|
"""A throwaway DB from config/schema.sql — never the live router.db.
|
||||||
|
|
||||||
|
``row_factory`` defaults to None (plain tuples) like the PRAGMA test
|
||||||
|
needs; the metrics round-trip test passes ``sqlite3.Row`` because
|
||||||
|
``metrics.recent_decisions`` builds its rows via ``dict(row)``.
|
||||||
|
"""
|
||||||
|
conn = sqlite3.connect(tmp_path / "schema-drift.db")
|
||||||
|
conn.executescript(SCHEMA_SQL)
|
||||||
|
conn.row_factory = row_factory
|
||||||
|
return conn
|
||||||
|
|
||||||
|
|
||||||
|
def test_schema_columns_all_have_surfacing_decisions(tmp_path):
|
||||||
|
"""Every route_decisions column is either surfaced or consciously not."""
|
||||||
|
conn = _db_from_schema(tmp_path)
|
||||||
|
schema_cols = {
|
||||||
|
row[1] for row in conn.execute("PRAGMA table_info(route_decisions)")
|
||||||
|
}
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
registered = set(SCHEMA_TO_MODEL_KEYS) | UNSURFACED_COLUMNS
|
||||||
|
missing = schema_cols - set(SCHEMA_TO_MODEL_KEYS) - UNSURFACED_COLUMNS
|
||||||
|
extra = registered - schema_cols
|
||||||
|
assert not missing, (
|
||||||
|
f"route_decisions column(s) added without a surfacing decision: "
|
||||||
|
f"{sorted(missing)}. Add each to SCHEMA_TO_MODEL_KEYS if the TUI "
|
||||||
|
f"should surface it (project it in tui_model.decision_row), or to "
|
||||||
|
f"UNSURFACED_COLUMNS with a reason comment."
|
||||||
|
)
|
||||||
|
assert not extra, (
|
||||||
|
f"registry registers column(s) that no longer exist in "
|
||||||
|
f"route_decisions: {sorted(extra)}."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_decision_row_projects_every_tracked_column():
|
||||||
|
"""decision_row surfaces every registered column's value faithfully."""
|
||||||
|
projected = tui_model.decision_row(dict(FULL_ROW))
|
||||||
|
|
||||||
|
expected_keys = set(SCHEMA_TO_MODEL_KEYS.values())
|
||||||
|
assert set(projected) == expected_keys, (
|
||||||
|
"tui_model.decision_row no longer matches the surfacing registry "
|
||||||
|
f"(symmetric difference: {sorted(expected_keys ^ set(projected))}).\n"
|
||||||
|
f" registry values: {sorted(expected_keys)}\n"
|
||||||
|
f" decision_row keys: {sorted(projected)}"
|
||||||
|
)
|
||||||
|
for col, key in SCHEMA_TO_MODEL_KEYS.items():
|
||||||
|
assert projected[key] == FULL_ROW[col], (
|
||||||
|
f"route_decisions column {col!r} must surface as decision_row "
|
||||||
|
f"key {key!r}: got {projected[key]!r}, want {FULL_ROW[col]!r}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_metrics_recent_decisions_selects_every_tracked_column(tmp_path):
|
||||||
|
"""The /metrics path carries every registered column end-to-end.
|
||||||
|
|
||||||
|
``request_id`` reaches the TUI ONLY through this SELECT (SSE publishes
|
||||||
|
before the backfill), so a narrowed SELECT silently erases forensics
|
||||||
|
fields a few seconds after a live decision arrives.
|
||||||
|
"""
|
||||||
|
conn = _db_from_schema(tmp_path, row_factory=sqlite3.Row)
|
||||||
|
cols = ",".join(FULL_ROW.keys())
|
||||||
|
placeholders = ",".join("?" * len(FULL_ROW))
|
||||||
|
conn.execute(
|
||||||
|
f"INSERT INTO route_decisions ({cols}) VALUES ({placeholders})",
|
||||||
|
tuple(FULL_ROW.values()),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
rows = metrics.recent_decisions(conn, limit=5)
|
||||||
|
conn.close()
|
||||||
|
assert rows, "the seeded FULL_ROW must come back from recent_decisions"
|
||||||
|
projected = tui_model.decision_row(rows[0])
|
||||||
|
|
||||||
|
unreachable = []
|
||||||
|
for col, key in SCHEMA_TO_MODEL_KEYS.items():
|
||||||
|
if projected.get(key) != FULL_ROW[col]:
|
||||||
|
unreachable.append(
|
||||||
|
f"route_decisions column {col!r} does not reach the TUI "
|
||||||
|
f"through /metrics — is it selected by "
|
||||||
|
f"metrics.recent_decisions()?"
|
||||||
|
)
|
||||||
|
assert not unreachable, "\n".join(unreachable)
|
||||||
291
tests/test_tui_warnings.py
Normal file
291
tests/test_tui_warnings.py
Normal file
@@ -0,0 +1,291 @@
|
|||||||
|
"""Tripwire: every warning class /metrics can emit must reach the panel.
|
||||||
|
|
||||||
|
The 2026-09-04 vision-ceiling incident was invisible for ~19 hours because a
|
||||||
|
whole warning family was computed correctly and never displayed. The fix there
|
||||||
|
was surfacing, not computing — so the guard has to be a test that fails when a
|
||||||
|
warning class exists but nothing renders it.
|
||||||
|
|
||||||
|
Two directions, both load-bearing:
|
||||||
|
|
||||||
|
* every REGISTERED class actually fires against a fixture that provokes all of
|
||||||
|
them — catches the fixture rotting or a message shape changing underneath;
|
||||||
|
* every EMITTED warning is registered — catches a NEW class being added to
|
||||||
|
``metrics.scoring_coverage`` without anyone deciding to surface it. This one
|
||||||
|
fails with the raw warning text, because the whole point is that nobody knew
|
||||||
|
the class existed.
|
||||||
|
|
||||||
|
Deliberately no assertions on warning COUNT: the numbers shift as fixture
|
||||||
|
semantics evolve, and item 4 of the spec asks for class coverage, not arity.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import copy
|
||||||
|
import re
|
||||||
|
import sqlite3
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from textual.widgets import Static
|
||||||
|
|
||||||
|
import metrics
|
||||||
|
import tui
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
||||||
|
|
||||||
|
|
||||||
|
# The registry. A class here without a matching emitted warning means the
|
||||||
|
# fixture rotted; an emitted warning without a class here means someone added
|
||||||
|
# a warning nobody decided to display.
|
||||||
|
WARN_CLASS_MATCHERS = {
|
||||||
|
"quota-over-plan": r"metered usage is \d+% of the",
|
||||||
|
"models-missing-energy": r"\d+/\d+ routable models have no reference-workload",
|
||||||
|
"models-missing-proficiency": r"\d+/\d+ routable models have no proficiency",
|
||||||
|
"catalog-stale": r"catalog last polled",
|
||||||
|
# Fixed-width lookbehind so this cannot swallow the capability warnings,
|
||||||
|
# which share the "tier N context ceiling" tail.
|
||||||
|
"demand-ceiling": r"(?<!capable )tier \d+ context ceiling",
|
||||||
|
"escalation-hazard": r"escalation hazard: tier",
|
||||||
|
"zero-eligible-models": r"tier \d+ has 0 eligible models",
|
||||||
|
"capability-subceiling": r"(vision|json_mode)-capable tier",
|
||||||
|
"rejection-new-pattern": r"new rejection pattern:",
|
||||||
|
"rejection-rate": r"rejection rate:",
|
||||||
|
}
|
||||||
|
|
||||||
|
CFG = SimpleNamespace(
|
||||||
|
routing=SimpleNamespace(
|
||||||
|
allowed_access_levels=["public"],
|
||||||
|
# scoring_coverage reads `.value` off this — it is an enum in real
|
||||||
|
# config, so a bare string is not a faithful stand-in.
|
||||||
|
default_flex_preference=SimpleNamespace(value="auto"),
|
||||||
|
),
|
||||||
|
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=None),
|
||||||
|
escalation=SimpleNamespace(enabled=True),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _model_row(model_id, tier, window, *, vision=0, availability="active"):
|
||||||
|
return (
|
||||||
|
model_id,
|
||||||
|
"neuralwatt",
|
||||||
|
model_id,
|
||||||
|
tier,
|
||||||
|
window,
|
||||||
|
window,
|
||||||
|
4096,
|
||||||
|
0.1,
|
||||||
|
0.3,
|
||||||
|
vision,
|
||||||
|
1,
|
||||||
|
"standard",
|
||||||
|
"default",
|
||||||
|
"full",
|
||||||
|
"public",
|
||||||
|
availability,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _seed_chernobyl(conn: sqlite3.Connection) -> None:
|
||||||
|
"""Seed a catalog and traffic that provoke ALL ten warning classes at once.
|
||||||
|
|
||||||
|
Named for what it is: a deliberately maximally-broken deployment. Every
|
||||||
|
timestamp is relative to now so the rejection windows (1h alert, 24h
|
||||||
|
baseline) stay valid whenever the suite runs.
|
||||||
|
"""
|
||||||
|
now = datetime.now(timezone.utc)
|
||||||
|
stale = (now - timedelta(days=3)).isoformat()
|
||||||
|
|
||||||
|
rows = [
|
||||||
|
# Covered model: has proficiency AND energy, so the "missing" counts
|
||||||
|
# come out as 2/3 rather than 3/3.
|
||||||
|
_model_row("cov-t1", 1, 20000),
|
||||||
|
# Uncovered: drives both missing- classes; its 5000 window is also the
|
||||||
|
# tier-2 ceiling behind the escalation hazard.
|
||||||
|
_model_row("gap-t2", 2, 5000),
|
||||||
|
# Vision-capable but small: the capability sub-ceiling.
|
||||||
|
_model_row("vis-t1", 1, 5000, vision=1),
|
||||||
|
# Deprecated: excluded from routable, so tier 3 has zero eligible.
|
||||||
|
_model_row("dead-t3", 3, 1000, availability="deprecated"),
|
||||||
|
]
|
||||||
|
for row in rows:
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
INSERT INTO models (
|
||||||
|
model_id, provider, base_model_id, tier, context_window,
|
||||||
|
effective_context_window, max_output_tokens,
|
||||||
|
cost_per_1m_prompt, cost_per_1m_completion,
|
||||||
|
supports_vision, supports_json_mode,
|
||||||
|
latency_class, reasoning_mode, context_variant,
|
||||||
|
access_level, availability, last_updated
|
||||||
|
) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
|
||||||
|
""",
|
||||||
|
(*row, stale),
|
||||||
|
)
|
||||||
|
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO proficiency "
|
||||||
|
"(model_id, provider, category, blended_score, last_updated) "
|
||||||
|
"VALUES ('cov-t1','neuralwatt','coding_general',0.9,?)",
|
||||||
|
(now.isoformat(),),
|
||||||
|
)
|
||||||
|
# 6.0 kWh against a 6.25 plan = 96% > 80%, firing the quota class. The
|
||||||
|
# 'seed_reference' task_category is also what marks a model as having
|
||||||
|
# reference-workload data, so this one row serves both purposes.
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO energy_observations "
|
||||||
|
"(model_id, provider, energy_kwh, task_category, observed_at) "
|
||||||
|
"VALUES ('cov-t1','neuralwatt',6.0,'seed_reference',?)",
|
||||||
|
((now - timedelta(days=1)).isoformat(),),
|
||||||
|
)
|
||||||
|
|
||||||
|
def _decision(**kw):
|
||||||
|
cols = {
|
||||||
|
"kind": "chat",
|
||||||
|
"task_category": "coding_general",
|
||||||
|
"task_tier": 1,
|
||||||
|
"required_context_tokens": 100,
|
||||||
|
"latency_tolerance": "interactive",
|
||||||
|
"selected_model": "cov-t1",
|
||||||
|
"selected_provider": "neuralwatt",
|
||||||
|
"observed_at": now.isoformat(),
|
||||||
|
"images": 0,
|
||||||
|
"json_mode": 0,
|
||||||
|
"rejected_reason": None,
|
||||||
|
}
|
||||||
|
cols.update(kw)
|
||||||
|
conn.execute(
|
||||||
|
f"INSERT INTO route_decisions ({','.join(cols)}) "
|
||||||
|
f"VALUES ({','.join('?' * len(cols))})",
|
||||||
|
tuple(cols.values()),
|
||||||
|
)
|
||||||
|
|
||||||
|
# One huge vision request: demand-ceiling (20000 < 999999), capability
|
||||||
|
# sub-ceiling (vision 5000 < 999999), and the escalation hazard all at once.
|
||||||
|
_decision(required_context_tokens=999999, images=1)
|
||||||
|
|
||||||
|
# Two rejections of a pattern with no baseline -> "new rejection pattern:".
|
||||||
|
for _ in range(2):
|
||||||
|
_decision(
|
||||||
|
selected_model=None,
|
||||||
|
selected_provider=None,
|
||||||
|
rejected_reason="context >= 99999 tokens; interactive",
|
||||||
|
)
|
||||||
|
|
||||||
|
# A familiar group: one in the baseline window, six in the alert window,
|
||||||
|
# crossing the min-count threshold -> "rejection rate:".
|
||||||
|
familiar = "tier >= 2; context >= 30000 tokens; interactive"
|
||||||
|
_decision(
|
||||||
|
task_tier=2,
|
||||||
|
selected_model=None,
|
||||||
|
selected_provider=None,
|
||||||
|
rejected_reason=familiar,
|
||||||
|
observed_at=(now - timedelta(hours=2)).isoformat(),
|
||||||
|
)
|
||||||
|
for _ in range(6):
|
||||||
|
_decision(
|
||||||
|
task_tier=2,
|
||||||
|
selected_model=None,
|
||||||
|
selected_provider=None,
|
||||||
|
rejected_reason=familiar,
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture()
|
||||||
|
def chernobyl(tmp_path):
|
||||||
|
conn = sqlite3.connect(tmp_path / "warn.db")
|
||||||
|
conn.executescript(SCHEMA_SQL)
|
||||||
|
conn.row_factory = sqlite3.Row
|
||||||
|
_seed_chernobyl(conn)
|
||||||
|
yield conn
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def _no_real_network(monkeypatch):
|
||||||
|
"""Even if a fetcher is mis-wired, never reach a real router."""
|
||||||
|
|
||||||
|
def _guard(base_url):
|
||||||
|
raise AssertionError(f"real fetch_metrics called with {base_url!r}")
|
||||||
|
|
||||||
|
monkeypatch.setattr(tui, "fetch_metrics", _guard)
|
||||||
|
|
||||||
|
|
||||||
|
def _run_app(app: tui.DashboardApp, body) -> None:
|
||||||
|
async def _go():
|
||||||
|
async with app.run_test() as pilot:
|
||||||
|
await pilot.pause()
|
||||||
|
body(app)
|
||||||
|
|
||||||
|
asyncio.run(_go())
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_registered_warning_class_fires(chernobyl):
|
||||||
|
"""Each registered class is provoked by the fixture.
|
||||||
|
|
||||||
|
A failure here means either the fixture rotted or a warning's message
|
||||||
|
shape changed under the matcher — both silently disable the tripwire.
|
||||||
|
"""
|
||||||
|
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
|
||||||
|
for name, pattern in WARN_CLASS_MATCHERS.items():
|
||||||
|
assert any(re.search(pattern, w) for w in warnings), (
|
||||||
|
f"registered warning class {name!r} did not fire.\n"
|
||||||
|
f"Either the chernobyl fixture no longer provokes it, or the "
|
||||||
|
f"message shape changed and the matcher needs updating.\n"
|
||||||
|
f"Pattern: {pattern}\nEmitted:\n " + "\n ".join(warnings)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_emitted_warning_belongs_to_a_registered_class(chernobyl):
|
||||||
|
"""No warning class exists without a decision about surfacing it.
|
||||||
|
|
||||||
|
This is the direction that catches the vision-ceiling failure mode: a new
|
||||||
|
class added to scoring_coverage that nothing displays. It fails with the
|
||||||
|
RAW text because the whole point is that nobody knew it existed.
|
||||||
|
"""
|
||||||
|
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
|
||||||
|
unmatched = [
|
||||||
|
w
|
||||||
|
for w in warnings
|
||||||
|
if not any(re.search(p, w) for p in WARN_CLASS_MATCHERS.values())
|
||||||
|
]
|
||||||
|
assert not unmatched, (
|
||||||
|
"a warning class was added to metrics.scoring_coverage without a "
|
||||||
|
"surfacing decision. Register a matcher in WARN_CLASS_MATCHERS (and a "
|
||||||
|
"chernobyl seed if it needs one).\nUNMATCHED:\n "
|
||||||
|
+ "\n ".join(unmatched)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_registered_warnings_render_in_the_panel(chernobyl):
|
||||||
|
"""The panel actually shows them — computing is not surfacing.
|
||||||
|
|
||||||
|
Drives the real app with the chernobyl warnings so the assertion covers
|
||||||
|
the render path, not just the metrics call.
|
||||||
|
"""
|
||||||
|
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
|
||||||
|
|
||||||
|
from tests.test_tui import _fixture # noqa: PLC0415 — shared payload shape
|
||||||
|
|
||||||
|
payload = copy.deepcopy(_fixture())
|
||||||
|
payload["coverage"]["warnings"] = warnings
|
||||||
|
|
||||||
|
class _Stub:
|
||||||
|
def __call__(self, base_url):
|
||||||
|
return payload
|
||||||
|
|
||||||
|
app = tui.DashboardApp(fetcher=_Stub())
|
||||||
|
|
||||||
|
def _assert(a):
|
||||||
|
panel = str(a.query_one("#warnings-panel", Static).content)
|
||||||
|
for name, pattern in WARN_CLASS_MATCHERS.items():
|
||||||
|
assert re.search(pattern, panel), (
|
||||||
|
f"warning class {name!r} is emitted by /metrics but does not "
|
||||||
|
f"render in #warnings-panel — computed but not surfaced, which "
|
||||||
|
f"is exactly the 2026-09-04 failure.\nPanel:\n{panel}"
|
||||||
|
)
|
||||||
|
|
||||||
|
_run_app(app, _assert)
|
||||||
Reference in New Issue
Block a user