feat(tui): catch the dashboard up to the schema #33
@@ -877,7 +877,11 @@ def recent_decisions(
|
||||
conn: sqlite3.Connection,
|
||||
limit: int = 50,
|
||||
) -> List[dict]:
|
||||
"""Last *N* rows from route_decisions, ordered by id DESC."""
|
||||
"""Last *N* rows from route_decisions, ordered by id DESC.
|
||||
|
||||
The row is the field set the TUI consumes; it is pinned by
|
||||
tests/test_tui_schema_drift.py.
|
||||
"""
|
||||
return [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
@@ -889,6 +893,7 @@ def recent_decisions(
|
||||
runner_up_models, est_cost_usd, est_proficiency,
|
||||
rejected_reason, session_key, tools, images, json_mode, streamed,
|
||||
flex_preference, flex_swapped, flex_forced,
|
||||
exploration, request_id,
|
||||
pinch_original_tokens, pinch_final_tokens, profile
|
||||
FROM route_decisions
|
||||
ORDER BY id DESC
|
||||
|
||||
142
src/tui.py
142
src/tui.py
@@ -24,6 +24,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime
|
||||
from typing import Callable, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -57,6 +58,43 @@ def _fmt_usd(v: Optional[float]) -> str:
|
||||
return f"{v:.6f}"
|
||||
|
||||
|
||||
def _fmt_runway(hours: float) -> str:
|
||||
"""Runway as hours below two days, days above. Money-scale, not micro."""
|
||||
if hours < 48:
|
||||
return f"~{hours:.1f}h"
|
||||
return f"~{hours / 24:.1f}d"
|
||||
|
||||
|
||||
def _format_quota_lead(
|
||||
balance: Optional[float],
|
||||
burn: Optional[float],
|
||||
runway_hours: Optional[float],
|
||||
low_warning: bool,
|
||||
) -> str:
|
||||
"""The quota panel's headline: balance, burn rate, projected runway.
|
||||
|
||||
Deliberately 2dp rather than ``_fmt_usd``'s 6 — this is money an operator
|
||||
acts on, not per-request micro-money.
|
||||
|
||||
``burn`` and ``runway_hours`` are legitimately ``None`` when the sample
|
||||
segment is too short to extrapolate (PR #30's guard against a wild rate
|
||||
right after a top-up). That is a real answer, not missing data, so it
|
||||
renders "burn n/a" and the caller shows ``runway_note`` alongside. The
|
||||
literal "None" must never reach the screen.
|
||||
"""
|
||||
if balance is None:
|
||||
return "balance n/a"
|
||||
parts = [f"balance ${balance:.2f} left"]
|
||||
parts.append(f"burn ${burn:.2f}/h" if burn is not None else "burn n/a")
|
||||
if runway_hours is not None:
|
||||
runway = f"runway {_fmt_runway(float(runway_hours))}"
|
||||
# Same markup idiom the legend already uses for `resets`.
|
||||
parts.append(
|
||||
f"[rgb(200,80,80)]{runway}[/rgb(200,80,80)]" if low_warning else runway
|
||||
)
|
||||
return " · ".join(parts)
|
||||
|
||||
|
||||
# Compact bottom keys legend. The 11 BINDINGS are collapsed to these short
|
||||
# (key, label) pairs — duplicate quit bindings (q/Q/ctrl+c) show once. Rendered
|
||||
# as a single markup string so Textual's Static wraps rather than truncates.
|
||||
@@ -95,6 +133,47 @@ def _flex_indicator(r: dict) -> str:
|
||||
return str(pref)
|
||||
|
||||
|
||||
def _short_time(value: object) -> str:
|
||||
"""``HH:MM:SS`` in LOCAL time from an ISO-8601 ``observed_at``.
|
||||
|
||||
The decision feed is dense and almost always same-day, so the date would
|
||||
crowd out ``selected`` — the column operators actually read. Local rather
|
||||
than UTC because this is a live feed watched next to a wall clock.
|
||||
|
||||
Returns ``""`` for a missing or unparseable value rather than raising: a
|
||||
malformed timestamp must not take down the whole table render.
|
||||
"""
|
||||
if not value:
|
||||
return ""
|
||||
try:
|
||||
parsed = datetime.fromisoformat(str(value))
|
||||
except (TypeError, ValueError):
|
||||
return ""
|
||||
if parsed.tzinfo is not None:
|
||||
parsed = parsed.astimezone()
|
||||
return parsed.strftime("%H:%M:%S")
|
||||
|
||||
|
||||
def _exploration_flag(r: dict) -> str:
|
||||
"""``E`` when epsilon-greedy picked this row instead of the ranking.
|
||||
|
||||
An exploratory pick is a deliberate random sample, NOT the router's
|
||||
judgement. Without a visible marker an operator reads it as the ranking's
|
||||
answer and draws the wrong conclusion about model quality.
|
||||
"""
|
||||
return "E" if r.get("exploration") else ""
|
||||
|
||||
|
||||
def _flags_indicator(r: dict) -> str:
|
||||
"""The combined flags cell: flex label and exploration marker.
|
||||
|
||||
One cell rather than two columns — the terminal is not wide, and the spec
|
||||
caps the table at one added column plus one flag.
|
||||
"""
|
||||
parts = [part for part in (_flex_indicator(r), _exploration_flag(r)) if part]
|
||||
return " ".join(parts)
|
||||
|
||||
|
||||
class DashboardApp(App):
|
||||
"""A terminal dashboard that reads GET /metrics and renders panels."""
|
||||
|
||||
@@ -155,6 +234,12 @@ class DashboardApp(App):
|
||||
#quota-progress {
|
||||
width: 100%;
|
||||
}
|
||||
#quota-lead {
|
||||
text-style: bold;
|
||||
}
|
||||
#quota-note {
|
||||
color: $text-muted;
|
||||
}
|
||||
#quota-legend {
|
||||
color: $text-muted;
|
||||
}
|
||||
@@ -207,6 +292,12 @@ class DashboardApp(App):
|
||||
yield Static("", id="loading-panel")
|
||||
yield Static("Quota burn", classes="panel-title")
|
||||
with Vertical(id="quota-panel"):
|
||||
# Balance and runway lead; the plan-percentage bar is demoted
|
||||
# below them. A bar against a routinely-exceeded plan implies
|
||||
# a ceiling that does not exist — the credit balance is the
|
||||
# real one (PR #30).
|
||||
yield Static(markup=True, id="quota-lead")
|
||||
yield Static(id="quota-note")
|
||||
yield ProgressBar(id="quota-progress")
|
||||
yield Static(markup=True, id="quota-legend")
|
||||
yield Static("Recent decisions (enter = details)", classes="panel-title")
|
||||
@@ -312,7 +403,16 @@ class DashboardApp(App):
|
||||
model_table.add_columns("model", "calls", "cost $", "kWh", "gCO2eq")
|
||||
decision_table = self.query_one("#decision-table", DataTable)
|
||||
decision_table.add_columns(
|
||||
"id", "kind", "category", "tier", "ctx", "selected", "est $", "flex"
|
||||
"time",
|
||||
"id",
|
||||
"kind",
|
||||
"category",
|
||||
"tier",
|
||||
"ctx",
|
||||
"profile",
|
||||
"selected",
|
||||
"est $",
|
||||
"flags",
|
||||
)
|
||||
decision_table.cursor_type = "row"
|
||||
decision_table.zebra_stripes = True
|
||||
@@ -509,6 +609,37 @@ class DashboardApp(App):
|
||||
metered = rows.get("metered_kwh_30d")
|
||||
frac = rows.get("fraction")
|
||||
calls = rows.get("calls")
|
||||
|
||||
# Balance/runway lead. Wired OUTSIDE the plan gate on purpose: the
|
||||
# credit balance is meaningful whether or not a kWh plan is set, and
|
||||
# it is the figure that actually stops traffic.
|
||||
lead = self.query_one("#quota-lead", Static)
|
||||
note = self.query_one("#quota-note", Static)
|
||||
if rows:
|
||||
balance = rows.get("balance_usd")
|
||||
burn = rows.get("burn_rate_usd_per_hour")
|
||||
lead.update(
|
||||
_format_quota_lead(
|
||||
balance,
|
||||
burn,
|
||||
rows.get("projected_hours_remaining"),
|
||||
bool(rows.get("runway_low_warning")),
|
||||
)
|
||||
)
|
||||
# Explain an absent burn rate rather than leaving a bare "n/a".
|
||||
if balance is not None and burn is None:
|
||||
note.update(
|
||||
str(rows.get("runway_note") or "burn estimate unavailable")
|
||||
)
|
||||
note.display = True
|
||||
else:
|
||||
note.update("")
|
||||
note.display = False
|
||||
else:
|
||||
lead.update("")
|
||||
note.update("")
|
||||
note.display = False
|
||||
|
||||
if plan is not None and float(plan) > 0:
|
||||
bar.total = float(plan)
|
||||
bar.progress = float(metered or 0)
|
||||
@@ -590,18 +721,23 @@ class DashboardApp(App):
|
||||
dt.clear()
|
||||
for r in self._last_model["recent_decisions"]:
|
||||
dt.add_row(
|
||||
_short_time(r.get("observed_at")),
|
||||
str(r.get("id")),
|
||||
str(r.get("kind")),
|
||||
str(r.get("category")),
|
||||
str(r.get("tier")),
|
||||
str(r.get("required_context_tokens")),
|
||||
# `or ""` so an absent profile is blank, never the literal "None".
|
||||
str(r.get("profile") or ""),
|
||||
str(r.get("selected")),
|
||||
_fmt_usd(r.get("est_cost_usd")),
|
||||
_flex_indicator(r),
|
||||
_flags_indicator(r),
|
||||
key=str(r.get("id")),
|
||||
)
|
||||
if not self._last_model["recent_decisions"]:
|
||||
dt.add_row("(no decisions)", "", "", "", "", "", "", "", key="_placeholder")
|
||||
dt.add_row(
|
||||
"(no decisions)", "", "", "", "", "", "", "", "", "", key="_placeholder"
|
||||
)
|
||||
self._restore_cursor(dt, saved_decision_key)
|
||||
|
||||
def _render_category_table(self) -> None:
|
||||
|
||||
@@ -143,6 +143,12 @@ def decision_row(r: dict) -> dict:
|
||||
"flex_preference": r.get("flex_preference"),
|
||||
"flex_swapped": r.get("flex_swapped"),
|
||||
"flex_forced": r.get("flex_forced"),
|
||||
"profile": r.get("profile"),
|
||||
"exploration": r.get("exploration"),
|
||||
"pinch_original_tokens": r.get("pinch_original_tokens"),
|
||||
"pinch_final_tokens": r.get("pinch_final_tokens"),
|
||||
"request_id": r.get("request_id"),
|
||||
"session_key": r.get("session_key"),
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -59,6 +59,9 @@ ROUTE_DECISIONS_COLUMNS = [
|
||||
"flex_preference",
|
||||
"flex_swapped",
|
||||
"flex_forced",
|
||||
# Enforcement for this list lives in tests/test_tui_schema_drift.py —
|
||||
# keep new route_decisions columns registered there, not just here.
|
||||
"request_id",
|
||||
"exploration",
|
||||
"pinch_original_tokens",
|
||||
"pinch_final_tokens",
|
||||
|
||||
@@ -11,7 +11,9 @@ file that may. No non-tui module imports textual.
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import copy
|
||||
import time
|
||||
from datetime import datetime
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
@@ -20,7 +22,7 @@ from textual.widgets import ProgressBar, Static
|
||||
import tui
|
||||
import tui_model
|
||||
from tui_model import build_model, fetch_metrics
|
||||
from tui_screens import VerdictMixScreen
|
||||
from tui_screens import DecisionDetailScreen, VerdictMixScreen
|
||||
|
||||
|
||||
def _fixture() -> dict:
|
||||
@@ -144,6 +146,46 @@ def _fixture() -> dict:
|
||||
}
|
||||
|
||||
|
||||
def _enriched_payload() -> dict:
|
||||
"""A deep copy of _fixture() whose recent_decisions rows carry the six
|
||||
T2 data-layer fields, mirroring what /metrics returns after the drift
|
||||
catch-up. Row 42 is the popup test's target row (exploration off);
|
||||
row 41 is the exploratory pick; row 40 stays a rejection row."""
|
||||
data = copy.deepcopy(_fixture())
|
||||
enrichments = [
|
||||
{
|
||||
"profile": "default",
|
||||
"exploration": 0,
|
||||
"request_id": "chatcmpl-fixture-42",
|
||||
"pinch_original_tokens": 120000,
|
||||
"pinch_final_tokens": 96122,
|
||||
"session_key": "sess-42",
|
||||
"observed_at": "2026-08-23T10:00:00+00:00",
|
||||
},
|
||||
{
|
||||
"profile": "default",
|
||||
"exploration": 1,
|
||||
"request_id": "chatcmpl-fixture-41",
|
||||
"pinch_original_tokens": 90000,
|
||||
"pinch_final_tokens": 87500,
|
||||
"session_key": "sess-41",
|
||||
"observed_at": "2026-08-23T09:59:00+00:00",
|
||||
},
|
||||
{
|
||||
"profile": "default",
|
||||
"exploration": 0,
|
||||
"request_id": None,
|
||||
"pinch_original_tokens": None,
|
||||
"pinch_final_tokens": None,
|
||||
"session_key": "sess-40",
|
||||
"observed_at": "2026-08-23T09:58:00+00:00",
|
||||
},
|
||||
]
|
||||
for row, extra in zip(data["recent_decisions"], enrichments):
|
||||
row.update(extra)
|
||||
return data
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Direct unit tests of the data layer (no TUI running).
|
||||
# --------------------------------------------------------------------------
|
||||
@@ -1056,6 +1098,78 @@ def test_show_decision_detail_pushes_modal_with_full_row():
|
||||
asyncio.run(_go())
|
||||
|
||||
|
||||
def test_detail_popup_carries_forensics_fields():
|
||||
"""The e-key popup carries the six T2 forensics fields — profile,
|
||||
exploration, pinch tokens, request_id, session_key — both on the live
|
||||
decision dict and in the rendered JSON for a metrics-shaped full row."""
|
||||
stub = _StubFetcher()
|
||||
stub.payload = _enriched_payload()
|
||||
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
||||
|
||||
async def _go():
|
||||
async with app.run_test() as pilot:
|
||||
await pilot.pause()
|
||||
app.query_one("#decision-table").focus()
|
||||
await pilot.pause()
|
||||
await pilot.press("e")
|
||||
await pilot.pause()
|
||||
from textual.screen import ModalScreen
|
||||
|
||||
assert isinstance(app.screen, ModalScreen)
|
||||
decision = app.screen.decision
|
||||
assert decision["request_id"] == "chatcmpl-fixture-42"
|
||||
assert decision["session_key"] == "sess-42"
|
||||
assert decision["pinch_original_tokens"] == 120000
|
||||
assert decision["pinch_final_tokens"] == 96122
|
||||
assert decision["profile"] == "default"
|
||||
assert decision["exploration"] == 0
|
||||
|
||||
asyncio.run(_go())
|
||||
|
||||
metrics_row = {
|
||||
"id": 43,
|
||||
"observed_at": "2026-09-05T12:34:56+00:00",
|
||||
"kind": "chat",
|
||||
"task_category": "coding_general",
|
||||
"task_tier": 2,
|
||||
"required_context_tokens": 123456,
|
||||
"confidence": 0.91,
|
||||
"classifier_ms": 1800,
|
||||
"classification_source": "classifier",
|
||||
"latency_tolerance": "interactive",
|
||||
"candidates_considered": 7,
|
||||
"selected_model": "fixture-model",
|
||||
"selected_provider": "neuralwatt",
|
||||
"runner_up_models": None,
|
||||
"est_cost_usd": 0.001,
|
||||
"est_proficiency": 0.9,
|
||||
"rejected_reason": None,
|
||||
"session_key": "sess-43",
|
||||
"tools": 0,
|
||||
"images": 0,
|
||||
"json_mode": 0,
|
||||
"streamed": 1,
|
||||
"flex_preference": "auto",
|
||||
"flex_swapped": 0,
|
||||
"flex_forced": 0,
|
||||
"request_id": "chatcmpl-fixture-43",
|
||||
"exploration": 1,
|
||||
"pinch_original_tokens": 120000,
|
||||
"pinch_final_tokens": 96122,
|
||||
"profile": "default",
|
||||
}
|
||||
text = DecisionDetailScreen(tui_model.decision_row(metrics_row))._render_text()
|
||||
for key in (
|
||||
"profile",
|
||||
"exploration",
|
||||
"pinch_original_tokens",
|
||||
"pinch_final_tokens",
|
||||
"request_id",
|
||||
"session_key",
|
||||
):
|
||||
assert key in text
|
||||
|
||||
|
||||
def test_app_run_test_ctrl_v_opens_verdict_popup():
|
||||
"""Pressing ``ctrl+v`` pushes a VerdictMixScreen modal showing the verdict
|
||||
mix rows from the current model, and ``escape`` dismisses it back to the
|
||||
@@ -1161,6 +1275,46 @@ def test_live_decision_inserts_row_at_front_and_rerenders():
|
||||
asyncio.run(_go())
|
||||
|
||||
|
||||
def test_live_decision_carries_new_fields():
|
||||
"""A live SSE decision keeps the six T2 fields through the shared
|
||||
decision_row projection, so a prompt `e` on a fresh decision shows the
|
||||
same forensics as a /metrics-sourced row."""
|
||||
stub = _StubFetcher()
|
||||
stub.payload = _fixture()
|
||||
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
||||
|
||||
async def _go():
|
||||
async with app.run_test() as pilot:
|
||||
await pilot.pause()
|
||||
app._handle_live_decision(
|
||||
{
|
||||
"id": 1001,
|
||||
"kind": "chat",
|
||||
"task_category": "coding_general",
|
||||
"task_tier": 2,
|
||||
"selected_model": "deepseek-v4-flash",
|
||||
"est_cost_usd": 0.0002,
|
||||
"profile": "default",
|
||||
"exploration": 1,
|
||||
"request_id": "chatcmpl-live-1001",
|
||||
"pinch_original_tokens": 120000,
|
||||
"pinch_final_tokens": 96122,
|
||||
"session_key": "sess-live-1001",
|
||||
}
|
||||
)
|
||||
await pilot.pause()
|
||||
row = app._last_model["recent_decisions"][0]
|
||||
assert row["id"] == 1001
|
||||
assert row["profile"] == "default"
|
||||
assert row["exploration"] == 1
|
||||
assert row["request_id"] == "chatcmpl-live-1001"
|
||||
assert row["pinch_original_tokens"] == 120000
|
||||
assert row["pinch_final_tokens"] == 96122
|
||||
assert row["session_key"] == "sess-live-1001"
|
||||
|
||||
asyncio.run(_go())
|
||||
|
||||
|
||||
def test_live_decision_caps_recent_decisions_at_fifty():
|
||||
"""The live feed never grows the in-memory list past the /metrics cap, so
|
||||
the dashboard's view stays consistent with a /metrics refresh."""
|
||||
@@ -1404,3 +1558,219 @@ def test_format_quota_legend_shows_next_reset_date():
|
||||
assert "2026-07-26" in legend # window start still shown
|
||||
assert "2026-09-06" in legend # billing reset
|
||||
assert "resets" in legend.lower()
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# T3 — decision table display: time column, profile column, exploration flag.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_decision_table_columns_and_order():
|
||||
"""The table declares exactly ten columns with `time` first.
|
||||
|
||||
Time leads because it is the natural scan axis for a live feed; `flex`
|
||||
became `flags` because exploration now shares that cell.
|
||||
"""
|
||||
stub = _StubFetcher()
|
||||
stub.payload = _enriched_payload()
|
||||
app = tui.DashboardApp(fetcher=stub)
|
||||
|
||||
def _assert(a):
|
||||
dt = a.query_one("#decision-table")
|
||||
labels = [str(c.label) for c in dt.columns.values()]
|
||||
assert labels == [
|
||||
"time",
|
||||
"id",
|
||||
"kind",
|
||||
"category",
|
||||
"tier",
|
||||
"ctx",
|
||||
"profile",
|
||||
"selected",
|
||||
"est $",
|
||||
"flags",
|
||||
]
|
||||
|
||||
_run_app(app, _assert)
|
||||
|
||||
|
||||
def test_short_time_renders_hhmmss_and_degrades_quietly():
|
||||
"""`_short_time` is HH:MM:SS local, and never raises on bad input.
|
||||
|
||||
A malformed timestamp must not take down the whole table render, so the
|
||||
unparseable cases return "" rather than propagating ValueError.
|
||||
"""
|
||||
rendered = tui._short_time("2026-08-23T10:00:00+00:00")
|
||||
assert len(rendered) == 8 and rendered.count(":") == 2
|
||||
# Local conversion, computed the same way the helper does it.
|
||||
expected = (
|
||||
datetime.fromisoformat("2026-08-23T10:00:00+00:00")
|
||||
.astimezone()
|
||||
.strftime("%H:%M:%S")
|
||||
)
|
||||
assert rendered == expected
|
||||
# No date component leaks into the cell.
|
||||
assert "2026" not in rendered
|
||||
for bad in (None, "", "not-a-timestamp", 12345):
|
||||
assert tui._short_time(bad) == ""
|
||||
|
||||
|
||||
def test_app_decision_table_profile_column_blank_not_none():
|
||||
"""`profile` renders its value, and renders BLANK when absent.
|
||||
|
||||
A literal "None" in a dense table reads as a real profile name.
|
||||
"""
|
||||
stub = _StubFetcher()
|
||||
data = _enriched_payload()
|
||||
data["recent_decisions"][1]["profile"] = None
|
||||
stub.payload = data
|
||||
app = tui.DashboardApp(fetcher=stub)
|
||||
|
||||
def _assert(a):
|
||||
dt = a.query_one("#decision-table")
|
||||
labels = [str(c.label) for c in dt.columns.values()]
|
||||
profile_at = labels.index("profile")
|
||||
assert str(dt.get_row_at(0)[profile_at]) == "default"
|
||||
# Absent profile is an empty cell, never the literal "None".
|
||||
assert str(dt.get_row_at(1)[profile_at]) == ""
|
||||
|
||||
_run_app(app, _assert)
|
||||
|
||||
|
||||
def test_exploration_flag_and_flags_indicator_compose():
|
||||
"""`E` marks an exploratory pick and composes with the flex label.
|
||||
|
||||
An epsilon-greedy pick is a random sample, not the ranking's judgement;
|
||||
without the marker an operator reads it as the router's answer.
|
||||
"""
|
||||
assert tui._exploration_flag({"exploration": 1}) == "E"
|
||||
assert tui._exploration_flag({"exploration": 0}) == ""
|
||||
assert tui._exploration_flag({}) == ""
|
||||
# Composition: flex label first, then the marker, space separated.
|
||||
both = tui._flags_indicator(
|
||||
{"flex_preference": "auto", "flex_swapped": 0, "exploration": 1}
|
||||
)
|
||||
assert both == "auto E"
|
||||
# Either alone survives; neither yields an empty cell.
|
||||
assert tui._flags_indicator({"exploration": 1}) == "E"
|
||||
assert tui._flags_indicator({"flex_preference": "auto"}) == "auto"
|
||||
assert tui._flags_indicator({}) == ""
|
||||
|
||||
|
||||
def test_app_decision_table_exploration_flag_renders():
|
||||
"""Row 41 is the exploratory pick in the fixture; its flags cell shows E."""
|
||||
stub = _StubFetcher()
|
||||
stub.payload = _enriched_payload()
|
||||
app = tui.DashboardApp(fetcher=stub)
|
||||
|
||||
def _assert(a):
|
||||
dt = a.query_one("#decision-table")
|
||||
exploratory = [str(c) for c in dt.get_row_at(1)]
|
||||
assert any("E" == cell or cell.endswith(" E") for cell in exploratory)
|
||||
non_exploratory = [str(c) for c in dt.get_row_at(0)]
|
||||
assert not any(cell == "E" or cell.endswith(" E") for cell in non_exploratory)
|
||||
|
||||
_run_app(app, _assert)
|
||||
|
||||
|
||||
def test_placeholder_row_matches_column_arity():
|
||||
"""The empty-state placeholder must have exactly ten cells.
|
||||
|
||||
A short placeholder row raises inside Textual at render time — an arity
|
||||
mismatch is a crash, not a cosmetic issue.
|
||||
"""
|
||||
stub = _StubFetcher()
|
||||
data = _enriched_payload()
|
||||
data["recent_decisions"] = []
|
||||
stub.payload = data
|
||||
app = tui.DashboardApp(fetcher=stub)
|
||||
|
||||
def _assert(a):
|
||||
dt = a.query_one("#decision-table")
|
||||
assert len(dt.get_row_at(0)) == len(dt.columns) == 10
|
||||
|
||||
_run_app(app, _assert)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# T4 — quota panel leads with balance and runway.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_format_quota_lead_states():
|
||||
"""All four lead states, including the one this deployment is actually in.
|
||||
|
||||
burn/runway are legitimately None when the sample segment is too short
|
||||
(PR #30). That is deliberate, not missing data, so the None path must
|
||||
still say something an operator can read.
|
||||
"""
|
||||
full = tui._format_quota_lead(8.50, 1.0, 8.5, False)
|
||||
assert "balance $8.50 left" in full
|
||||
assert "burn $1.00/h" in full
|
||||
assert "runway ~8.5h" in full
|
||||
|
||||
none_burn = tui._format_quota_lead(8.50, None, None, False)
|
||||
assert "burn n/a" in none_burn
|
||||
assert "None" not in none_burn
|
||||
|
||||
assert tui._format_quota_lead(None, 1.0, 8.5, False) == "balance n/a"
|
||||
assert "runway ~20.8d" in tui._format_quota_lead(100.0, 1.0, 500.0, False)
|
||||
|
||||
low = tui._format_quota_lead(8.50, 4.0, 1.5, True)
|
||||
assert "[rgb(200,80,80)]runway ~1.5h" in low
|
||||
|
||||
|
||||
def test_quota_panel_burn_unavailable_renders_note_verbatim():
|
||||
"""When burn is None the note renders VERBATIM — never blank, 0, or None.
|
||||
|
||||
This is the state the live deployment is in right now, so it is the first
|
||||
thing an operator sees, not an edge case.
|
||||
"""
|
||||
stub = _StubFetcher()
|
||||
data = _fixture()
|
||||
note = "burn estimate unavailable: segment after last balance increase spans only 9 minutes"
|
||||
for block in (data["quota"], data["coverage"]["quota"]):
|
||||
block["balance_usd"] = 7.5
|
||||
block["burn_rate_usd_per_hour"] = None
|
||||
block["projected_hours_remaining"] = None
|
||||
block["runway_low_warning"] = False
|
||||
block["runway_note"] = note
|
||||
stub.payload = data
|
||||
app = tui.DashboardApp(fetcher=stub)
|
||||
|
||||
def _assert(a):
|
||||
lead_text = str(a.query_one("#quota-lead", Static).content)
|
||||
note_text = str(a.query_one("#quota-note", Static).content)
|
||||
assert "balance $7.50 left" in lead_text
|
||||
assert "burn n/a" in lead_text
|
||||
assert "None" not in lead_text
|
||||
# The note is passed through unchanged, not summarised.
|
||||
assert note_text == note
|
||||
assert note_text.strip() != ""
|
||||
assert note_text.strip() != "0"
|
||||
assert note_text.strip() != "None"
|
||||
|
||||
_run_app(app, _assert)
|
||||
|
||||
|
||||
def test_quota_panel_note_hidden_when_burn_is_available():
|
||||
"""With a real burn rate there is nothing to explain, so no note line."""
|
||||
stub = _StubFetcher()
|
||||
data = _fixture()
|
||||
for block in (data["quota"], data["coverage"]["quota"]):
|
||||
block["balance_usd"] = 12.0
|
||||
block["burn_rate_usd_per_hour"] = 0.5
|
||||
block["projected_hours_remaining"] = 24.0
|
||||
block["runway_low_warning"] = False
|
||||
block["runway_note"] = None
|
||||
stub.payload = data
|
||||
app = tui.DashboardApp(fetcher=stub)
|
||||
|
||||
def _assert(a):
|
||||
lead_text = str(a.query_one("#quota-lead", Static).content)
|
||||
assert "balance $12.00 left" in lead_text
|
||||
assert "burn $0.50/h" in lead_text
|
||||
assert "runway ~24.0h" in lead_text
|
||||
assert a.query_one("#quota-note", Static).display is False
|
||||
|
||||
_run_app(app, _assert)
|
||||
|
||||
200
tests/test_tui_schema_drift.py
Normal file
200
tests/test_tui_schema_drift.py
Normal file
@@ -0,0 +1,200 @@
|
||||
"""Schema-drift guard: every route_decisions column reaches the TUI on purpose.
|
||||
|
||||
Five-plus columns shipped to ``route_decisions`` across recent merges without
|
||||
any of them reaching the dashboard, and ``tests/test_route_decisions.py``'s own
|
||||
hand-maintained column list had ALREADY drifted (missing ``request_id``) — the
|
||||
drift recurs, so this file is the enforcement, not the catch-up.
|
||||
|
||||
Two registries, not a magic diff:
|
||||
|
||||
- ``SCHEMA_TO_MODEL_KEYS`` — schema column -> the ``tui_model.decision_row``
|
||||
key that surfaces it. This is the recorded DECISION about surfacing. Three
|
||||
keys are intentionally renamed (``task_category`` -> ``category``,
|
||||
``task_tier`` -> ``tier``, ``selected_model`` -> ``selected``), which is
|
||||
exactly why the mapping is explicit and never derived by string identity.
|
||||
- ``UNSURFACED_COLUMNS`` — columns deliberately NOT surfaced, each with a
|
||||
one-line reason. Empty today.
|
||||
|
||||
Every new schema column must land in one of the two registries or
|
||||
``test_schema_columns_all_have_surfacing_decisions`` fails naming the column;
|
||||
every registry entry must actually round-trip through ``decision_row`` and the
|
||||
``/metrics`` SELECT or the other two tests fail naming the key.
|
||||
|
||||
Offline: temp SQLite created from ``config/schema.sql``, never the live
|
||||
router.db (pattern borrowed from tests/test_route_decisions.py).
|
||||
"""
|
||||
|
||||
import sqlite3
|
||||
from pathlib import Path
|
||||
|
||||
import metrics
|
||||
import tui_model
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
||||
|
||||
# route_decisions column -> the tui_model.decision_row key that surfaces it.
|
||||
# Schema order (config/schema.sql route_decisions declaration). Add a new
|
||||
# schema column here when the TUI should surface it, or to
|
||||
# UNSURFACED_COLUMNS when it deliberately should not.
|
||||
SCHEMA_TO_MODEL_KEYS = {
|
||||
"id": "id",
|
||||
"observed_at": "observed_at",
|
||||
"kind": "kind",
|
||||
"task_category": "category",
|
||||
"task_tier": "tier",
|
||||
"required_context_tokens": "required_context_tokens",
|
||||
"confidence": "confidence",
|
||||
"classifier_ms": "classifier_ms",
|
||||
"classification_source": "classification_source",
|
||||
"latency_tolerance": "latency_tolerance",
|
||||
"candidates_considered": "candidates_considered",
|
||||
"selected_model": "selected",
|
||||
"selected_provider": "selected_provider",
|
||||
"runner_up_models": "runner_up_models",
|
||||
"est_cost_usd": "est_cost_usd",
|
||||
"est_proficiency": "est_proficiency",
|
||||
"rejected_reason": "rejected_reason",
|
||||
"session_key": "session_key",
|
||||
"tools": "tools",
|
||||
"images": "images",
|
||||
"json_mode": "json_mode",
|
||||
"streamed": "streamed",
|
||||
"flex_preference": "flex_preference",
|
||||
"flex_swapped": "flex_swapped",
|
||||
"flex_forced": "flex_forced",
|
||||
"request_id": "request_id",
|
||||
"exploration": "exploration",
|
||||
"pinch_original_tokens": "pinch_original_tokens",
|
||||
"pinch_final_tokens": "pinch_final_tokens",
|
||||
"profile": "profile",
|
||||
}
|
||||
# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a
|
||||
# one-line reason (callers must keep the comment). Empty today: every column is
|
||||
# surfaced via decision_row. A new schema column that lands in NEITHER registry
|
||||
# fails the drift test.
|
||||
UNSURFACED_COLUMNS: set[str] = set()
|
||||
|
||||
# Every column seeded non-NULL: the single source of truth for the INSERT and
|
||||
# the round-trip assertions, so the tests prove the column flows, not that
|
||||
# None flows.
|
||||
FULL_ROW = {
|
||||
"id": 1,
|
||||
"observed_at": "2026-09-05T12:34:56+00:00",
|
||||
"kind": "chat",
|
||||
"task_category": "coding_general",
|
||||
"task_tier": 2,
|
||||
"required_context_tokens": 123456,
|
||||
"confidence": 0.91,
|
||||
"classifier_ms": 1800,
|
||||
"classification_source": "classifier",
|
||||
"latency_tolerance": "interactive",
|
||||
"candidates_considered": 7,
|
||||
"selected_model": "fixture-model",
|
||||
"selected_provider": "neuralwatt",
|
||||
"runner_up_models": '[{"model_id": "runner", "provider": "neuralwatt"}]',
|
||||
"est_cost_usd": 0.00123,
|
||||
"est_proficiency": 0.88,
|
||||
# Schema-legal even for a selected row (no CHECK constraint); non-NULL so the
|
||||
# round-trip assertion proves the column flows, not that None flows.
|
||||
"rejected_reason": "fixture rejected reason",
|
||||
"session_key": "fixture-session-key",
|
||||
"tools": 0,
|
||||
"images": 0,
|
||||
"json_mode": 1,
|
||||
"streamed": 1,
|
||||
"flex_preference": "auto",
|
||||
"flex_swapped": 0,
|
||||
"flex_forced": 0,
|
||||
"request_id": "chatcmpl-fixture-42",
|
||||
"exploration": 1,
|
||||
"pinch_original_tokens": 120000,
|
||||
"pinch_final_tokens": 96122,
|
||||
"profile": "default",
|
||||
}
|
||||
|
||||
|
||||
def _db_from_schema(tmp_path, row_factory=None) -> sqlite3.Connection:
|
||||
"""A throwaway DB from config/schema.sql — never the live router.db.
|
||||
|
||||
``row_factory`` defaults to None (plain tuples) like the PRAGMA test
|
||||
needs; the metrics round-trip test passes ``sqlite3.Row`` because
|
||||
``metrics.recent_decisions`` builds its rows via ``dict(row)``.
|
||||
"""
|
||||
conn = sqlite3.connect(tmp_path / "schema-drift.db")
|
||||
conn.executescript(SCHEMA_SQL)
|
||||
conn.row_factory = row_factory
|
||||
return conn
|
||||
|
||||
|
||||
def test_schema_columns_all_have_surfacing_decisions(tmp_path):
|
||||
"""Every route_decisions column is either surfaced or consciously not."""
|
||||
conn = _db_from_schema(tmp_path)
|
||||
schema_cols = {
|
||||
row[1] for row in conn.execute("PRAGMA table_info(route_decisions)")
|
||||
}
|
||||
conn.close()
|
||||
|
||||
registered = set(SCHEMA_TO_MODEL_KEYS) | UNSURFACED_COLUMNS
|
||||
missing = schema_cols - set(SCHEMA_TO_MODEL_KEYS) - UNSURFACED_COLUMNS
|
||||
extra = registered - schema_cols
|
||||
assert not missing, (
|
||||
f"route_decisions column(s) added without a surfacing decision: "
|
||||
f"{sorted(missing)}. Add each to SCHEMA_TO_MODEL_KEYS if the TUI "
|
||||
f"should surface it (project it in tui_model.decision_row), or to "
|
||||
f"UNSURFACED_COLUMNS with a reason comment."
|
||||
)
|
||||
assert not extra, (
|
||||
f"registry registers column(s) that no longer exist in "
|
||||
f"route_decisions: {sorted(extra)}."
|
||||
)
|
||||
|
||||
|
||||
def test_decision_row_projects_every_tracked_column():
|
||||
"""decision_row surfaces every registered column's value faithfully."""
|
||||
projected = tui_model.decision_row(dict(FULL_ROW))
|
||||
|
||||
expected_keys = set(SCHEMA_TO_MODEL_KEYS.values())
|
||||
assert set(projected) == expected_keys, (
|
||||
"tui_model.decision_row no longer matches the surfacing registry "
|
||||
f"(symmetric difference: {sorted(expected_keys ^ set(projected))}).\n"
|
||||
f" registry values: {sorted(expected_keys)}\n"
|
||||
f" decision_row keys: {sorted(projected)}"
|
||||
)
|
||||
for col, key in SCHEMA_TO_MODEL_KEYS.items():
|
||||
assert projected[key] == FULL_ROW[col], (
|
||||
f"route_decisions column {col!r} must surface as decision_row "
|
||||
f"key {key!r}: got {projected[key]!r}, want {FULL_ROW[col]!r}"
|
||||
)
|
||||
|
||||
|
||||
def test_metrics_recent_decisions_selects_every_tracked_column(tmp_path):
|
||||
"""The /metrics path carries every registered column end-to-end.
|
||||
|
||||
``request_id`` reaches the TUI ONLY through this SELECT (SSE publishes
|
||||
before the backfill), so a narrowed SELECT silently erases forensics
|
||||
fields a few seconds after a live decision arrives.
|
||||
"""
|
||||
conn = _db_from_schema(tmp_path, row_factory=sqlite3.Row)
|
||||
cols = ",".join(FULL_ROW.keys())
|
||||
placeholders = ",".join("?" * len(FULL_ROW))
|
||||
conn.execute(
|
||||
f"INSERT INTO route_decisions ({cols}) VALUES ({placeholders})",
|
||||
tuple(FULL_ROW.values()),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
rows = metrics.recent_decisions(conn, limit=5)
|
||||
conn.close()
|
||||
assert rows, "the seeded FULL_ROW must come back from recent_decisions"
|
||||
projected = tui_model.decision_row(rows[0])
|
||||
|
||||
unreachable = []
|
||||
for col, key in SCHEMA_TO_MODEL_KEYS.items():
|
||||
if projected.get(key) != FULL_ROW[col]:
|
||||
unreachable.append(
|
||||
f"route_decisions column {col!r} does not reach the TUI "
|
||||
f"through /metrics — is it selected by "
|
||||
f"metrics.recent_decisions()?"
|
||||
)
|
||||
assert not unreachable, "\n".join(unreachable)
|
||||
291
tests/test_tui_warnings.py
Normal file
291
tests/test_tui_warnings.py
Normal file
@@ -0,0 +1,291 @@
|
||||
"""Tripwire: every warning class /metrics can emit must reach the panel.
|
||||
|
||||
The 2026-09-04 vision-ceiling incident was invisible for ~19 hours because a
|
||||
whole warning family was computed correctly and never displayed. The fix there
|
||||
was surfacing, not computing — so the guard has to be a test that fails when a
|
||||
warning class exists but nothing renders it.
|
||||
|
||||
Two directions, both load-bearing:
|
||||
|
||||
* every REGISTERED class actually fires against a fixture that provokes all of
|
||||
them — catches the fixture rotting or a message shape changing underneath;
|
||||
* every EMITTED warning is registered — catches a NEW class being added to
|
||||
``metrics.scoring_coverage`` without anyone deciding to surface it. This one
|
||||
fails with the raw warning text, because the whole point is that nobody knew
|
||||
the class existed.
|
||||
|
||||
Deliberately no assertions on warning COUNT: the numbers shift as fixture
|
||||
semantics evolve, and item 4 of the spec asks for class coverage, not arity.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import copy
|
||||
import re
|
||||
import sqlite3
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
from textual.widgets import Static
|
||||
|
||||
import metrics
|
||||
import tui
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
||||
|
||||
|
||||
# The registry. A class here without a matching emitted warning means the
|
||||
# fixture rotted; an emitted warning without a class here means someone added
|
||||
# a warning nobody decided to display.
|
||||
WARN_CLASS_MATCHERS = {
|
||||
"quota-over-plan": r"metered usage is \d+% of the",
|
||||
"models-missing-energy": r"\d+/\d+ routable models have no reference-workload",
|
||||
"models-missing-proficiency": r"\d+/\d+ routable models have no proficiency",
|
||||
"catalog-stale": r"catalog last polled",
|
||||
# Fixed-width lookbehind so this cannot swallow the capability warnings,
|
||||
# which share the "tier N context ceiling" tail.
|
||||
"demand-ceiling": r"(?<!capable )tier \d+ context ceiling",
|
||||
"escalation-hazard": r"escalation hazard: tier",
|
||||
"zero-eligible-models": r"tier \d+ has 0 eligible models",
|
||||
"capability-subceiling": r"(vision|json_mode)-capable tier",
|
||||
"rejection-new-pattern": r"new rejection pattern:",
|
||||
"rejection-rate": r"rejection rate:",
|
||||
}
|
||||
|
||||
CFG = SimpleNamespace(
|
||||
routing=SimpleNamespace(
|
||||
allowed_access_levels=["public"],
|
||||
# scoring_coverage reads `.value` off this — it is an enum in real
|
||||
# config, so a bare string is not a faithful stand-in.
|
||||
default_flex_preference=SimpleNamespace(value="auto"),
|
||||
),
|
||||
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=None),
|
||||
escalation=SimpleNamespace(enabled=True),
|
||||
)
|
||||
|
||||
|
||||
def _model_row(model_id, tier, window, *, vision=0, availability="active"):
|
||||
return (
|
||||
model_id,
|
||||
"neuralwatt",
|
||||
model_id,
|
||||
tier,
|
||||
window,
|
||||
window,
|
||||
4096,
|
||||
0.1,
|
||||
0.3,
|
||||
vision,
|
||||
1,
|
||||
"standard",
|
||||
"default",
|
||||
"full",
|
||||
"public",
|
||||
availability,
|
||||
)
|
||||
|
||||
|
||||
def _seed_chernobyl(conn: sqlite3.Connection) -> None:
|
||||
"""Seed a catalog and traffic that provoke ALL ten warning classes at once.
|
||||
|
||||
Named for what it is: a deliberately maximally-broken deployment. Every
|
||||
timestamp is relative to now so the rejection windows (1h alert, 24h
|
||||
baseline) stay valid whenever the suite runs.
|
||||
"""
|
||||
now = datetime.now(timezone.utc)
|
||||
stale = (now - timedelta(days=3)).isoformat()
|
||||
|
||||
rows = [
|
||||
# Covered model: has proficiency AND energy, so the "missing" counts
|
||||
# come out as 2/3 rather than 3/3.
|
||||
_model_row("cov-t1", 1, 20000),
|
||||
# Uncovered: drives both missing- classes; its 5000 window is also the
|
||||
# tier-2 ceiling behind the escalation hazard.
|
||||
_model_row("gap-t2", 2, 5000),
|
||||
# Vision-capable but small: the capability sub-ceiling.
|
||||
_model_row("vis-t1", 1, 5000, vision=1),
|
||||
# Deprecated: excluded from routable, so tier 3 has zero eligible.
|
||||
_model_row("dead-t3", 3, 1000, availability="deprecated"),
|
||||
]
|
||||
for row in rows:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO models (
|
||||
model_id, provider, base_model_id, tier, context_window,
|
||||
effective_context_window, max_output_tokens,
|
||||
cost_per_1m_prompt, cost_per_1m_completion,
|
||||
supports_vision, supports_json_mode,
|
||||
latency_class, reasoning_mode, context_variant,
|
||||
access_level, availability, last_updated
|
||||
) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
|
||||
""",
|
||||
(*row, stale),
|
||||
)
|
||||
|
||||
conn.execute(
|
||||
"INSERT INTO proficiency "
|
||||
"(model_id, provider, category, blended_score, last_updated) "
|
||||
"VALUES ('cov-t1','neuralwatt','coding_general',0.9,?)",
|
||||
(now.isoformat(),),
|
||||
)
|
||||
# 6.0 kWh against a 6.25 plan = 96% > 80%, firing the quota class. The
|
||||
# 'seed_reference' task_category is also what marks a model as having
|
||||
# reference-workload data, so this one row serves both purposes.
|
||||
conn.execute(
|
||||
"INSERT INTO energy_observations "
|
||||
"(model_id, provider, energy_kwh, task_category, observed_at) "
|
||||
"VALUES ('cov-t1','neuralwatt',6.0,'seed_reference',?)",
|
||||
((now - timedelta(days=1)).isoformat(),),
|
||||
)
|
||||
|
||||
def _decision(**kw):
|
||||
cols = {
|
||||
"kind": "chat",
|
||||
"task_category": "coding_general",
|
||||
"task_tier": 1,
|
||||
"required_context_tokens": 100,
|
||||
"latency_tolerance": "interactive",
|
||||
"selected_model": "cov-t1",
|
||||
"selected_provider": "neuralwatt",
|
||||
"observed_at": now.isoformat(),
|
||||
"images": 0,
|
||||
"json_mode": 0,
|
||||
"rejected_reason": None,
|
||||
}
|
||||
cols.update(kw)
|
||||
conn.execute(
|
||||
f"INSERT INTO route_decisions ({','.join(cols)}) "
|
||||
f"VALUES ({','.join('?' * len(cols))})",
|
||||
tuple(cols.values()),
|
||||
)
|
||||
|
||||
# One huge vision request: demand-ceiling (20000 < 999999), capability
|
||||
# sub-ceiling (vision 5000 < 999999), and the escalation hazard all at once.
|
||||
_decision(required_context_tokens=999999, images=1)
|
||||
|
||||
# Two rejections of a pattern with no baseline -> "new rejection pattern:".
|
||||
for _ in range(2):
|
||||
_decision(
|
||||
selected_model=None,
|
||||
selected_provider=None,
|
||||
rejected_reason="context >= 99999 tokens; interactive",
|
||||
)
|
||||
|
||||
# A familiar group: one in the baseline window, six in the alert window,
|
||||
# crossing the min-count threshold -> "rejection rate:".
|
||||
familiar = "tier >= 2; context >= 30000 tokens; interactive"
|
||||
_decision(
|
||||
task_tier=2,
|
||||
selected_model=None,
|
||||
selected_provider=None,
|
||||
rejected_reason=familiar,
|
||||
observed_at=(now - timedelta(hours=2)).isoformat(),
|
||||
)
|
||||
for _ in range(6):
|
||||
_decision(
|
||||
task_tier=2,
|
||||
selected_model=None,
|
||||
selected_provider=None,
|
||||
rejected_reason=familiar,
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def chernobyl(tmp_path):
|
||||
conn = sqlite3.connect(tmp_path / "warn.db")
|
||||
conn.executescript(SCHEMA_SQL)
|
||||
conn.row_factory = sqlite3.Row
|
||||
_seed_chernobyl(conn)
|
||||
yield conn
|
||||
conn.close()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_real_network(monkeypatch):
|
||||
"""Even if a fetcher is mis-wired, never reach a real router."""
|
||||
|
||||
def _guard(base_url):
|
||||
raise AssertionError(f"real fetch_metrics called with {base_url!r}")
|
||||
|
||||
monkeypatch.setattr(tui, "fetch_metrics", _guard)
|
||||
|
||||
|
||||
def _run_app(app: tui.DashboardApp, body) -> None:
|
||||
async def _go():
|
||||
async with app.run_test() as pilot:
|
||||
await pilot.pause()
|
||||
body(app)
|
||||
|
||||
asyncio.run(_go())
|
||||
|
||||
|
||||
def test_every_registered_warning_class_fires(chernobyl):
|
||||
"""Each registered class is provoked by the fixture.
|
||||
|
||||
A failure here means either the fixture rotted or a warning's message
|
||||
shape changed under the matcher — both silently disable the tripwire.
|
||||
"""
|
||||
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
|
||||
for name, pattern in WARN_CLASS_MATCHERS.items():
|
||||
assert any(re.search(pattern, w) for w in warnings), (
|
||||
f"registered warning class {name!r} did not fire.\n"
|
||||
f"Either the chernobyl fixture no longer provokes it, or the "
|
||||
f"message shape changed and the matcher needs updating.\n"
|
||||
f"Pattern: {pattern}\nEmitted:\n " + "\n ".join(warnings)
|
||||
)
|
||||
|
||||
|
||||
def test_every_emitted_warning_belongs_to_a_registered_class(chernobyl):
|
||||
"""No warning class exists without a decision about surfacing it.
|
||||
|
||||
This is the direction that catches the vision-ceiling failure mode: a new
|
||||
class added to scoring_coverage that nothing displays. It fails with the
|
||||
RAW text because the whole point is that nobody knew it existed.
|
||||
"""
|
||||
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
|
||||
unmatched = [
|
||||
w
|
||||
for w in warnings
|
||||
if not any(re.search(p, w) for p in WARN_CLASS_MATCHERS.values())
|
||||
]
|
||||
assert not unmatched, (
|
||||
"a warning class was added to metrics.scoring_coverage without a "
|
||||
"surfacing decision. Register a matcher in WARN_CLASS_MATCHERS (and a "
|
||||
"chernobyl seed if it needs one).\nUNMATCHED:\n "
|
||||
+ "\n ".join(unmatched)
|
||||
)
|
||||
|
||||
|
||||
def test_registered_warnings_render_in_the_panel(chernobyl):
|
||||
"""The panel actually shows them — computing is not surfacing.
|
||||
|
||||
Drives the real app with the chernobyl warnings so the assertion covers
|
||||
the render path, not just the metrics call.
|
||||
"""
|
||||
warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"]
|
||||
|
||||
from tests.test_tui import _fixture # noqa: PLC0415 — shared payload shape
|
||||
|
||||
payload = copy.deepcopy(_fixture())
|
||||
payload["coverage"]["warnings"] = warnings
|
||||
|
||||
class _Stub:
|
||||
def __call__(self, base_url):
|
||||
return payload
|
||||
|
||||
app = tui.DashboardApp(fetcher=_Stub())
|
||||
|
||||
def _assert(a):
|
||||
panel = str(a.query_one("#warnings-panel", Static).content)
|
||||
for name, pattern in WARN_CLASS_MATCHERS.items():
|
||||
assert re.search(pattern, panel), (
|
||||
f"warning class {name!r} is emitted by /metrics but does not "
|
||||
f"render in #warnings-panel — computed but not surfaced, which "
|
||||
f"is exactly the 2026-09-04 failure.\nPanel:\n{panel}"
|
||||
)
|
||||
|
||||
_run_app(app, _assert)
|
||||
Reference in New Issue
Block a user