diff --git a/src/metrics.py b/src/metrics.py index 0d948b9..3f0054a 100644 --- a/src/metrics.py +++ b/src/metrics.py @@ -877,7 +877,11 @@ def recent_decisions( conn: sqlite3.Connection, limit: int = 50, ) -> List[dict]: - """Last *N* rows from route_decisions, ordered by id DESC.""" + """Last *N* rows from route_decisions, ordered by id DESC. + + The row is the field set the TUI consumes; it is pinned by + tests/test_tui_schema_drift.py. + """ return [ dict(row) for row in conn.execute( @@ -889,6 +893,7 @@ def recent_decisions( runner_up_models, est_cost_usd, est_proficiency, rejected_reason, session_key, tools, images, json_mode, streamed, flex_preference, flex_swapped, flex_forced, + exploration, request_id, pinch_original_tokens, pinch_final_tokens, profile FROM route_decisions ORDER BY id DESC diff --git a/src/tui.py b/src/tui.py index fca5d1f..8f3a761 100644 --- a/src/tui.py +++ b/src/tui.py @@ -24,6 +24,7 @@ from __future__ import annotations import logging import os +from datetime import datetime from typing import Callable, Optional logger = logging.getLogger(__name__) @@ -57,6 +58,43 @@ def _fmt_usd(v: Optional[float]) -> str: return f"{v:.6f}" +def _fmt_runway(hours: float) -> str: + """Runway as hours below two days, days above. Money-scale, not micro.""" + if hours < 48: + return f"~{hours:.1f}h" + return f"~{hours / 24:.1f}d" + + +def _format_quota_lead( + balance: Optional[float], + burn: Optional[float], + runway_hours: Optional[float], + low_warning: bool, +) -> str: + """The quota panel's headline: balance, burn rate, projected runway. + + Deliberately 2dp rather than ``_fmt_usd``'s 6 — this is money an operator + acts on, not per-request micro-money. + + ``burn`` and ``runway_hours`` are legitimately ``None`` when the sample + segment is too short to extrapolate (PR #30's guard against a wild rate + right after a top-up). That is a real answer, not missing data, so it + renders "burn n/a" and the caller shows ``runway_note`` alongside. The + literal "None" must never reach the screen. + """ + if balance is None: + return "balance n/a" + parts = [f"balance ${balance:.2f} left"] + parts.append(f"burn ${burn:.2f}/h" if burn is not None else "burn n/a") + if runway_hours is not None: + runway = f"runway {_fmt_runway(float(runway_hours))}" + # Same markup idiom the legend already uses for `resets`. + parts.append( + f"[rgb(200,80,80)]{runway}[/rgb(200,80,80)]" if low_warning else runway + ) + return " · ".join(parts) + + # Compact bottom keys legend. The 11 BINDINGS are collapsed to these short # (key, label) pairs — duplicate quit bindings (q/Q/ctrl+c) show once. Rendered # as a single markup string so Textual's Static wraps rather than truncates. @@ -95,6 +133,47 @@ def _flex_indicator(r: dict) -> str: return str(pref) +def _short_time(value: object) -> str: + """``HH:MM:SS`` in LOCAL time from an ISO-8601 ``observed_at``. + + The decision feed is dense and almost always same-day, so the date would + crowd out ``selected`` — the column operators actually read. Local rather + than UTC because this is a live feed watched next to a wall clock. + + Returns ``""`` for a missing or unparseable value rather than raising: a + malformed timestamp must not take down the whole table render. + """ + if not value: + return "" + try: + parsed = datetime.fromisoformat(str(value)) + except (TypeError, ValueError): + return "" + if parsed.tzinfo is not None: + parsed = parsed.astimezone() + return parsed.strftime("%H:%M:%S") + + +def _exploration_flag(r: dict) -> str: + """``E`` when epsilon-greedy picked this row instead of the ranking. + + An exploratory pick is a deliberate random sample, NOT the router's + judgement. Without a visible marker an operator reads it as the ranking's + answer and draws the wrong conclusion about model quality. + """ + return "E" if r.get("exploration") else "" + + +def _flags_indicator(r: dict) -> str: + """The combined flags cell: flex label and exploration marker. + + One cell rather than two columns — the terminal is not wide, and the spec + caps the table at one added column plus one flag. + """ + parts = [part for part in (_flex_indicator(r), _exploration_flag(r)) if part] + return " ".join(parts) + + class DashboardApp(App): """A terminal dashboard that reads GET /metrics and renders panels.""" @@ -155,6 +234,12 @@ class DashboardApp(App): #quota-progress { width: 100%; } + #quota-lead { + text-style: bold; + } + #quota-note { + color: $text-muted; + } #quota-legend { color: $text-muted; } @@ -207,6 +292,12 @@ class DashboardApp(App): yield Static("", id="loading-panel") yield Static("Quota burn", classes="panel-title") with Vertical(id="quota-panel"): + # Balance and runway lead; the plan-percentage bar is demoted + # below them. A bar against a routinely-exceeded plan implies + # a ceiling that does not exist — the credit balance is the + # real one (PR #30). + yield Static(markup=True, id="quota-lead") + yield Static(id="quota-note") yield ProgressBar(id="quota-progress") yield Static(markup=True, id="quota-legend") yield Static("Recent decisions (enter = details)", classes="panel-title") @@ -312,7 +403,16 @@ class DashboardApp(App): model_table.add_columns("model", "calls", "cost $", "kWh", "gCO2eq") decision_table = self.query_one("#decision-table", DataTable) decision_table.add_columns( - "id", "kind", "category", "tier", "ctx", "selected", "est $", "flex" + "time", + "id", + "kind", + "category", + "tier", + "ctx", + "profile", + "selected", + "est $", + "flags", ) decision_table.cursor_type = "row" decision_table.zebra_stripes = True @@ -509,6 +609,37 @@ class DashboardApp(App): metered = rows.get("metered_kwh_30d") frac = rows.get("fraction") calls = rows.get("calls") + + # Balance/runway lead. Wired OUTSIDE the plan gate on purpose: the + # credit balance is meaningful whether or not a kWh plan is set, and + # it is the figure that actually stops traffic. + lead = self.query_one("#quota-lead", Static) + note = self.query_one("#quota-note", Static) + if rows: + balance = rows.get("balance_usd") + burn = rows.get("burn_rate_usd_per_hour") + lead.update( + _format_quota_lead( + balance, + burn, + rows.get("projected_hours_remaining"), + bool(rows.get("runway_low_warning")), + ) + ) + # Explain an absent burn rate rather than leaving a bare "n/a". + if balance is not None and burn is None: + note.update( + str(rows.get("runway_note") or "burn estimate unavailable") + ) + note.display = True + else: + note.update("") + note.display = False + else: + lead.update("") + note.update("") + note.display = False + if plan is not None and float(plan) > 0: bar.total = float(plan) bar.progress = float(metered or 0) @@ -590,18 +721,23 @@ class DashboardApp(App): dt.clear() for r in self._last_model["recent_decisions"]: dt.add_row( + _short_time(r.get("observed_at")), str(r.get("id")), str(r.get("kind")), str(r.get("category")), str(r.get("tier")), str(r.get("required_context_tokens")), + # `or ""` so an absent profile is blank, never the literal "None". + str(r.get("profile") or ""), str(r.get("selected")), _fmt_usd(r.get("est_cost_usd")), - _flex_indicator(r), + _flags_indicator(r), key=str(r.get("id")), ) if not self._last_model["recent_decisions"]: - dt.add_row("(no decisions)", "", "", "", "", "", "", "", key="_placeholder") + dt.add_row( + "(no decisions)", "", "", "", "", "", "", "", "", "", key="_placeholder" + ) self._restore_cursor(dt, saved_decision_key) def _render_category_table(self) -> None: diff --git a/src/tui_model.py b/src/tui_model.py index 9f3a293..9ef672e 100644 --- a/src/tui_model.py +++ b/src/tui_model.py @@ -143,6 +143,12 @@ def decision_row(r: dict) -> dict: "flex_preference": r.get("flex_preference"), "flex_swapped": r.get("flex_swapped"), "flex_forced": r.get("flex_forced"), + "profile": r.get("profile"), + "exploration": r.get("exploration"), + "pinch_original_tokens": r.get("pinch_original_tokens"), + "pinch_final_tokens": r.get("pinch_final_tokens"), + "request_id": r.get("request_id"), + "session_key": r.get("session_key"), } diff --git a/tests/test_route_decisions.py b/tests/test_route_decisions.py index 3588f59..d22cec1 100644 --- a/tests/test_route_decisions.py +++ b/tests/test_route_decisions.py @@ -59,6 +59,9 @@ ROUTE_DECISIONS_COLUMNS = [ "flex_preference", "flex_swapped", "flex_forced", + # Enforcement for this list lives in tests/test_tui_schema_drift.py — + # keep new route_decisions columns registered there, not just here. + "request_id", "exploration", "pinch_original_tokens", "pinch_final_tokens", diff --git a/tests/test_tui.py b/tests/test_tui.py index 44c1ea5..9873d7f 100644 --- a/tests/test_tui.py +++ b/tests/test_tui.py @@ -11,7 +11,9 @@ file that may. No non-tui module imports textual. from __future__ import annotations import asyncio +import copy import time +from datetime import datetime import pytest import requests @@ -20,7 +22,7 @@ from textual.widgets import ProgressBar, Static import tui import tui_model from tui_model import build_model, fetch_metrics -from tui_screens import VerdictMixScreen +from tui_screens import DecisionDetailScreen, VerdictMixScreen def _fixture() -> dict: @@ -144,6 +146,46 @@ def _fixture() -> dict: } +def _enriched_payload() -> dict: + """A deep copy of _fixture() whose recent_decisions rows carry the six + T2 data-layer fields, mirroring what /metrics returns after the drift + catch-up. Row 42 is the popup test's target row (exploration off); + row 41 is the exploratory pick; row 40 stays a rejection row.""" + data = copy.deepcopy(_fixture()) + enrichments = [ + { + "profile": "default", + "exploration": 0, + "request_id": "chatcmpl-fixture-42", + "pinch_original_tokens": 120000, + "pinch_final_tokens": 96122, + "session_key": "sess-42", + "observed_at": "2026-08-23T10:00:00+00:00", + }, + { + "profile": "default", + "exploration": 1, + "request_id": "chatcmpl-fixture-41", + "pinch_original_tokens": 90000, + "pinch_final_tokens": 87500, + "session_key": "sess-41", + "observed_at": "2026-08-23T09:59:00+00:00", + }, + { + "profile": "default", + "exploration": 0, + "request_id": None, + "pinch_original_tokens": None, + "pinch_final_tokens": None, + "session_key": "sess-40", + "observed_at": "2026-08-23T09:58:00+00:00", + }, + ] + for row, extra in zip(data["recent_decisions"], enrichments): + row.update(extra) + return data + + # -------------------------------------------------------------------------- # Direct unit tests of the data layer (no TUI running). # -------------------------------------------------------------------------- @@ -1056,6 +1098,78 @@ def test_show_decision_detail_pushes_modal_with_full_row(): asyncio.run(_go()) +def test_detail_popup_carries_forensics_fields(): + """The e-key popup carries the six T2 forensics fields — profile, + exploration, pinch tokens, request_id, session_key — both on the live + decision dict and in the rendered JSON for a metrics-shaped full row.""" + stub = _StubFetcher() + stub.payload = _enriched_payload() + app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) + + async def _go(): + async with app.run_test() as pilot: + await pilot.pause() + app.query_one("#decision-table").focus() + await pilot.pause() + await pilot.press("e") + await pilot.pause() + from textual.screen import ModalScreen + + assert isinstance(app.screen, ModalScreen) + decision = app.screen.decision + assert decision["request_id"] == "chatcmpl-fixture-42" + assert decision["session_key"] == "sess-42" + assert decision["pinch_original_tokens"] == 120000 + assert decision["pinch_final_tokens"] == 96122 + assert decision["profile"] == "default" + assert decision["exploration"] == 0 + + asyncio.run(_go()) + + metrics_row = { + "id": 43, + "observed_at": "2026-09-05T12:34:56+00:00", + "kind": "chat", + "task_category": "coding_general", + "task_tier": 2, + "required_context_tokens": 123456, + "confidence": 0.91, + "classifier_ms": 1800, + "classification_source": "classifier", + "latency_tolerance": "interactive", + "candidates_considered": 7, + "selected_model": "fixture-model", + "selected_provider": "neuralwatt", + "runner_up_models": None, + "est_cost_usd": 0.001, + "est_proficiency": 0.9, + "rejected_reason": None, + "session_key": "sess-43", + "tools": 0, + "images": 0, + "json_mode": 0, + "streamed": 1, + "flex_preference": "auto", + "flex_swapped": 0, + "flex_forced": 0, + "request_id": "chatcmpl-fixture-43", + "exploration": 1, + "pinch_original_tokens": 120000, + "pinch_final_tokens": 96122, + "profile": "default", + } + text = DecisionDetailScreen(tui_model.decision_row(metrics_row))._render_text() + for key in ( + "profile", + "exploration", + "pinch_original_tokens", + "pinch_final_tokens", + "request_id", + "session_key", + ): + assert key in text + + def test_app_run_test_ctrl_v_opens_verdict_popup(): """Pressing ``ctrl+v`` pushes a VerdictMixScreen modal showing the verdict mix rows from the current model, and ``escape`` dismisses it back to the @@ -1161,6 +1275,46 @@ def test_live_decision_inserts_row_at_front_and_rerenders(): asyncio.run(_go()) +def test_live_decision_carries_new_fields(): + """A live SSE decision keeps the six T2 fields through the shared + decision_row projection, so a prompt `e` on a fresh decision shows the + same forensics as a /metrics-sourced row.""" + stub = _StubFetcher() + stub.payload = _fixture() + app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) + + async def _go(): + async with app.run_test() as pilot: + await pilot.pause() + app._handle_live_decision( + { + "id": 1001, + "kind": "chat", + "task_category": "coding_general", + "task_tier": 2, + "selected_model": "deepseek-v4-flash", + "est_cost_usd": 0.0002, + "profile": "default", + "exploration": 1, + "request_id": "chatcmpl-live-1001", + "pinch_original_tokens": 120000, + "pinch_final_tokens": 96122, + "session_key": "sess-live-1001", + } + ) + await pilot.pause() + row = app._last_model["recent_decisions"][0] + assert row["id"] == 1001 + assert row["profile"] == "default" + assert row["exploration"] == 1 + assert row["request_id"] == "chatcmpl-live-1001" + assert row["pinch_original_tokens"] == 120000 + assert row["pinch_final_tokens"] == 96122 + assert row["session_key"] == "sess-live-1001" + + asyncio.run(_go()) + + def test_live_decision_caps_recent_decisions_at_fifty(): """The live feed never grows the in-memory list past the /metrics cap, so the dashboard's view stays consistent with a /metrics refresh.""" @@ -1404,3 +1558,219 @@ def test_format_quota_legend_shows_next_reset_date(): assert "2026-07-26" in legend # window start still shown assert "2026-09-06" in legend # billing reset assert "resets" in legend.lower() + + +# -------------------------------------------------------------------------- +# T3 — decision table display: time column, profile column, exploration flag. +# -------------------------------------------------------------------------- + + +def test_decision_table_columns_and_order(): + """The table declares exactly ten columns with `time` first. + + Time leads because it is the natural scan axis for a live feed; `flex` + became `flags` because exploration now shares that cell. + """ + stub = _StubFetcher() + stub.payload = _enriched_payload() + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + dt = a.query_one("#decision-table") + labels = [str(c.label) for c in dt.columns.values()] + assert labels == [ + "time", + "id", + "kind", + "category", + "tier", + "ctx", + "profile", + "selected", + "est $", + "flags", + ] + + _run_app(app, _assert) + + +def test_short_time_renders_hhmmss_and_degrades_quietly(): + """`_short_time` is HH:MM:SS local, and never raises on bad input. + + A malformed timestamp must not take down the whole table render, so the + unparseable cases return "" rather than propagating ValueError. + """ + rendered = tui._short_time("2026-08-23T10:00:00+00:00") + assert len(rendered) == 8 and rendered.count(":") == 2 + # Local conversion, computed the same way the helper does it. + expected = ( + datetime.fromisoformat("2026-08-23T10:00:00+00:00") + .astimezone() + .strftime("%H:%M:%S") + ) + assert rendered == expected + # No date component leaks into the cell. + assert "2026" not in rendered + for bad in (None, "", "not-a-timestamp", 12345): + assert tui._short_time(bad) == "" + + +def test_app_decision_table_profile_column_blank_not_none(): + """`profile` renders its value, and renders BLANK when absent. + + A literal "None" in a dense table reads as a real profile name. + """ + stub = _StubFetcher() + data = _enriched_payload() + data["recent_decisions"][1]["profile"] = None + stub.payload = data + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + dt = a.query_one("#decision-table") + labels = [str(c.label) for c in dt.columns.values()] + profile_at = labels.index("profile") + assert str(dt.get_row_at(0)[profile_at]) == "default" + # Absent profile is an empty cell, never the literal "None". + assert str(dt.get_row_at(1)[profile_at]) == "" + + _run_app(app, _assert) + + +def test_exploration_flag_and_flags_indicator_compose(): + """`E` marks an exploratory pick and composes with the flex label. + + An epsilon-greedy pick is a random sample, not the ranking's judgement; + without the marker an operator reads it as the router's answer. + """ + assert tui._exploration_flag({"exploration": 1}) == "E" + assert tui._exploration_flag({"exploration": 0}) == "" + assert tui._exploration_flag({}) == "" + # Composition: flex label first, then the marker, space separated. + both = tui._flags_indicator( + {"flex_preference": "auto", "flex_swapped": 0, "exploration": 1} + ) + assert both == "auto E" + # Either alone survives; neither yields an empty cell. + assert tui._flags_indicator({"exploration": 1}) == "E" + assert tui._flags_indicator({"flex_preference": "auto"}) == "auto" + assert tui._flags_indicator({}) == "" + + +def test_app_decision_table_exploration_flag_renders(): + """Row 41 is the exploratory pick in the fixture; its flags cell shows E.""" + stub = _StubFetcher() + stub.payload = _enriched_payload() + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + dt = a.query_one("#decision-table") + exploratory = [str(c) for c in dt.get_row_at(1)] + assert any("E" == cell or cell.endswith(" E") for cell in exploratory) + non_exploratory = [str(c) for c in dt.get_row_at(0)] + assert not any(cell == "E" or cell.endswith(" E") for cell in non_exploratory) + + _run_app(app, _assert) + + +def test_placeholder_row_matches_column_arity(): + """The empty-state placeholder must have exactly ten cells. + + A short placeholder row raises inside Textual at render time — an arity + mismatch is a crash, not a cosmetic issue. + """ + stub = _StubFetcher() + data = _enriched_payload() + data["recent_decisions"] = [] + stub.payload = data + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + dt = a.query_one("#decision-table") + assert len(dt.get_row_at(0)) == len(dt.columns) == 10 + + _run_app(app, _assert) + + +# -------------------------------------------------------------------------- +# T4 — quota panel leads with balance and runway. +# -------------------------------------------------------------------------- + + +def test_format_quota_lead_states(): + """All four lead states, including the one this deployment is actually in. + + burn/runway are legitimately None when the sample segment is too short + (PR #30). That is deliberate, not missing data, so the None path must + still say something an operator can read. + """ + full = tui._format_quota_lead(8.50, 1.0, 8.5, False) + assert "balance $8.50 left" in full + assert "burn $1.00/h" in full + assert "runway ~8.5h" in full + + none_burn = tui._format_quota_lead(8.50, None, None, False) + assert "burn n/a" in none_burn + assert "None" not in none_burn + + assert tui._format_quota_lead(None, 1.0, 8.5, False) == "balance n/a" + assert "runway ~20.8d" in tui._format_quota_lead(100.0, 1.0, 500.0, False) + + low = tui._format_quota_lead(8.50, 4.0, 1.5, True) + assert "[rgb(200,80,80)]runway ~1.5h" in low + + +def test_quota_panel_burn_unavailable_renders_note_verbatim(): + """When burn is None the note renders VERBATIM — never blank, 0, or None. + + This is the state the live deployment is in right now, so it is the first + thing an operator sees, not an edge case. + """ + stub = _StubFetcher() + data = _fixture() + note = "burn estimate unavailable: segment after last balance increase spans only 9 minutes" + for block in (data["quota"], data["coverage"]["quota"]): + block["balance_usd"] = 7.5 + block["burn_rate_usd_per_hour"] = None + block["projected_hours_remaining"] = None + block["runway_low_warning"] = False + block["runway_note"] = note + stub.payload = data + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + lead_text = str(a.query_one("#quota-lead", Static).content) + note_text = str(a.query_one("#quota-note", Static).content) + assert "balance $7.50 left" in lead_text + assert "burn n/a" in lead_text + assert "None" not in lead_text + # The note is passed through unchanged, not summarised. + assert note_text == note + assert note_text.strip() != "" + assert note_text.strip() != "0" + assert note_text.strip() != "None" + + _run_app(app, _assert) + + +def test_quota_panel_note_hidden_when_burn_is_available(): + """With a real burn rate there is nothing to explain, so no note line.""" + stub = _StubFetcher() + data = _fixture() + for block in (data["quota"], data["coverage"]["quota"]): + block["balance_usd"] = 12.0 + block["burn_rate_usd_per_hour"] = 0.5 + block["projected_hours_remaining"] = 24.0 + block["runway_low_warning"] = False + block["runway_note"] = None + stub.payload = data + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + lead_text = str(a.query_one("#quota-lead", Static).content) + assert "balance $12.00 left" in lead_text + assert "burn $0.50/h" in lead_text + assert "runway ~24.0h" in lead_text + assert a.query_one("#quota-note", Static).display is False + + _run_app(app, _assert) diff --git a/tests/test_tui_schema_drift.py b/tests/test_tui_schema_drift.py new file mode 100644 index 0000000..66e49e3 --- /dev/null +++ b/tests/test_tui_schema_drift.py @@ -0,0 +1,200 @@ +"""Schema-drift guard: every route_decisions column reaches the TUI on purpose. + +Five-plus columns shipped to ``route_decisions`` across recent merges without +any of them reaching the dashboard, and ``tests/test_route_decisions.py``'s own +hand-maintained column list had ALREADY drifted (missing ``request_id``) — the +drift recurs, so this file is the enforcement, not the catch-up. + +Two registries, not a magic diff: + +- ``SCHEMA_TO_MODEL_KEYS`` — schema column -> the ``tui_model.decision_row`` + key that surfaces it. This is the recorded DECISION about surfacing. Three + keys are intentionally renamed (``task_category`` -> ``category``, + ``task_tier`` -> ``tier``, ``selected_model`` -> ``selected``), which is + exactly why the mapping is explicit and never derived by string identity. +- ``UNSURFACED_COLUMNS`` — columns deliberately NOT surfaced, each with a + one-line reason. Empty today. + +Every new schema column must land in one of the two registries or +``test_schema_columns_all_have_surfacing_decisions`` fails naming the column; +every registry entry must actually round-trip through ``decision_row`` and the +``/metrics`` SELECT or the other two tests fail naming the key. + +Offline: temp SQLite created from ``config/schema.sql``, never the live +router.db (pattern borrowed from tests/test_route_decisions.py). +""" + +import sqlite3 +from pathlib import Path + +import metrics +import tui_model + +ROOT = Path(__file__).resolve().parent.parent +SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text() + +# route_decisions column -> the tui_model.decision_row key that surfaces it. +# Schema order (config/schema.sql route_decisions declaration). Add a new +# schema column here when the TUI should surface it, or to +# UNSURFACED_COLUMNS when it deliberately should not. +SCHEMA_TO_MODEL_KEYS = { + "id": "id", + "observed_at": "observed_at", + "kind": "kind", + "task_category": "category", + "task_tier": "tier", + "required_context_tokens": "required_context_tokens", + "confidence": "confidence", + "classifier_ms": "classifier_ms", + "classification_source": "classification_source", + "latency_tolerance": "latency_tolerance", + "candidates_considered": "candidates_considered", + "selected_model": "selected", + "selected_provider": "selected_provider", + "runner_up_models": "runner_up_models", + "est_cost_usd": "est_cost_usd", + "est_proficiency": "est_proficiency", + "rejected_reason": "rejected_reason", + "session_key": "session_key", + "tools": "tools", + "images": "images", + "json_mode": "json_mode", + "streamed": "streamed", + "flex_preference": "flex_preference", + "flex_swapped": "flex_swapped", + "flex_forced": "flex_forced", + "request_id": "request_id", + "exploration": "exploration", + "pinch_original_tokens": "pinch_original_tokens", + "pinch_final_tokens": "pinch_final_tokens", + "profile": "profile", +} +# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a +# one-line reason (callers must keep the comment). Empty today: every column is +# surfaced via decision_row. A new schema column that lands in NEITHER registry +# fails the drift test. +UNSURFACED_COLUMNS: set[str] = set() + +# Every column seeded non-NULL: the single source of truth for the INSERT and +# the round-trip assertions, so the tests prove the column flows, not that +# None flows. +FULL_ROW = { + "id": 1, + "observed_at": "2026-09-05T12:34:56+00:00", + "kind": "chat", + "task_category": "coding_general", + "task_tier": 2, + "required_context_tokens": 123456, + "confidence": 0.91, + "classifier_ms": 1800, + "classification_source": "classifier", + "latency_tolerance": "interactive", + "candidates_considered": 7, + "selected_model": "fixture-model", + "selected_provider": "neuralwatt", + "runner_up_models": '[{"model_id": "runner", "provider": "neuralwatt"}]', + "est_cost_usd": 0.00123, + "est_proficiency": 0.88, + # Schema-legal even for a selected row (no CHECK constraint); non-NULL so the + # round-trip assertion proves the column flows, not that None flows. + "rejected_reason": "fixture rejected reason", + "session_key": "fixture-session-key", + "tools": 0, + "images": 0, + "json_mode": 1, + "streamed": 1, + "flex_preference": "auto", + "flex_swapped": 0, + "flex_forced": 0, + "request_id": "chatcmpl-fixture-42", + "exploration": 1, + "pinch_original_tokens": 120000, + "pinch_final_tokens": 96122, + "profile": "default", +} + + +def _db_from_schema(tmp_path, row_factory=None) -> sqlite3.Connection: + """A throwaway DB from config/schema.sql — never the live router.db. + + ``row_factory`` defaults to None (plain tuples) like the PRAGMA test + needs; the metrics round-trip test passes ``sqlite3.Row`` because + ``metrics.recent_decisions`` builds its rows via ``dict(row)``. + """ + conn = sqlite3.connect(tmp_path / "schema-drift.db") + conn.executescript(SCHEMA_SQL) + conn.row_factory = row_factory + return conn + + +def test_schema_columns_all_have_surfacing_decisions(tmp_path): + """Every route_decisions column is either surfaced or consciously not.""" + conn = _db_from_schema(tmp_path) + schema_cols = { + row[1] for row in conn.execute("PRAGMA table_info(route_decisions)") + } + conn.close() + + registered = set(SCHEMA_TO_MODEL_KEYS) | UNSURFACED_COLUMNS + missing = schema_cols - set(SCHEMA_TO_MODEL_KEYS) - UNSURFACED_COLUMNS + extra = registered - schema_cols + assert not missing, ( + f"route_decisions column(s) added without a surfacing decision: " + f"{sorted(missing)}. Add each to SCHEMA_TO_MODEL_KEYS if the TUI " + f"should surface it (project it in tui_model.decision_row), or to " + f"UNSURFACED_COLUMNS with a reason comment." + ) + assert not extra, ( + f"registry registers column(s) that no longer exist in " + f"route_decisions: {sorted(extra)}." + ) + + +def test_decision_row_projects_every_tracked_column(): + """decision_row surfaces every registered column's value faithfully.""" + projected = tui_model.decision_row(dict(FULL_ROW)) + + expected_keys = set(SCHEMA_TO_MODEL_KEYS.values()) + assert set(projected) == expected_keys, ( + "tui_model.decision_row no longer matches the surfacing registry " + f"(symmetric difference: {sorted(expected_keys ^ set(projected))}).\n" + f" registry values: {sorted(expected_keys)}\n" + f" decision_row keys: {sorted(projected)}" + ) + for col, key in SCHEMA_TO_MODEL_KEYS.items(): + assert projected[key] == FULL_ROW[col], ( + f"route_decisions column {col!r} must surface as decision_row " + f"key {key!r}: got {projected[key]!r}, want {FULL_ROW[col]!r}" + ) + + +def test_metrics_recent_decisions_selects_every_tracked_column(tmp_path): + """The /metrics path carries every registered column end-to-end. + + ``request_id`` reaches the TUI ONLY through this SELECT (SSE publishes + before the backfill), so a narrowed SELECT silently erases forensics + fields a few seconds after a live decision arrives. + """ + conn = _db_from_schema(tmp_path, row_factory=sqlite3.Row) + cols = ",".join(FULL_ROW.keys()) + placeholders = ",".join("?" * len(FULL_ROW)) + conn.execute( + f"INSERT INTO route_decisions ({cols}) VALUES ({placeholders})", + tuple(FULL_ROW.values()), + ) + conn.commit() + + rows = metrics.recent_decisions(conn, limit=5) + conn.close() + assert rows, "the seeded FULL_ROW must come back from recent_decisions" + projected = tui_model.decision_row(rows[0]) + + unreachable = [] + for col, key in SCHEMA_TO_MODEL_KEYS.items(): + if projected.get(key) != FULL_ROW[col]: + unreachable.append( + f"route_decisions column {col!r} does not reach the TUI " + f"through /metrics — is it selected by " + f"metrics.recent_decisions()?" + ) + assert not unreachable, "\n".join(unreachable) diff --git a/tests/test_tui_warnings.py b/tests/test_tui_warnings.py new file mode 100644 index 0000000..be8416d --- /dev/null +++ b/tests/test_tui_warnings.py @@ -0,0 +1,291 @@ +"""Tripwire: every warning class /metrics can emit must reach the panel. + +The 2026-09-04 vision-ceiling incident was invisible for ~19 hours because a +whole warning family was computed correctly and never displayed. The fix there +was surfacing, not computing — so the guard has to be a test that fails when a +warning class exists but nothing renders it. + +Two directions, both load-bearing: + +* every REGISTERED class actually fires against a fixture that provokes all of + them — catches the fixture rotting or a message shape changing underneath; +* every EMITTED warning is registered — catches a NEW class being added to + ``metrics.scoring_coverage`` without anyone deciding to surface it. This one + fails with the raw warning text, because the whole point is that nobody knew + the class existed. + +Deliberately no assertions on warning COUNT: the numbers shift as fixture +semantics evolve, and item 4 of the spec asks for class coverage, not arity. +""" +from __future__ import annotations + +import asyncio +import copy +import re +import sqlite3 +from datetime import datetime, timedelta, timezone +from pathlib import Path +from types import SimpleNamespace + +import pytest +from textual.widgets import Static + +import metrics +import tui + +ROOT = Path(__file__).resolve().parent.parent +SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text() + + +# The registry. A class here without a matching emitted warning means the +# fixture rotted; an emitted warning without a class here means someone added +# a warning nobody decided to display. +WARN_CLASS_MATCHERS = { + "quota-over-plan": r"metered usage is \d+% of the", + "models-missing-energy": r"\d+/\d+ routable models have no reference-workload", + "models-missing-proficiency": r"\d+/\d+ routable models have no proficiency", + "catalog-stale": r"catalog last polled", + # Fixed-width lookbehind so this cannot swallow the capability warnings, + # which share the "tier N context ceiling" tail. + "demand-ceiling": r"(? None: + """Seed a catalog and traffic that provoke ALL ten warning classes at once. + + Named for what it is: a deliberately maximally-broken deployment. Every + timestamp is relative to now so the rejection windows (1h alert, 24h + baseline) stay valid whenever the suite runs. + """ + now = datetime.now(timezone.utc) + stale = (now - timedelta(days=3)).isoformat() + + rows = [ + # Covered model: has proficiency AND energy, so the "missing" counts + # come out as 2/3 rather than 3/3. + _model_row("cov-t1", 1, 20000), + # Uncovered: drives both missing- classes; its 5000 window is also the + # tier-2 ceiling behind the escalation hazard. + _model_row("gap-t2", 2, 5000), + # Vision-capable but small: the capability sub-ceiling. + _model_row("vis-t1", 1, 5000, vision=1), + # Deprecated: excluded from routable, so tier 3 has zero eligible. + _model_row("dead-t3", 3, 1000, availability="deprecated"), + ] + for row in rows: + conn.execute( + """ + INSERT INTO models ( + model_id, provider, base_model_id, tier, context_window, + effective_context_window, max_output_tokens, + cost_per_1m_prompt, cost_per_1m_completion, + supports_vision, supports_json_mode, + latency_class, reasoning_mode, context_variant, + access_level, availability, last_updated + ) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?) + """, + (*row, stale), + ) + + conn.execute( + "INSERT INTO proficiency " + "(model_id, provider, category, blended_score, last_updated) " + "VALUES ('cov-t1','neuralwatt','coding_general',0.9,?)", + (now.isoformat(),), + ) + # 6.0 kWh against a 6.25 plan = 96% > 80%, firing the quota class. The + # 'seed_reference' task_category is also what marks a model as having + # reference-workload data, so this one row serves both purposes. + conn.execute( + "INSERT INTO energy_observations " + "(model_id, provider, energy_kwh, task_category, observed_at) " + "VALUES ('cov-t1','neuralwatt',6.0,'seed_reference',?)", + ((now - timedelta(days=1)).isoformat(),), + ) + + def _decision(**kw): + cols = { + "kind": "chat", + "task_category": "coding_general", + "task_tier": 1, + "required_context_tokens": 100, + "latency_tolerance": "interactive", + "selected_model": "cov-t1", + "selected_provider": "neuralwatt", + "observed_at": now.isoformat(), + "images": 0, + "json_mode": 0, + "rejected_reason": None, + } + cols.update(kw) + conn.execute( + f"INSERT INTO route_decisions ({','.join(cols)}) " + f"VALUES ({','.join('?' * len(cols))})", + tuple(cols.values()), + ) + + # One huge vision request: demand-ceiling (20000 < 999999), capability + # sub-ceiling (vision 5000 < 999999), and the escalation hazard all at once. + _decision(required_context_tokens=999999, images=1) + + # Two rejections of a pattern with no baseline -> "new rejection pattern:". + for _ in range(2): + _decision( + selected_model=None, + selected_provider=None, + rejected_reason="context >= 99999 tokens; interactive", + ) + + # A familiar group: one in the baseline window, six in the alert window, + # crossing the min-count threshold -> "rejection rate:". + familiar = "tier >= 2; context >= 30000 tokens; interactive" + _decision( + task_tier=2, + selected_model=None, + selected_provider=None, + rejected_reason=familiar, + observed_at=(now - timedelta(hours=2)).isoformat(), + ) + for _ in range(6): + _decision( + task_tier=2, + selected_model=None, + selected_provider=None, + rejected_reason=familiar, + ) + conn.commit() + + +@pytest.fixture() +def chernobyl(tmp_path): + conn = sqlite3.connect(tmp_path / "warn.db") + conn.executescript(SCHEMA_SQL) + conn.row_factory = sqlite3.Row + _seed_chernobyl(conn) + yield conn + conn.close() + + +@pytest.fixture(autouse=True) +def _no_real_network(monkeypatch): + """Even if a fetcher is mis-wired, never reach a real router.""" + + def _guard(base_url): + raise AssertionError(f"real fetch_metrics called with {base_url!r}") + + monkeypatch.setattr(tui, "fetch_metrics", _guard) + + +def _run_app(app: tui.DashboardApp, body) -> None: + async def _go(): + async with app.run_test() as pilot: + await pilot.pause() + body(app) + + asyncio.run(_go()) + + +def test_every_registered_warning_class_fires(chernobyl): + """Each registered class is provoked by the fixture. + + A failure here means either the fixture rotted or a warning's message + shape changed under the matcher — both silently disable the tripwire. + """ + warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] + for name, pattern in WARN_CLASS_MATCHERS.items(): + assert any(re.search(pattern, w) for w in warnings), ( + f"registered warning class {name!r} did not fire.\n" + f"Either the chernobyl fixture no longer provokes it, or the " + f"message shape changed and the matcher needs updating.\n" + f"Pattern: {pattern}\nEmitted:\n " + "\n ".join(warnings) + ) + + +def test_every_emitted_warning_belongs_to_a_registered_class(chernobyl): + """No warning class exists without a decision about surfacing it. + + This is the direction that catches the vision-ceiling failure mode: a new + class added to scoring_coverage that nothing displays. It fails with the + RAW text because the whole point is that nobody knew it existed. + """ + warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] + unmatched = [ + w + for w in warnings + if not any(re.search(p, w) for p in WARN_CLASS_MATCHERS.values()) + ] + assert not unmatched, ( + "a warning class was added to metrics.scoring_coverage without a " + "surfacing decision. Register a matcher in WARN_CLASS_MATCHERS (and a " + "chernobyl seed if it needs one).\nUNMATCHED:\n " + + "\n ".join(unmatched) + ) + + +def test_registered_warnings_render_in_the_panel(chernobyl): + """The panel actually shows them — computing is not surfacing. + + Drives the real app with the chernobyl warnings so the assertion covers + the render path, not just the metrics call. + """ + warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] + + from tests.test_tui import _fixture # noqa: PLC0415 — shared payload shape + + payload = copy.deepcopy(_fixture()) + payload["coverage"]["warnings"] = warnings + + class _Stub: + def __call__(self, base_url): + return payload + + app = tui.DashboardApp(fetcher=_Stub()) + + def _assert(a): + panel = str(a.query_one("#warnings-panel", Static).content) + for name, pattern in WARN_CLASS_MATCHERS.items(): + assert re.search(pattern, panel), ( + f"warning class {name!r} is emitted by /metrics but does not " + f"render in #warnings-panel — computed but not surfaced, which " + f"is exactly the 2026-09-04 failure.\nPanel:\n{panel}" + ) + + _run_app(app, _assert)