From 537eb5ac87f6ca20020f71c396522e915044c67b Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 5 Sep 2026 02:09:34 -0400 Subject: [PATCH 1/4] feat(tui): surface six drifted route_decisions columns behind a schema drift test --- src/metrics.py | 7 +- src/tui_model.py | 6 + tests/test_route_decisions.py | 3 + tests/test_tui.py | 155 ++++++++++++++++++++++++- tests/test_tui_schema_drift.py | 200 +++++++++++++++++++++++++++++++++ 5 files changed, 369 insertions(+), 2 deletions(-) create mode 100644 tests/test_tui_schema_drift.py diff --git a/src/metrics.py b/src/metrics.py index 0d948b9..3f0054a 100644 --- a/src/metrics.py +++ b/src/metrics.py @@ -877,7 +877,11 @@ def recent_decisions( conn: sqlite3.Connection, limit: int = 50, ) -> List[dict]: - """Last *N* rows from route_decisions, ordered by id DESC.""" + """Last *N* rows from route_decisions, ordered by id DESC. + + The row is the field set the TUI consumes; it is pinned by + tests/test_tui_schema_drift.py. + """ return [ dict(row) for row in conn.execute( @@ -889,6 +893,7 @@ def recent_decisions( runner_up_models, est_cost_usd, est_proficiency, rejected_reason, session_key, tools, images, json_mode, streamed, flex_preference, flex_swapped, flex_forced, + exploration, request_id, pinch_original_tokens, pinch_final_tokens, profile FROM route_decisions ORDER BY id DESC diff --git a/src/tui_model.py b/src/tui_model.py index 9f3a293..9ef672e 100644 --- a/src/tui_model.py +++ b/src/tui_model.py @@ -143,6 +143,12 @@ def decision_row(r: dict) -> dict: "flex_preference": r.get("flex_preference"), "flex_swapped": r.get("flex_swapped"), "flex_forced": r.get("flex_forced"), + "profile": r.get("profile"), + "exploration": r.get("exploration"), + "pinch_original_tokens": r.get("pinch_original_tokens"), + "pinch_final_tokens": r.get("pinch_final_tokens"), + "request_id": r.get("request_id"), + "session_key": r.get("session_key"), } diff --git a/tests/test_route_decisions.py b/tests/test_route_decisions.py index 3588f59..d22cec1 100644 --- a/tests/test_route_decisions.py +++ b/tests/test_route_decisions.py @@ -59,6 +59,9 @@ ROUTE_DECISIONS_COLUMNS = [ "flex_preference", "flex_swapped", "flex_forced", + # Enforcement for this list lives in tests/test_tui_schema_drift.py — + # keep new route_decisions columns registered there, not just here. + "request_id", "exploration", "pinch_original_tokens", "pinch_final_tokens", diff --git a/tests/test_tui.py b/tests/test_tui.py index 44c1ea5..d8cd496 100644 --- a/tests/test_tui.py +++ b/tests/test_tui.py @@ -11,6 +11,7 @@ file that may. No non-tui module imports textual. from __future__ import annotations import asyncio +import copy import time import pytest @@ -20,7 +21,7 @@ from textual.widgets import ProgressBar, Static import tui import tui_model from tui_model import build_model, fetch_metrics -from tui_screens import VerdictMixScreen +from tui_screens import DecisionDetailScreen, VerdictMixScreen def _fixture() -> dict: @@ -144,6 +145,46 @@ def _fixture() -> dict: } +def _enriched_payload() -> dict: + """A deep copy of _fixture() whose recent_decisions rows carry the six + T2 data-layer fields, mirroring what /metrics returns after the drift + catch-up. Row 42 is the popup test's target row (exploration off); + row 41 is the exploratory pick; row 40 stays a rejection row.""" + data = copy.deepcopy(_fixture()) + enrichments = [ + { + "profile": "default", + "exploration": 0, + "request_id": "chatcmpl-fixture-42", + "pinch_original_tokens": 120000, + "pinch_final_tokens": 96122, + "session_key": "sess-42", + "observed_at": "2026-08-23T10:00:00+00:00", + }, + { + "profile": "default", + "exploration": 1, + "request_id": "chatcmpl-fixture-41", + "pinch_original_tokens": 90000, + "pinch_final_tokens": 87500, + "session_key": "sess-41", + "observed_at": "2026-08-23T09:59:00+00:00", + }, + { + "profile": "default", + "exploration": 0, + "request_id": None, + "pinch_original_tokens": None, + "pinch_final_tokens": None, + "session_key": "sess-40", + "observed_at": "2026-08-23T09:58:00+00:00", + }, + ] + for row, extra in zip(data["recent_decisions"], enrichments): + row.update(extra) + return data + + # -------------------------------------------------------------------------- # Direct unit tests of the data layer (no TUI running). # -------------------------------------------------------------------------- @@ -1056,6 +1097,78 @@ def test_show_decision_detail_pushes_modal_with_full_row(): asyncio.run(_go()) +def test_detail_popup_carries_forensics_fields(): + """The e-key popup carries the six T2 forensics fields — profile, + exploration, pinch tokens, request_id, session_key — both on the live + decision dict and in the rendered JSON for a metrics-shaped full row.""" + stub = _StubFetcher() + stub.payload = _enriched_payload() + app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) + + async def _go(): + async with app.run_test() as pilot: + await pilot.pause() + app.query_one("#decision-table").focus() + await pilot.pause() + await pilot.press("e") + await pilot.pause() + from textual.screen import ModalScreen + + assert isinstance(app.screen, ModalScreen) + decision = app.screen.decision + assert decision["request_id"] == "chatcmpl-fixture-42" + assert decision["session_key"] == "sess-42" + assert decision["pinch_original_tokens"] == 120000 + assert decision["pinch_final_tokens"] == 96122 + assert decision["profile"] == "default" + assert decision["exploration"] == 0 + + asyncio.run(_go()) + + metrics_row = { + "id": 43, + "observed_at": "2026-09-05T12:34:56+00:00", + "kind": "chat", + "task_category": "coding_general", + "task_tier": 2, + "required_context_tokens": 123456, + "confidence": 0.91, + "classifier_ms": 1800, + "classification_source": "classifier", + "latency_tolerance": "interactive", + "candidates_considered": 7, + "selected_model": "fixture-model", + "selected_provider": "neuralwatt", + "runner_up_models": None, + "est_cost_usd": 0.001, + "est_proficiency": 0.9, + "rejected_reason": None, + "session_key": "sess-43", + "tools": 0, + "images": 0, + "json_mode": 0, + "streamed": 1, + "flex_preference": "auto", + "flex_swapped": 0, + "flex_forced": 0, + "request_id": "chatcmpl-fixture-43", + "exploration": 1, + "pinch_original_tokens": 120000, + "pinch_final_tokens": 96122, + "profile": "default", + } + text = DecisionDetailScreen(tui_model.decision_row(metrics_row))._render_text() + for key in ( + "profile", + "exploration", + "pinch_original_tokens", + "pinch_final_tokens", + "request_id", + "session_key", + ): + assert key in text + + def test_app_run_test_ctrl_v_opens_verdict_popup(): """Pressing ``ctrl+v`` pushes a VerdictMixScreen modal showing the verdict mix rows from the current model, and ``escape`` dismisses it back to the @@ -1161,6 +1274,46 @@ def test_live_decision_inserts_row_at_front_and_rerenders(): asyncio.run(_go()) +def test_live_decision_carries_new_fields(): + """A live SSE decision keeps the six T2 fields through the shared + decision_row projection, so a prompt `e` on a fresh decision shows the + same forensics as a /metrics-sourced row.""" + stub = _StubFetcher() + stub.payload = _fixture() + app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) + + async def _go(): + async with app.run_test() as pilot: + await pilot.pause() + app._handle_live_decision( + { + "id": 1001, + "kind": "chat", + "task_category": "coding_general", + "task_tier": 2, + "selected_model": "deepseek-v4-flash", + "est_cost_usd": 0.0002, + "profile": "default", + "exploration": 1, + "request_id": "chatcmpl-live-1001", + "pinch_original_tokens": 120000, + "pinch_final_tokens": 96122, + "session_key": "sess-live-1001", + } + ) + await pilot.pause() + row = app._last_model["recent_decisions"][0] + assert row["id"] == 1001 + assert row["profile"] == "default" + assert row["exploration"] == 1 + assert row["request_id"] == "chatcmpl-live-1001" + assert row["pinch_original_tokens"] == 120000 + assert row["pinch_final_tokens"] == 96122 + assert row["session_key"] == "sess-live-1001" + + asyncio.run(_go()) + + def test_live_decision_caps_recent_decisions_at_fifty(): """The live feed never grows the in-memory list past the /metrics cap, so the dashboard's view stays consistent with a /metrics refresh.""" diff --git a/tests/test_tui_schema_drift.py b/tests/test_tui_schema_drift.py new file mode 100644 index 0000000..66e49e3 --- /dev/null +++ b/tests/test_tui_schema_drift.py @@ -0,0 +1,200 @@ +"""Schema-drift guard: every route_decisions column reaches the TUI on purpose. + +Five-plus columns shipped to ``route_decisions`` across recent merges without +any of them reaching the dashboard, and ``tests/test_route_decisions.py``'s own +hand-maintained column list had ALREADY drifted (missing ``request_id``) — the +drift recurs, so this file is the enforcement, not the catch-up. + +Two registries, not a magic diff: + +- ``SCHEMA_TO_MODEL_KEYS`` — schema column -> the ``tui_model.decision_row`` + key that surfaces it. This is the recorded DECISION about surfacing. Three + keys are intentionally renamed (``task_category`` -> ``category``, + ``task_tier`` -> ``tier``, ``selected_model`` -> ``selected``), which is + exactly why the mapping is explicit and never derived by string identity. +- ``UNSURFACED_COLUMNS`` — columns deliberately NOT surfaced, each with a + one-line reason. Empty today. + +Every new schema column must land in one of the two registries or +``test_schema_columns_all_have_surfacing_decisions`` fails naming the column; +every registry entry must actually round-trip through ``decision_row`` and the +``/metrics`` SELECT or the other two tests fail naming the key. + +Offline: temp SQLite created from ``config/schema.sql``, never the live +router.db (pattern borrowed from tests/test_route_decisions.py). +""" + +import sqlite3 +from pathlib import Path + +import metrics +import tui_model + +ROOT = Path(__file__).resolve().parent.parent +SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text() + +# route_decisions column -> the tui_model.decision_row key that surfaces it. +# Schema order (config/schema.sql route_decisions declaration). Add a new +# schema column here when the TUI should surface it, or to +# UNSURFACED_COLUMNS when it deliberately should not. +SCHEMA_TO_MODEL_KEYS = { + "id": "id", + "observed_at": "observed_at", + "kind": "kind", + "task_category": "category", + "task_tier": "tier", + "required_context_tokens": "required_context_tokens", + "confidence": "confidence", + "classifier_ms": "classifier_ms", + "classification_source": "classification_source", + "latency_tolerance": "latency_tolerance", + "candidates_considered": "candidates_considered", + "selected_model": "selected", + "selected_provider": "selected_provider", + "runner_up_models": "runner_up_models", + "est_cost_usd": "est_cost_usd", + "est_proficiency": "est_proficiency", + "rejected_reason": "rejected_reason", + "session_key": "session_key", + "tools": "tools", + "images": "images", + "json_mode": "json_mode", + "streamed": "streamed", + "flex_preference": "flex_preference", + "flex_swapped": "flex_swapped", + "flex_forced": "flex_forced", + "request_id": "request_id", + "exploration": "exploration", + "pinch_original_tokens": "pinch_original_tokens", + "pinch_final_tokens": "pinch_final_tokens", + "profile": "profile", +} +# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a +# one-line reason (callers must keep the comment). Empty today: every column is +# surfaced via decision_row. A new schema column that lands in NEITHER registry +# fails the drift test. +UNSURFACED_COLUMNS: set[str] = set() + +# Every column seeded non-NULL: the single source of truth for the INSERT and +# the round-trip assertions, so the tests prove the column flows, not that +# None flows. +FULL_ROW = { + "id": 1, + "observed_at": "2026-09-05T12:34:56+00:00", + "kind": "chat", + "task_category": "coding_general", + "task_tier": 2, + "required_context_tokens": 123456, + "confidence": 0.91, + "classifier_ms": 1800, + "classification_source": "classifier", + "latency_tolerance": "interactive", + "candidates_considered": 7, + "selected_model": "fixture-model", + "selected_provider": "neuralwatt", + "runner_up_models": '[{"model_id": "runner", "provider": "neuralwatt"}]', + "est_cost_usd": 0.00123, + "est_proficiency": 0.88, + # Schema-legal even for a selected row (no CHECK constraint); non-NULL so the + # round-trip assertion proves the column flows, not that None flows. + "rejected_reason": "fixture rejected reason", + "session_key": "fixture-session-key", + "tools": 0, + "images": 0, + "json_mode": 1, + "streamed": 1, + "flex_preference": "auto", + "flex_swapped": 0, + "flex_forced": 0, + "request_id": "chatcmpl-fixture-42", + "exploration": 1, + "pinch_original_tokens": 120000, + "pinch_final_tokens": 96122, + "profile": "default", +} + + +def _db_from_schema(tmp_path, row_factory=None) -> sqlite3.Connection: + """A throwaway DB from config/schema.sql — never the live router.db. + + ``row_factory`` defaults to None (plain tuples) like the PRAGMA test + needs; the metrics round-trip test passes ``sqlite3.Row`` because + ``metrics.recent_decisions`` builds its rows via ``dict(row)``. + """ + conn = sqlite3.connect(tmp_path / "schema-drift.db") + conn.executescript(SCHEMA_SQL) + conn.row_factory = row_factory + return conn + + +def test_schema_columns_all_have_surfacing_decisions(tmp_path): + """Every route_decisions column is either surfaced or consciously not.""" + conn = _db_from_schema(tmp_path) + schema_cols = { + row[1] for row in conn.execute("PRAGMA table_info(route_decisions)") + } + conn.close() + + registered = set(SCHEMA_TO_MODEL_KEYS) | UNSURFACED_COLUMNS + missing = schema_cols - set(SCHEMA_TO_MODEL_KEYS) - UNSURFACED_COLUMNS + extra = registered - schema_cols + assert not missing, ( + f"route_decisions column(s) added without a surfacing decision: " + f"{sorted(missing)}. Add each to SCHEMA_TO_MODEL_KEYS if the TUI " + f"should surface it (project it in tui_model.decision_row), or to " + f"UNSURFACED_COLUMNS with a reason comment." + ) + assert not extra, ( + f"registry registers column(s) that no longer exist in " + f"route_decisions: {sorted(extra)}." + ) + + +def test_decision_row_projects_every_tracked_column(): + """decision_row surfaces every registered column's value faithfully.""" + projected = tui_model.decision_row(dict(FULL_ROW)) + + expected_keys = set(SCHEMA_TO_MODEL_KEYS.values()) + assert set(projected) == expected_keys, ( + "tui_model.decision_row no longer matches the surfacing registry " + f"(symmetric difference: {sorted(expected_keys ^ set(projected))}).\n" + f" registry values: {sorted(expected_keys)}\n" + f" decision_row keys: {sorted(projected)}" + ) + for col, key in SCHEMA_TO_MODEL_KEYS.items(): + assert projected[key] == FULL_ROW[col], ( + f"route_decisions column {col!r} must surface as decision_row " + f"key {key!r}: got {projected[key]!r}, want {FULL_ROW[col]!r}" + ) + + +def test_metrics_recent_decisions_selects_every_tracked_column(tmp_path): + """The /metrics path carries every registered column end-to-end. + + ``request_id`` reaches the TUI ONLY through this SELECT (SSE publishes + before the backfill), so a narrowed SELECT silently erases forensics + fields a few seconds after a live decision arrives. + """ + conn = _db_from_schema(tmp_path, row_factory=sqlite3.Row) + cols = ",".join(FULL_ROW.keys()) + placeholders = ",".join("?" * len(FULL_ROW)) + conn.execute( + f"INSERT INTO route_decisions ({cols}) VALUES ({placeholders})", + tuple(FULL_ROW.values()), + ) + conn.commit() + + rows = metrics.recent_decisions(conn, limit=5) + conn.close() + assert rows, "the seeded FULL_ROW must come back from recent_decisions" + projected = tui_model.decision_row(rows[0]) + + unreachable = [] + for col, key in SCHEMA_TO_MODEL_KEYS.items(): + if projected.get(key) != FULL_ROW[col]: + unreachable.append( + f"route_decisions column {col!r} does not reach the TUI " + f"through /metrics — is it selected by " + f"metrics.recent_decisions()?" + ) + assert not unreachable, "\n".join(unreachable) -- 2.49.1 From bc3c7f4be88afea3b9a1307439eb2b570ec26b0c Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 5 Sep 2026 02:17:01 -0400 Subject: [PATCH 2/4] feat(tui): decision table time and profile columns, exploration flag The decision feed had no timestamp at all and the eight columns predated five schema additions. Adds `time` first (short HH:MM:SS local -- the date is almost always today and would crowd out `selected`, the column operators actually read) and `profile` after `ctx`, since a decision cannot be read without knowing which profile filtered the candidate set. `flex` becomes `flags` and now composes the flex label with an `E` marker for epsilon-greedy picks. One cell rather than two columns: the terminal is not wide, and an exploratory pick must be visually distinguishable or an operator reads a deliberate random sample as the router's judgement. `profile` renders `or ""` so an absent value is blank rather than the literal "None", which in a dense table reads as a real profile name. Placeholder row widened to arity 10 -- a short placeholder raises inside Textual at render time rather than looking wrong. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U --- src/tui.py | 62 +++++++++++++++++++-- tests/test_tui.py | 133 ++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 192 insertions(+), 3 deletions(-) diff --git a/src/tui.py b/src/tui.py index fca5d1f..c87caf4 100644 --- a/src/tui.py +++ b/src/tui.py @@ -24,6 +24,7 @@ from __future__ import annotations import logging import os +from datetime import datetime from typing import Callable, Optional logger = logging.getLogger(__name__) @@ -95,6 +96,47 @@ def _flex_indicator(r: dict) -> str: return str(pref) +def _short_time(value: object) -> str: + """``HH:MM:SS`` in LOCAL time from an ISO-8601 ``observed_at``. + + The decision feed is dense and almost always same-day, so the date would + crowd out ``selected`` — the column operators actually read. Local rather + than UTC because this is a live feed watched next to a wall clock. + + Returns ``""`` for a missing or unparseable value rather than raising: a + malformed timestamp must not take down the whole table render. + """ + if not value: + return "" + try: + parsed = datetime.fromisoformat(str(value)) + except (TypeError, ValueError): + return "" + if parsed.tzinfo is not None: + parsed = parsed.astimezone() + return parsed.strftime("%H:%M:%S") + + +def _exploration_flag(r: dict) -> str: + """``E`` when epsilon-greedy picked this row instead of the ranking. + + An exploratory pick is a deliberate random sample, NOT the router's + judgement. Without a visible marker an operator reads it as the ranking's + answer and draws the wrong conclusion about model quality. + """ + return "E" if r.get("exploration") else "" + + +def _flags_indicator(r: dict) -> str: + """The combined flags cell: flex label and exploration marker. + + One cell rather than two columns — the terminal is not wide, and the spec + caps the table at one added column plus one flag. + """ + parts = [part for part in (_flex_indicator(r), _exploration_flag(r)) if part] + return " ".join(parts) + + class DashboardApp(App): """A terminal dashboard that reads GET /metrics and renders panels.""" @@ -312,7 +354,16 @@ class DashboardApp(App): model_table.add_columns("model", "calls", "cost $", "kWh", "gCO2eq") decision_table = self.query_one("#decision-table", DataTable) decision_table.add_columns( - "id", "kind", "category", "tier", "ctx", "selected", "est $", "flex" + "time", + "id", + "kind", + "category", + "tier", + "ctx", + "profile", + "selected", + "est $", + "flags", ) decision_table.cursor_type = "row" decision_table.zebra_stripes = True @@ -590,18 +641,23 @@ class DashboardApp(App): dt.clear() for r in self._last_model["recent_decisions"]: dt.add_row( + _short_time(r.get("observed_at")), str(r.get("id")), str(r.get("kind")), str(r.get("category")), str(r.get("tier")), str(r.get("required_context_tokens")), + # `or ""` so an absent profile is blank, never the literal "None". + str(r.get("profile") or ""), str(r.get("selected")), _fmt_usd(r.get("est_cost_usd")), - _flex_indicator(r), + _flags_indicator(r), key=str(r.get("id")), ) if not self._last_model["recent_decisions"]: - dt.add_row("(no decisions)", "", "", "", "", "", "", "", key="_placeholder") + dt.add_row( + "(no decisions)", "", "", "", "", "", "", "", "", "", key="_placeholder" + ) self._restore_cursor(dt, saved_decision_key) def _render_category_table(self) -> None: diff --git a/tests/test_tui.py b/tests/test_tui.py index d8cd496..9a38d45 100644 --- a/tests/test_tui.py +++ b/tests/test_tui.py @@ -13,6 +13,7 @@ from __future__ import annotations import asyncio import copy import time +from datetime import datetime import pytest import requests @@ -1557,3 +1558,135 @@ def test_format_quota_legend_shows_next_reset_date(): assert "2026-07-26" in legend # window start still shown assert "2026-09-06" in legend # billing reset assert "resets" in legend.lower() + + +# -------------------------------------------------------------------------- +# T3 — decision table display: time column, profile column, exploration flag. +# -------------------------------------------------------------------------- + + +def test_decision_table_columns_and_order(): + """The table declares exactly ten columns with `time` first. + + Time leads because it is the natural scan axis for a live feed; `flex` + became `flags` because exploration now shares that cell. + """ + stub = _StubFetcher() + stub.payload = _enriched_payload() + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + dt = a.query_one("#decision-table") + labels = [str(c.label) for c in dt.columns.values()] + assert labels == [ + "time", + "id", + "kind", + "category", + "tier", + "ctx", + "profile", + "selected", + "est $", + "flags", + ] + + _run_app(app, _assert) + + +def test_short_time_renders_hhmmss_and_degrades_quietly(): + """`_short_time` is HH:MM:SS local, and never raises on bad input. + + A malformed timestamp must not take down the whole table render, so the + unparseable cases return "" rather than propagating ValueError. + """ + rendered = tui._short_time("2026-08-23T10:00:00+00:00") + assert len(rendered) == 8 and rendered.count(":") == 2 + # Local conversion, computed the same way the helper does it. + expected = ( + datetime.fromisoformat("2026-08-23T10:00:00+00:00") + .astimezone() + .strftime("%H:%M:%S") + ) + assert rendered == expected + # No date component leaks into the cell. + assert "2026" not in rendered + for bad in (None, "", "not-a-timestamp", 12345): + assert tui._short_time(bad) == "" + + +def test_app_decision_table_profile_column_blank_not_none(): + """`profile` renders its value, and renders BLANK when absent. + + A literal "None" in a dense table reads as a real profile name. + """ + stub = _StubFetcher() + data = _enriched_payload() + data["recent_decisions"][1]["profile"] = None + stub.payload = data + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + dt = a.query_one("#decision-table") + labels = [str(c.label) for c in dt.columns.values()] + profile_at = labels.index("profile") + assert str(dt.get_row_at(0)[profile_at]) == "default" + # Absent profile is an empty cell, never the literal "None". + assert str(dt.get_row_at(1)[profile_at]) == "" + + _run_app(app, _assert) + + +def test_exploration_flag_and_flags_indicator_compose(): + """`E` marks an exploratory pick and composes with the flex label. + + An epsilon-greedy pick is a random sample, not the ranking's judgement; + without the marker an operator reads it as the router's answer. + """ + assert tui._exploration_flag({"exploration": 1}) == "E" + assert tui._exploration_flag({"exploration": 0}) == "" + assert tui._exploration_flag({}) == "" + # Composition: flex label first, then the marker, space separated. + both = tui._flags_indicator( + {"flex_preference": "auto", "flex_swapped": 0, "exploration": 1} + ) + assert both == "auto E" + # Either alone survives; neither yields an empty cell. + assert tui._flags_indicator({"exploration": 1}) == "E" + assert tui._flags_indicator({"flex_preference": "auto"}) == "auto" + assert tui._flags_indicator({}) == "" + + +def test_app_decision_table_exploration_flag_renders(): + """Row 41 is the exploratory pick in the fixture; its flags cell shows E.""" + stub = _StubFetcher() + stub.payload = _enriched_payload() + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + dt = a.query_one("#decision-table") + exploratory = [str(c) for c in dt.get_row_at(1)] + assert any("E" == cell or cell.endswith(" E") for cell in exploratory) + non_exploratory = [str(c) for c in dt.get_row_at(0)] + assert not any(cell == "E" or cell.endswith(" E") for cell in non_exploratory) + + _run_app(app, _assert) + + +def test_placeholder_row_matches_column_arity(): + """The empty-state placeholder must have exactly ten cells. + + A short placeholder row raises inside Textual at render time — an arity + mismatch is a crash, not a cosmetic issue. + """ + stub = _StubFetcher() + data = _enriched_payload() + data["recent_decisions"] = [] + stub.payload = data + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + dt = a.query_one("#decision-table") + assert len(dt.get_row_at(0)) == len(dt.columns) == 10 + + _run_app(app, _assert) -- 2.49.1 From a9bfad00dea1f66e024d3b8852a8ef40fe411db4 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 5 Sep 2026 02:19:40 -0400 Subject: [PATCH 3/4] feat(tui): quota panel leads with balance and runway The panel led with a progress bar against plan_kwh, which is misleading: usage routinely exceeds 100% and nothing fails, because overage bills against a credit balance. PR #30 made balance, burn rate and runway available on /metrics; this surfaces them. Balance and runway now lead, in a bold #quota-lead line, with the plan percentage bar demoted below. Wired outside the plan-configured gate on purpose -- the credit balance is meaningful whether or not a kWh plan is set, and it is the figure that actually stops traffic. burn and runway are legitimately None when the sample segment is too short to extrapolate, which is PR #30's guard against a wild rate right after a top-up. On the live deployment that is the CURRENT state, so it is the first thing an operator sees rather than an edge case. The lead renders "burn n/a" and #quota-note carries runway_note verbatim -- never blank, never a bare 0, never the literal None. Tests assert all three. Formatting is 2dp rather than _fmt_usd's 6: money an operator acts on, not per-request micro-money. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U --- src/tui.py | 80 ++++++++++++++++++++++++++++++++++++++++++++ tests/test_tui.py | 84 +++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 164 insertions(+) diff --git a/src/tui.py b/src/tui.py index c87caf4..8f3a761 100644 --- a/src/tui.py +++ b/src/tui.py @@ -58,6 +58,43 @@ def _fmt_usd(v: Optional[float]) -> str: return f"{v:.6f}" +def _fmt_runway(hours: float) -> str: + """Runway as hours below two days, days above. Money-scale, not micro.""" + if hours < 48: + return f"~{hours:.1f}h" + return f"~{hours / 24:.1f}d" + + +def _format_quota_lead( + balance: Optional[float], + burn: Optional[float], + runway_hours: Optional[float], + low_warning: bool, +) -> str: + """The quota panel's headline: balance, burn rate, projected runway. + + Deliberately 2dp rather than ``_fmt_usd``'s 6 — this is money an operator + acts on, not per-request micro-money. + + ``burn`` and ``runway_hours`` are legitimately ``None`` when the sample + segment is too short to extrapolate (PR #30's guard against a wild rate + right after a top-up). That is a real answer, not missing data, so it + renders "burn n/a" and the caller shows ``runway_note`` alongside. The + literal "None" must never reach the screen. + """ + if balance is None: + return "balance n/a" + parts = [f"balance ${balance:.2f} left"] + parts.append(f"burn ${burn:.2f}/h" if burn is not None else "burn n/a") + if runway_hours is not None: + runway = f"runway {_fmt_runway(float(runway_hours))}" + # Same markup idiom the legend already uses for `resets`. + parts.append( + f"[rgb(200,80,80)]{runway}[/rgb(200,80,80)]" if low_warning else runway + ) + return " · ".join(parts) + + # Compact bottom keys legend. The 11 BINDINGS are collapsed to these short # (key, label) pairs — duplicate quit bindings (q/Q/ctrl+c) show once. Rendered # as a single markup string so Textual's Static wraps rather than truncates. @@ -197,6 +234,12 @@ class DashboardApp(App): #quota-progress { width: 100%; } + #quota-lead { + text-style: bold; + } + #quota-note { + color: $text-muted; + } #quota-legend { color: $text-muted; } @@ -249,6 +292,12 @@ class DashboardApp(App): yield Static("", id="loading-panel") yield Static("Quota burn", classes="panel-title") with Vertical(id="quota-panel"): + # Balance and runway lead; the plan-percentage bar is demoted + # below them. A bar against a routinely-exceeded plan implies + # a ceiling that does not exist — the credit balance is the + # real one (PR #30). + yield Static(markup=True, id="quota-lead") + yield Static(id="quota-note") yield ProgressBar(id="quota-progress") yield Static(markup=True, id="quota-legend") yield Static("Recent decisions (enter = details)", classes="panel-title") @@ -560,6 +609,37 @@ class DashboardApp(App): metered = rows.get("metered_kwh_30d") frac = rows.get("fraction") calls = rows.get("calls") + + # Balance/runway lead. Wired OUTSIDE the plan gate on purpose: the + # credit balance is meaningful whether or not a kWh plan is set, and + # it is the figure that actually stops traffic. + lead = self.query_one("#quota-lead", Static) + note = self.query_one("#quota-note", Static) + if rows: + balance = rows.get("balance_usd") + burn = rows.get("burn_rate_usd_per_hour") + lead.update( + _format_quota_lead( + balance, + burn, + rows.get("projected_hours_remaining"), + bool(rows.get("runway_low_warning")), + ) + ) + # Explain an absent burn rate rather than leaving a bare "n/a". + if balance is not None and burn is None: + note.update( + str(rows.get("runway_note") or "burn estimate unavailable") + ) + note.display = True + else: + note.update("") + note.display = False + else: + lead.update("") + note.update("") + note.display = False + if plan is not None and float(plan) > 0: bar.total = float(plan) bar.progress = float(metered or 0) diff --git a/tests/test_tui.py b/tests/test_tui.py index 9a38d45..9873d7f 100644 --- a/tests/test_tui.py +++ b/tests/test_tui.py @@ -1690,3 +1690,87 @@ def test_placeholder_row_matches_column_arity(): assert len(dt.get_row_at(0)) == len(dt.columns) == 10 _run_app(app, _assert) + + +# -------------------------------------------------------------------------- +# T4 — quota panel leads with balance and runway. +# -------------------------------------------------------------------------- + + +def test_format_quota_lead_states(): + """All four lead states, including the one this deployment is actually in. + + burn/runway are legitimately None when the sample segment is too short + (PR #30). That is deliberate, not missing data, so the None path must + still say something an operator can read. + """ + full = tui._format_quota_lead(8.50, 1.0, 8.5, False) + assert "balance $8.50 left" in full + assert "burn $1.00/h" in full + assert "runway ~8.5h" in full + + none_burn = tui._format_quota_lead(8.50, None, None, False) + assert "burn n/a" in none_burn + assert "None" not in none_burn + + assert tui._format_quota_lead(None, 1.0, 8.5, False) == "balance n/a" + assert "runway ~20.8d" in tui._format_quota_lead(100.0, 1.0, 500.0, False) + + low = tui._format_quota_lead(8.50, 4.0, 1.5, True) + assert "[rgb(200,80,80)]runway ~1.5h" in low + + +def test_quota_panel_burn_unavailable_renders_note_verbatim(): + """When burn is None the note renders VERBATIM — never blank, 0, or None. + + This is the state the live deployment is in right now, so it is the first + thing an operator sees, not an edge case. + """ + stub = _StubFetcher() + data = _fixture() + note = "burn estimate unavailable: segment after last balance increase spans only 9 minutes" + for block in (data["quota"], data["coverage"]["quota"]): + block["balance_usd"] = 7.5 + block["burn_rate_usd_per_hour"] = None + block["projected_hours_remaining"] = None + block["runway_low_warning"] = False + block["runway_note"] = note + stub.payload = data + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + lead_text = str(a.query_one("#quota-lead", Static).content) + note_text = str(a.query_one("#quota-note", Static).content) + assert "balance $7.50 left" in lead_text + assert "burn n/a" in lead_text + assert "None" not in lead_text + # The note is passed through unchanged, not summarised. + assert note_text == note + assert note_text.strip() != "" + assert note_text.strip() != "0" + assert note_text.strip() != "None" + + _run_app(app, _assert) + + +def test_quota_panel_note_hidden_when_burn_is_available(): + """With a real burn rate there is nothing to explain, so no note line.""" + stub = _StubFetcher() + data = _fixture() + for block in (data["quota"], data["coverage"]["quota"]): + block["balance_usd"] = 12.0 + block["burn_rate_usd_per_hour"] = 0.5 + block["projected_hours_remaining"] = 24.0 + block["runway_low_warning"] = False + block["runway_note"] = None + stub.payload = data + app = tui.DashboardApp(fetcher=stub) + + def _assert(a): + lead_text = str(a.query_one("#quota-lead", Static).content) + assert "balance $12.00 left" in lead_text + assert "burn $0.50/h" in lead_text + assert "runway ~24.0h" in lead_text + assert a.query_one("#quota-note", Static).display is False + + _run_app(app, _assert) -- 2.49.1 From 4c8e1abd59f6285afaf5ed7f9557af55e2a59d69 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 5 Sep 2026 02:24:14 -0400 Subject: [PATCH 4/4] test(tui): pin all /metrics warning classes render in the panel The 2026-09-04 vision-ceiling incident was invisible for ~19 hours because a whole warning family was computed correctly and never displayed. The fix there was surfacing, not computing -- so the guard has to be a test that fails when a class exists but nothing renders it. Two directions, both load-bearing. Every REGISTERED class must fire against a fixture that provokes all ten, catching the fixture rotting or a message shape drifting under a matcher. And every EMITTED warning must be registered, catching a new class added to scoring_coverage that nothing displays -- this one fails with the RAW warning text, because the whole point is that nobody knew the class existed. No assertions on warning count: the numbers move as fixture semantics evolve, and what matters is class coverage, not arity. Three fixes to the fixture the plan specified, found by running it: proficiency.last_updated is NOT NULL; a seed_reference row is identified by task_category rather than a `kind` column that does not exist; and scoring_coverage reads `.value` off default_flex_preference, which is an enum in real config, so a bare string is not a faithful stand-in. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U --- tests/test_tui_warnings.py | 291 +++++++++++++++++++++++++++++++++++++ 1 file changed, 291 insertions(+) create mode 100644 tests/test_tui_warnings.py diff --git a/tests/test_tui_warnings.py b/tests/test_tui_warnings.py new file mode 100644 index 0000000..be8416d --- /dev/null +++ b/tests/test_tui_warnings.py @@ -0,0 +1,291 @@ +"""Tripwire: every warning class /metrics can emit must reach the panel. + +The 2026-09-04 vision-ceiling incident was invisible for ~19 hours because a +whole warning family was computed correctly and never displayed. The fix there +was surfacing, not computing — so the guard has to be a test that fails when a +warning class exists but nothing renders it. + +Two directions, both load-bearing: + +* every REGISTERED class actually fires against a fixture that provokes all of + them — catches the fixture rotting or a message shape changing underneath; +* every EMITTED warning is registered — catches a NEW class being added to + ``metrics.scoring_coverage`` without anyone deciding to surface it. This one + fails with the raw warning text, because the whole point is that nobody knew + the class existed. + +Deliberately no assertions on warning COUNT: the numbers shift as fixture +semantics evolve, and item 4 of the spec asks for class coverage, not arity. +""" +from __future__ import annotations + +import asyncio +import copy +import re +import sqlite3 +from datetime import datetime, timedelta, timezone +from pathlib import Path +from types import SimpleNamespace + +import pytest +from textual.widgets import Static + +import metrics +import tui + +ROOT = Path(__file__).resolve().parent.parent +SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text() + + +# The registry. A class here without a matching emitted warning means the +# fixture rotted; an emitted warning without a class here means someone added +# a warning nobody decided to display. +WARN_CLASS_MATCHERS = { + "quota-over-plan": r"metered usage is \d+% of the", + "models-missing-energy": r"\d+/\d+ routable models have no reference-workload", + "models-missing-proficiency": r"\d+/\d+ routable models have no proficiency", + "catalog-stale": r"catalog last polled", + # Fixed-width lookbehind so this cannot swallow the capability warnings, + # which share the "tier N context ceiling" tail. + "demand-ceiling": r"(? None: + """Seed a catalog and traffic that provoke ALL ten warning classes at once. + + Named for what it is: a deliberately maximally-broken deployment. Every + timestamp is relative to now so the rejection windows (1h alert, 24h + baseline) stay valid whenever the suite runs. + """ + now = datetime.now(timezone.utc) + stale = (now - timedelta(days=3)).isoformat() + + rows = [ + # Covered model: has proficiency AND energy, so the "missing" counts + # come out as 2/3 rather than 3/3. + _model_row("cov-t1", 1, 20000), + # Uncovered: drives both missing- classes; its 5000 window is also the + # tier-2 ceiling behind the escalation hazard. + _model_row("gap-t2", 2, 5000), + # Vision-capable but small: the capability sub-ceiling. + _model_row("vis-t1", 1, 5000, vision=1), + # Deprecated: excluded from routable, so tier 3 has zero eligible. + _model_row("dead-t3", 3, 1000, availability="deprecated"), + ] + for row in rows: + conn.execute( + """ + INSERT INTO models ( + model_id, provider, base_model_id, tier, context_window, + effective_context_window, max_output_tokens, + cost_per_1m_prompt, cost_per_1m_completion, + supports_vision, supports_json_mode, + latency_class, reasoning_mode, context_variant, + access_level, availability, last_updated + ) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?) + """, + (*row, stale), + ) + + conn.execute( + "INSERT INTO proficiency " + "(model_id, provider, category, blended_score, last_updated) " + "VALUES ('cov-t1','neuralwatt','coding_general',0.9,?)", + (now.isoformat(),), + ) + # 6.0 kWh against a 6.25 plan = 96% > 80%, firing the quota class. The + # 'seed_reference' task_category is also what marks a model as having + # reference-workload data, so this one row serves both purposes. + conn.execute( + "INSERT INTO energy_observations " + "(model_id, provider, energy_kwh, task_category, observed_at) " + "VALUES ('cov-t1','neuralwatt',6.0,'seed_reference',?)", + ((now - timedelta(days=1)).isoformat(),), + ) + + def _decision(**kw): + cols = { + "kind": "chat", + "task_category": "coding_general", + "task_tier": 1, + "required_context_tokens": 100, + "latency_tolerance": "interactive", + "selected_model": "cov-t1", + "selected_provider": "neuralwatt", + "observed_at": now.isoformat(), + "images": 0, + "json_mode": 0, + "rejected_reason": None, + } + cols.update(kw) + conn.execute( + f"INSERT INTO route_decisions ({','.join(cols)}) " + f"VALUES ({','.join('?' * len(cols))})", + tuple(cols.values()), + ) + + # One huge vision request: demand-ceiling (20000 < 999999), capability + # sub-ceiling (vision 5000 < 999999), and the escalation hazard all at once. + _decision(required_context_tokens=999999, images=1) + + # Two rejections of a pattern with no baseline -> "new rejection pattern:". + for _ in range(2): + _decision( + selected_model=None, + selected_provider=None, + rejected_reason="context >= 99999 tokens; interactive", + ) + + # A familiar group: one in the baseline window, six in the alert window, + # crossing the min-count threshold -> "rejection rate:". + familiar = "tier >= 2; context >= 30000 tokens; interactive" + _decision( + task_tier=2, + selected_model=None, + selected_provider=None, + rejected_reason=familiar, + observed_at=(now - timedelta(hours=2)).isoformat(), + ) + for _ in range(6): + _decision( + task_tier=2, + selected_model=None, + selected_provider=None, + rejected_reason=familiar, + ) + conn.commit() + + +@pytest.fixture() +def chernobyl(tmp_path): + conn = sqlite3.connect(tmp_path / "warn.db") + conn.executescript(SCHEMA_SQL) + conn.row_factory = sqlite3.Row + _seed_chernobyl(conn) + yield conn + conn.close() + + +@pytest.fixture(autouse=True) +def _no_real_network(monkeypatch): + """Even if a fetcher is mis-wired, never reach a real router.""" + + def _guard(base_url): + raise AssertionError(f"real fetch_metrics called with {base_url!r}") + + monkeypatch.setattr(tui, "fetch_metrics", _guard) + + +def _run_app(app: tui.DashboardApp, body) -> None: + async def _go(): + async with app.run_test() as pilot: + await pilot.pause() + body(app) + + asyncio.run(_go()) + + +def test_every_registered_warning_class_fires(chernobyl): + """Each registered class is provoked by the fixture. + + A failure here means either the fixture rotted or a warning's message + shape changed under the matcher — both silently disable the tripwire. + """ + warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] + for name, pattern in WARN_CLASS_MATCHERS.items(): + assert any(re.search(pattern, w) for w in warnings), ( + f"registered warning class {name!r} did not fire.\n" + f"Either the chernobyl fixture no longer provokes it, or the " + f"message shape changed and the matcher needs updating.\n" + f"Pattern: {pattern}\nEmitted:\n " + "\n ".join(warnings) + ) + + +def test_every_emitted_warning_belongs_to_a_registered_class(chernobyl): + """No warning class exists without a decision about surfacing it. + + This is the direction that catches the vision-ceiling failure mode: a new + class added to scoring_coverage that nothing displays. It fails with the + RAW text because the whole point is that nobody knew it existed. + """ + warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] + unmatched = [ + w + for w in warnings + if not any(re.search(p, w) for p in WARN_CLASS_MATCHERS.values()) + ] + assert not unmatched, ( + "a warning class was added to metrics.scoring_coverage without a " + "surfacing decision. Register a matcher in WARN_CLASS_MATCHERS (and a " + "chernobyl seed if it needs one).\nUNMATCHED:\n " + + "\n ".join(unmatched) + ) + + +def test_registered_warnings_render_in_the_panel(chernobyl): + """The panel actually shows them — computing is not surfacing. + + Drives the real app with the chernobyl warnings so the assertion covers + the render path, not just the metrics call. + """ + warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] + + from tests.test_tui import _fixture # noqa: PLC0415 — shared payload shape + + payload = copy.deepcopy(_fixture()) + payload["coverage"]["warnings"] = warnings + + class _Stub: + def __call__(self, base_url): + return payload + + app = tui.DashboardApp(fetcher=_Stub()) + + def _assert(a): + panel = str(a.query_one("#warnings-panel", Static).content) + for name, pattern in WARN_CLASS_MATCHERS.items(): + assert re.search(pattern, panel), ( + f"warning class {name!r} is emitted by /metrics but does not " + f"render in #warnings-panel — computed but not surfaced, which " + f"is exactly the 2026-09-04 failure.\nPanel:\n{panel}" + ) + + _run_app(app, _assert) -- 2.49.1