"""Tests for the Textual monitoring dashboard (tui.py). Offline, no real router / network. The data layer (``build_model`` and ``fetch_metrics``) is importable without a running TUI, so nearly all assertions are on the rendered panel PAYLOADS (plain dicts of display rows), not on pixels. ``App.run_test`` drives the app itself with a stubbed fetcher. This file imports ``tui`` and ``textual`` deliberately — it is the ONE test file that may. No non-tui module imports textual. """ from __future__ import annotations import asyncio import copy import time from datetime import datetime import pytest import requests from textual.widgets import DataTable, ProgressBar, Static import tui import tui_model from tui_model import build_model, fetch_metrics from tui_screens import DecisionDetailScreen, VerdictMixScreen def _fixture() -> dict: """A representative /admin/api/snapshot payload (quota_accounts shape).""" return { "quota": { "period": { "start": "2026-07-26", "next_reset": "2026-08-26", "elapsed_fraction": 0.5, "source": "billing_reset_day", }, "accounts": [ { "provider": "neuralwatt", "shape": "metered_plan", "spend_usd": {"period": 1.25, "attribution_coverage": 1.0}, "plan": {"kwh_per_period": 6.25, "used_kwh": 1.25, "used_fraction": 0.2}, "pool": None, "burn": { "burn_rate_usd_per_hour": 1.0, "projected_hours_remaining": 8.5, }, "credit": { "balance_usd": 8.50, "balance_at": "2026-08-23T09:58:00+00:00", "balance_source": "telemetry", }, "energy": {"kwh_30d": 1.25, "calls_30d": 18}, } ], "spend": { "by_provider_usd": {"neuralwatt": 1.25}, "total_usd": 1.25, "estimated_usd": 0.0, "estimate_ratio": None, }, "alarm": { "kind": "none", "severity": "info", "headline": "$1.25 period spend", }, }, "coverage": { "routable_models": 13, "with_energy_data": 10, "with_proficiency_data": 12, "quota": { "plan_kwh": 6.25, "metered_kwh_30d": 1.25, "metered_fraction_of_plan": 0.2, "metered_calls_30d": 18, "reset_date": "2026-07-26", "note": "router-metered only", "total_balance_usd": 8.50, "by_provider": { "neuralwatt": { "balance_usd": 8.50, "balance_at": "2026-08-23T09:58:00+00:00", "balance_source": "telemetry", "burn_window_hours": 24, "burn_rate_usd_per_hour": 1.0, "projected_hours_remaining": 8.5, "runway_low_warning": False, "runway_note": None, } }, "window_start_30d": "2026-07-24", "metered_kwh_period": 0.5, }, "warnings": [ "3/13 routable models have no reference-workload observations", "1/13 routable models have no proficiency data", ], "flex_default": "auto", }, "recent_decisions": [ { "id": 42, "observed_at": "2026-08-23T10:00:00+00:00", "kind": "chat", "task_category": "coding_general", "task_tier": 2, "selected_model": "deepseek-v4-flash", "selected_provider": "neuralwatt", "est_cost_usd": 0.00016, "flex_preference": "force-flex", "flex_swapped": 1, "flex_forced": 1, }, { "id": 41, "observed_at": "2026-08-23T09:59:00+00:00", "kind": "route", "task_category": "docs_writing", "task_tier": 3, "selected_model": "kimi-k2.7-code", "selected_provider": "neuralwatt", "est_cost_usd": 0.0136, "flex_preference": "auto", "flex_swapped": 0, "flex_forced": 0, }, { "id": 40, "observed_at": "2026-08-23T09:58:00+00:00", "kind": "chat", "task_category": "debugging", "task_tier": 1, "selected_model": None, "selected_provider": None, "est_cost_usd": None, }, ], "per_model": [ { "model_id": "deepseek-v4-flash", "provider": "neuralwatt", "calls": 12, "sum_cost_usd": 0.0012, "sum_energy_kwh": 2.5e-05, "sum_carbon_g_co2eq": 1.2e-04, }, { "model_id": "kimi-k3", "provider": "neuralwatt", "calls": 4, "sum_cost_usd": 0.08, "sum_energy_kwh": 1.0e-03, "sum_carbon_g_co2eq": 3.0e-03, }, { "model_id": "null-energy-model", "provider": "openrouter", "calls": 5, "sum_cost_usd": None, "sum_energy_kwh": None, "sum_carbon_g_co2eq": None, }, ], "verdict_mix": {"ok": 5, "unverifiable": 2, "truncated": 1}, "top_proficiency": [ {"model_id": "deepseek-v4-flash", "provider": "neuralwatt", "blended_score": 1.0, "source": "self_eval_thin", "self_eval_samples": 3} ], "pinch": { "calls_30d": 120, "pruned_calls_30d": 30, "share_pruned": 0.25, "total_tokens_saved": 50000, "median_tokens_saved": 1200, "dollars_saved_usd_30d": 0.0125, }, "generated_at": "2026-08-23T10:01:00+00:00", } def _enriched_payload() -> dict: """A deep copy of _fixture() whose recent_decisions rows carry the six T2 data-layer fields, mirroring what /metrics returns after the drift catch-up. Row 42 is the popup test's target row (exploration off); row 41 is the exploratory pick; row 40 stays a rejection row.""" data = copy.deepcopy(_fixture()) enrichments = [ { "profile": "default", "exploration": 0, "request_id": "chatcmpl-fixture-42", "pinch_original_tokens": 120000, "pinch_final_tokens": 96122, "session_key": "sess-42", "observed_at": "2026-08-23T10:00:00+00:00", }, { "profile": "default", "exploration": 1, "request_id": "chatcmpl-fixture-41", "pinch_original_tokens": 90000, "pinch_final_tokens": 87500, "session_key": "sess-41", "observed_at": "2026-08-23T09:59:00+00:00", }, { "profile": "default", "exploration": 0, "request_id": None, "pinch_original_tokens": None, "pinch_final_tokens": None, "session_key": "sess-40", "observed_at": "2026-08-23T09:58:00+00:00", }, ] for row, extra in zip(data["recent_decisions"], enrichments): row.update(extra) return data # -------------------------------------------------------------------------- # Direct unit tests of the data layer (no TUI running). # -------------------------------------------------------------------------- def test_build_model_quota_panel(): m = build_model(_fixture()) rows = m["quota"] # Plan info is surfaced via per-account rows joined = " ".join(r["label"] + "=" + str(r["value"]) for r in rows) assert "period_start=2026-07-26" in joined assert "next_reset=2026-08-26" in joined assert "spend_total_usd=1.25" in joined assert "neuralwatt plan_kwh_per_period=6.25" in joined assert "neuralwatt used_kwh=1.25" in joined assert "neuralwatt used_fraction=0.2" in joined assert "neuralwatt calls_30d=18" in joined def test_build_model_pinch_panel(): m = build_model(_fixture()) rows = m["pinch"] by_label = {r["label"]: r["value"] for r in rows} assert by_label["share_pruned"] == 0.25 assert by_label["total_tokens_saved"] == 50000 assert by_label["median_tokens_saved"] == 1200 assert by_label["dollars_saved_usd_30d"] == 0.0125 def test_build_model_pinch_panel_empty_when_none(): data = _fixture() data["pinch"] = None m = build_model(data) assert m["pinch"] == [] def test_build_model_quota_panel_includes_period_and_spend(): m = build_model(_fixture()) rows = m["quota"] by_label = {r["label"]: r["value"] for r in rows} assert by_label["period_start"] == "2026-07-26" assert by_label["next_reset"] == "2026-08-26" assert by_label["spend_total_usd"] == 1.25 def test_build_model_quota_panel_includes_provider_rows(): """The per-provider quota shape reaches the TUI data model.""" m = build_model(_fixture()) rows = m["quota"] by_label = {r["label"]: r["value"] for r in rows} assert by_label["neuralwatt shape"] == "metered_plan" assert by_label["neuralwatt plan_kwh_per_period"] == 6.25 assert by_label["neuralwatt used_kwh"] == 1.25 assert by_label["neuralwatt used_fraction"] == 0.2 assert by_label["neuralwatt burn_rate_usd_per_hour"] == 1.0 assert by_label["neuralwatt projected_hours_remaining"] == 8.5 def test_build_model_quota_panel_drops_flat_balance_keys(): """Old flat balance keys must not leak into the TUI row list.""" m = build_model(_fixture()) labels = {r["label"] for r in m["quota"]} assert "balance_usd" not in labels assert "burn_rate_usd_per_hour" not in labels assert "total_balance_usd" not in labels def test_build_model_per_model_lists_seeded_models(): m = build_model(_fixture()) rows = m["per_model"] assert rows[0]["model"] == "deepseek-v4-flash" assert rows[0]["calls"] == 12 assert rows[1]["model"] == "kimi-k3" # every row keeps numeric cost/energy/carbon for display assert rows[0]["cost_usd"] == 0.0012 assert rows[0]["energy_kwh"] == 2.5e-05 assert rows[0]["carbon_g_co2eq"] == 1.2e-04 def test_build_model_per_model_handles_null_energy(): """A per-model row with all-NULL cost/energy/carbon must not crash and must preserve the None values for the rendering layer.""" m = build_model(_fixture()) rows = m["per_model"] null_row = [r for r in rows if r["model"] == "null-energy-model"] assert len(null_row) == 1 nr = null_row[0] assert nr["cost_usd"] is None assert nr["energy_kwh"] is None assert nr["carbon_g_co2eq"] is None def test_build_model_verdict_mix(): m = build_model(_fixture()) mix = m["verdict_mix"] by_verdict = {r["verdict"]: r["count"] for r in mix} assert by_verdict == {"ok": 5, "unverifiable": 2, "truncated": 1} def test_build_model_recent_decisions_top_rows(): m = build_model(_fixture()) rows = m["recent_decisions"] # DESC by id: first row is id 42 assert rows[0]["id"] == 42 assert rows[0]["kind"] == "chat" assert rows[0]["category"] == "coding_general" assert rows[0]["tier"] == 2 assert rows[0]["selected"] == "deepseek-v4-flash" assert rows[1]["kind"] == "route" assert rows[2]["selected"] == "none" # no-candidate row renders "none" def test_build_model_warnings_from_coverage(): m = build_model(_fixture()) warnings = m["warnings"] assert len(warnings) == 2 assert "reference-workload" in warnings[0] assert "proficiency data" in warnings[1] def test_build_model_handles_missing_quota(): """Empty DB / null plan: the quota section is an empty list (renderer hides it).""" data = _fixture() data["quota"] = None data["coverage"]["quota"] = None m = build_model(data) assert m["quota"] == [] def test_fetch_metrics_returns_parsed_dict(monkeypatch): """fetch_metrics hits the right URL and returns parsed JSON.""" captured = {} class _FakeResp: def raise_for_status(self): return None def json(self): return {"quota": None, "ok": True} def _fake_get(url, timeout=None): captured["url"] = url captured["timeout"] = timeout return _FakeResp() monkeypatch.setattr(tui_model.requests, "get", _fake_get) out = fetch_metrics("http://testhost:8081") assert out == {"quota": None, "ok": True} assert captured["url"] == "http://testhost:8081/metrics" def test_fetch_metrics_raises_on_http_error(monkeypatch): class _Err: def raise_for_status(self): raise RuntimeError("500") monkeypatch.setattr(tui_model.requests, "get", lambda *a, **k: _Err()) with pytest.raises(Exception): fetch_metrics("http://x") def test_fetch_metrics_raises_on_network_error(monkeypatch): def _boom(*a, **k): raise ConnectionError("refused") monkeypatch.setattr(tui_model.requests, "get", _boom) with pytest.raises(ConnectionError): fetch_metrics("http://x") # -------------------------------------------------------------------------- # App-level tests via App.run_test with a stubbed fetch_metrics. # -------------------------------------------------------------------------- class _StubFetcher: """Swappable fake for fetch_metrics the App calls.""" def __init__(self): self.payload = None self.error = None self.calls = 0 def __call__(self, base_url): self.calls += 1 if self.error is not None: raise self.error return self.payload @pytest.mark.parametrize("fetcher_arg", ["callable", "subclass"]) def test_app_run_test_populates_quota_and_model_panels(fetcher_arg): stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): stash = getattr(a, "_last_model", None) assert stash is not None, "build_model result was not stashed on the app" # quota plan number surfaced quota_text = " ".join( f"{r['label']}={r['value']}" for r in stash["quota"] ) assert "neuralwatt plan_kwh_per_period=6.25" in quota_text # per-model lists the seeded models models = {r["model"] for r in stash["per_model"]} assert {"deepseek-v4-flash", "kimi-k3"} <= models assert stub.calls == 1 _run_app(app, _assert) def test_app_run_test_recent_and_warnings_panels(): stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): stash = a._last_model assert stash["recent_decisions"][0]["selected"] == "deepseek-v4-flash" assert len(stash["warnings"]) == 2 _run_app(app, _assert) def test_app_run_test_duplicate_model_keys_do_not_crash(): """Duplicate model ids in the /metrics per_model payload must not raise DuplicateKeyError when rendered into #model-table. The source may be a poller/anomaly, so the TUI dedups defensively.""" stub = _StubFetcher() payload = _fixture() payload["per_model"].append(dict(payload["per_model"][0], calls=99)) stub.payload = payload app = tui.DashboardApp(fetcher=stub) def _assert(a): mt = a.query_one("#model-table") assert mt.row_count == 3, ( f"expected 3 deduped rows, got {mt.row_count}" ) assert len(payload["per_model"]) == 4 assert len(a._last_model["per_model"]) == 4 _run_app(app, _assert) def test_app_run_test_null_energy_row_renders_safely(): """A per-model row with NULL cost/energy/carbon must render as 'n/a' for cost and em-dash for energy/carbon without crashing.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): mt = a.query_one("#model-table", DataTable) # Find the null-energy-model row by scanning model names (col 0) null_row_idx = None for idx in range(mt.row_count): cell = mt.get_cell_at((idx, 0)) if "null-energy-model" in str(cell): null_row_idx = idx break assert null_row_idx is not None, "null-energy-model row not found" cost_cell = str(mt.get_cell_at((null_row_idx, 2))) kwh_cell = str(mt.get_cell_at((null_row_idx, 3))) co2_cell = str(mt.get_cell_at((null_row_idx, 4))) assert "n/a" in cost_cell, f"expected n/a for cost, got {cost_cell}" assert "\u2014" in kwh_cell, f"expected em-dash for kWh, got {kwh_cell}" assert "\u2014" in co2_cell, f"expected em-dash for gCO2eq, got {co2_cell}" _run_app(app, _assert) def test_app_run_test_quota_panel_progress_bar_and_legend(): """The quota panel renders a ProgressBar and a legend carrying the plan. ``#quota-panel`` is a container holding a ``ProgressBar`` (``#quota-progress``) and a ``Static`` legend (``#quota-legend``). The bar is filled to the metered kWh against the plan kWh total, and the legend shows the metered / plan / fraction / calls summary. """ stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): container = a.query_one("#quota-panel") assert container is not None bar = a.query_one("#quota-progress", ProgressBar) assert bar.progress == 1.25 assert bar.total == 6.25 legend = a.query_one("#quota-legend", Static) text = str(legend.content) assert "6.25" in text assert "1.25" in text assert "period since 2026-07-26" in text _run_app(app, _assert) def test_format_quota_legend_omits_none_when_unmetered(): """With unmetered (None) values the legend never shows the literal "None". The ``metered``/``frac`` rows can be absent (None) for a plan that has not been metered yet; they must render as ``n/a`` and the returned string must not contain the substring ``"None"``. """ app = tui.DashboardApp(fetcher=_StubFetcher()) legend = app._format_quota_legend( plan=6.25, metered=None, frac=None, flex_default=None, ) assert "None" not in legend assert "n/a" in legend def test_format_quota_legend_happy_path_contains_numbers(): """A fully-populated legend carries the metered / plan values.""" app = tui.DashboardApp(fetcher=_StubFetcher()) legend = app._format_quota_legend( plan=6.25, metered=1.25, frac=0.2, ) assert "6.25" in legend assert "1.25" in legend assert "20%" in legend assert "None" not in legend def test_format_quota_legend_distinguishes_period_from_window(): """The legend shows metered vs plan — period and window labels are now appended in _render, not inside _format_quota_legend.""" app = tui.DashboardApp(fetcher=_StubFetcher()) legend = app._format_quota_legend( plan=6.25, metered=1.25, frac=0.2, ) assert "6.25" in legend assert "1.25" in legend assert "None" not in legend def test_format_quota_legend_shows_flex_default(): """The configured flex default is surfaced in the quota legend readout.""" app = tui.DashboardApp(fetcher=_StubFetcher()) legend = app._format_quota_legend( plan=6.25, metered=1.25, frac=0.2, flex_default="auto", ) assert "flex default auto" in legend def test_format_quota_legend_omits_flex_when_absent(): """No flex default in the payload -> the legend does not claim one.""" app = tui.DashboardApp(fetcher=_StubFetcher()) legend = app._format_quota_legend( plan=6.25, metered=1.25, frac=0.2, ) assert "flex default" not in legend def test_app_run_test_quota_bar_hidden_when_plan_unconfigured(): """No plan configured -> the ProgressBar is hidden (display False). The bar is not removed from the DOM; its ``display`` is toggled so the "quota not configured" legend still shows and the widget state is kept for when a plan is later configured. """ stub = _StubFetcher() data = _fixture() data["quota"]["accounts"] = [] # no metered_plan account → bar hidden data["coverage"] = {"quota": None, "warnings": [], "flex_default": None} stub.payload = data app = tui.DashboardApp(fetcher=stub) def _assert(a): bar = a.query_one("#quota-progress", ProgressBar) assert bar.display is False, "bar should be hidden when plan unconfigured" legend = a.query_one("#quota-legend", Static) assert "quota not configured" in str(legend.content) _run_app(app, _assert) def test_app_run_test_failure_shows_error_and_does_not_crash(): """fetch raising -> error panel visible, run_test completes without raising.""" stub = _StubFetcher() stub.error = ConnectionError("cannot reach router") app = tui.DashboardApp(fetcher=stub, base_url="http://127.0.0.1:8080") def _assert(a): error_widget = a.query_one("#error-panel") assert "cannot reach router" in str(error_widget.content) assert a._last_error is not None _run_app(app, _assert) # must not raise def test_app_run_test_decision_table_focused_on_mount(): """#decision-table is focused by default after the app mounts.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() table = app.query_one("#decision-table") assert table.has_focus, "#decision-table should have focus on mount" asyncio.run(_go()) def _run_app(app: tui.DashboardApp, body) -> None: """Drive the app via Textual App.run_test synchronously. ``body(app)`` runs while the app is mounted, so queries and the data model are live. Assertion failures inside propagate out of ``asyncio.run`` as normal test failures. """ async def _go(): async with app.run_test() as pilot: await pilot.pause() body(app) asyncio.run(_go()) # -------------------------------------------------------------------------- # Auto-refresh, error resilience and keyboard controls. # # These drive the app through App.run_test with a tiny REFRESH_SECONDS and no # real wall-clock sleep. Textual 8.2.8 schedules set_interval timers on the # asyncio event loop, so repeatedly awaiting ``pilot.pause()`` lets due ticks # fire without the test asserting on elapsed time or calling time.sleep(). # Assertions are on the re-rendered panel payload (``app._last_model``), not # on mock-call counts. # -------------------------------------------------------------------------- class _CountingFetcher: """Hands back a metrics payload whose first per-model call count equals the invocation number, so each fetch produces a distinct, inspectable payload with no script to exhaust.""" def __init__(self): self.calls = 0 def __call__(self, base_url): self.calls += 1 return _variant(self.calls) def _variant(calls: int) -> dict: """Return a metrics payload whose per-model call count is ``calls``.""" data = _fixture() data["per_model"][0]["calls"] = calls return data def _first_model_calls(app) -> int: """Read the re-rendered per-model payload's first row call count.""" return app._last_model["per_model"][0]["calls"] def test_auto_refresh_rerenders_updated_payload(): """Interval ticks re-fetch and re-render: the re-rendered panel tracks the fetcher's latest payload. No real sleep: the interval is tiny and every tick fires while the event loop is pumped through ``pilot.pause()``. The assertion reads the actual re-rendered payload back, not a mock-call count. """ fetcher = _CountingFetcher() app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=0.05) async def _go(): async with app.run_test() as pilot: await pilot.pause() # >1 fetch means an auto-tick fired beyond the on_mount refresh. for _ in range(30): await pilot.pause() assert fetcher.calls > 1 # The displayed panel reflects the fetcher's newest payload. assert _first_model_calls(app) == fetcher.calls assert app._refreshing is False asyncio.run(_go()) def test_error_resilience_keeps_last_good_data_and_recovers(): """A transient fetch failure keeps the app running and the last good data displayed; a later successful refresh re-renders the new payload. A large interval means no stray auto-ticks, so the 2nd fetch is exactly the failing one, driven deterministically through the same ``_on_interval`` callback the timer invokes — no sleep, no timing race. """ calls = {"n": 0} def _scripted(base_url): calls["n"] += 1 if calls["n"] == 2: raise ConnectionError("transient blip") return _variant(calls["n"]) app = tui.DashboardApp(fetcher=_scripted, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() assert _first_model_calls(app) == 1 # initial good fetch displayed # advance one tick (the failing 2nd fetch) app._on_interval() await pilot.pause() assert app._last_error is not None, "transient failure was not seen" assert not app._exit, "app must not exit on a transient failure" err_widget = app.query_one("#error-panel") assert "cannot reach router" in str(err_widget.content) assert "visible" in err_widget.classes # last good data is still the displayed payload assert _first_model_calls(app) == 1 # recover on the next refresh (3rd fetch, now successful) app._on_interval() await pilot.pause() assert _first_model_calls(app) == 3 assert app._last_error is None asyncio.run(_go()) def test_force_refresh_binding_reloads_on_r(): """Pressing ``r`` immediately re-fetches and re-renders a new payload.""" fetcher = _CountingFetcher() # A large interval ensures only the forced refresh advances the payload. app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() baseline = fetcher.calls await pilot.press("r") await pilot.pause() assert fetcher.calls == baseline + 1 assert _first_model_calls(app) == fetcher.calls asyncio.run(_go()) @pytest.mark.parametrize("key", ["q", "Q", "ctrl+c"]) def test_quit_bindings_exit_app(key): """q, Q and Ctrl+C all quit the running app.""" fetcher = _CountingFetcher() app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() assert not app._exit await pilot.press(key) assert app._exit asyncio.run(_go()) @pytest.mark.parametrize( "key,panel", [ ("1", "model-table"), ("2", "decision-table"), ("3", "category-table"), ("4", "quota-panel"), ("5", "warnings-panel"), ], ) def test_number_bindings_focus_panel(key, panel): """Number keys 1-5 focus the corresponding panel.""" fetcher = _CountingFetcher() app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() await pilot.press(key) await pilot.pause() widget = app.query_one(f"#{panel}") assert widget.has_focus, f"{panel} should have focus after {key!r}" asyncio.run(_go()) # -------------------------------------------------------------------------- # Cursor persistence across refreshes (decision table). # -------------------------------------------------------------------------- def _decisions_payload(*ids: int) -> dict: """A /metrics payload whose recent_decisions carry the given ids (newest first), all otherwise identical to the fixture's first row shape.""" base = _fixture()["recent_decisions"][0] return {"recent_decisions": [dict(base, id=i) for i in ids]} def test_cursor_persistence_across_refresh(): """Highlighting a row survives a refresh that shifts it: after a new decision is prepended, the cursor follows the same logical decision (by stable row key ``str(id)``) instead of resetting to row 0.""" payloads = [_decisions_payload(42, 41, 40), _decisions_payload(43, 42, 41, 40)] class _Scripted: def __init__(self): self.calls = 0 def __call__(self, base_url): p = payloads[self.calls] self.calls += 1 return p fetcher = _Scripted() app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() dt = app.query_one("#decision-table") # Row 1 is id 41; highlight it. dt.focus() dt.move_cursor(row=1) await pilot.pause() assert dt.cursor_coordinate.row == 1 # Logically the highlighted decision id. highlighted_id = app._last_model["recent_decisions"][1]["id"] assert highlighted_id == 41 # Refresh to a payload with a new decision prepended. app.action_refresh() await pilot.pause() assert fetcher.calls == 2 # id 41 now sits at index 2; the cursor must have followed it. assert app._last_model["recent_decisions"][2]["id"] == 41 assert dt.cursor_coordinate.row == 2, ( f"cursor reset to {dt.cursor_coordinate.row}; expected 2 " "(same logical decision across the refresh)" ) asyncio.run(_go()) def test_cursor_persistence_clamps_when_selected_row_removed(): """When the highlighted decision disappears from the payload on refresh, the cursor rests at row 0 (the default ``clear()`` already set) instead of clamping to a stale anchor.""" payloads = [_decisions_payload(42, 41, 40), _decisions_payload(40, 39)] class _Scripted: def __init__(self): self.calls = 0 def __call__(self, base_url): p = payloads[self.calls] self.calls += 1 return p fetcher = _Scripted() app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() dt = app.query_one("#decision-table") dt.focus() dt.move_cursor(row=1) # id 41 await pilot.pause() assert dt.cursor_coordinate.row == 1 app.action_refresh() # id 41 is dropped in the new payload await pilot.pause() # The cursor rests at the cleared default row 0. assert dt.cursor_coordinate.row == 0, ( f"cursor at {dt.cursor_coordinate.row}; expected 0 " "(dropped row should rest at the cleared default)" ) # The table reflects the two-row payload. assert app._last_model["recent_decisions"][0]["id"] == 40 asyncio.run(_go()) @pytest.fixture(autouse=True) def _no_real_network(monkeypatch): """Safety net: even if the fetcher is mis-wired, never hit a real router.""" def _guard(base_url): raise AssertionError(f"real fetch_metrics called with {base_url!r}") monkeypatch.setattr(tui, "fetch_metrics", _guard) # Also guard the SSE consumer's requests so a stray DecisionStream never # reaches the network even if a test forgets to disable live_events. import tui_sse def _sse_guard(*args, **kwargs): raise AssertionError( f"real requests.get called from tui_sse with {args!r} {kwargs!r}" ) monkeypatch.setattr(tui_sse.requests, "get", _sse_guard) # -------------------------------------------------------------------------- # Category breakdown and enriched decision fields (pure data-layer tests). # -------------------------------------------------------------------------- def test_build_category_breakdown_majority_and_share(): """One category, two different winners: majority is the most common and the share is its fraction of the count.""" decisions = [ {"id": 3, "category": "coding_general", "tier": 2, "selected": "a"}, {"id": 2, "category": "coding_general", "tier": 2, "selected": "a"}, {"id": 1, "category": "coding_general", "tier": 2, "selected": "b"}, ] rows = tui_model.build_category_breakdown(decisions) assert len(rows) == 1 row = rows[0] assert row["category"] == "coding_general" assert row["tier"] == 2 assert row["count"] == 3 assert row["majority"] == "a" assert row["share"] == round(2 / 3, 2) def test_build_category_breakdown_separates_tiers(): """Same category, different tiers are separate buckets.""" decisions = [ {"id": 2, "category": "coding_general", "tier": 1, "selected": "tiny"}, {"id": 1, "category": "coding_general", "tier": 3, "selected": "big"}, ] rows = tui_model.build_category_breakdown(decisions) assert len(rows) == 2 tiers = {r["tier"] for r in rows} assert tiers == {1, 3} def test_build_category_breakdown_handles_empty_and_none_selected(): """No decisions: empty list. Decisions with no selected model get majority 'none' and share 1.0 (they all count toward the bucket).""" assert tui_model.build_category_breakdown([]) == [] rows = tui_model.build_category_breakdown( [{"id": 1, "category": "x", "tier": 1, "selected": None}] ) assert rows[0]["majority"] == "none" def test_build_model_recent_decisions_carry_enriched_fields(): """The enriched /metrics row fields must reach the TUI data model so the detail popup can render the full decision in full.""" data = _fixture() # Ensure the fixture's first row has the fields the new model surfaces. data["recent_decisions"][0].update( { "required_context_tokens": 50000, "confidence": 0.92, "classifier_ms": 1800, "classification_source": "classifier", "latency_tolerance": "interactive", "candidates_considered": 8, "runner_up_models": '[{"model_id":"kimi-k3","provider":"neuralwatt"}]', "est_proficiency": 0.9, "rejected_reason": None, "tools": 0, "images": 0, "json_mode": 0, "streamed": 1, } ) m = build_model(data) row = m["recent_decisions"][0] assert row["required_context_tokens"] == 50000 assert row["confidence"] == 0.92 assert row["runner_up_models"].startswith("[{") assert row["streamed"] == 1 # The breakdown is always present (even an empty list proves the key). assert "category_breakdown" in m def test_decision_row_projects_flex_fields(): """decision_row must project the three flex telemetry fields so they reach both the /metrics path and the live SSE path (shared by build_model).""" row = tui_model.decision_row( { "flex_preference": "prefer-flex", "flex_swapped": 1, "flex_forced": 0, } ) assert row["flex_preference"] == "prefer-flex" assert row["flex_swapped"] == 1 assert row["flex_forced"] == 0 def test_decision_row_missing_flex_fields_default_to_none(): """Rows without flex columns (older payloads) yield None, not a crash.""" row = tui_model.decision_row({"id": 1}) assert row["flex_preference"] is None assert row["flex_swapped"] is None assert row["flex_forced"] is None def test_build_model_threads_flex_default(): """build_model carries the configured flex default through from coverage.""" m = build_model(_fixture()) assert m["flex_default"] == "auto" def test_flex_indicator_labels(): """The compact flex cell: plain preference by default, markers on a swap.""" assert tui._flex_indicator({"flex_preference": "auto", "flex_swapped": 0, "flex_forced": 0}) == "auto" assert tui._flex_indicator({"flex_preference": "prefer-flex", "flex_swapped": 1, "flex_forced": 0}) == "prefer-flex!" assert tui._flex_indicator({"flex_preference": "force-flex", "flex_swapped": 1, "flex_forced": 1}) == "force!" assert tui._flex_indicator({"id": 1}) == "" def test_app_decision_table_renders_flex_column(): """The decision table renders a flex indicator per row and the configured flex default appears in the UI.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): dt = a.query_one("#decision-table") first_row_text = " ".join(str(c) for c in dt.get_row_at(0)) assert "force!" in first_row_text second_row_text = " ".join(str(c) for c in dt.get_row_at(1)) assert "auto" in second_row_text legend = a.query_one("#quota-legend", Static) assert "flex default auto" in str(legend.content) _run_app(app, _assert) def test_keys_legend_built_from_short_pairs_dedups_quit(): """The bottom keys legend is built from compact short pairs and collapses the three quit bindings (q/Q/ctrl+c) to a single 'quit'.""" legend = tui._keys_legend() assert isinstance(legend, str) # every short key is present for key in ("q", "r", "e", "ctrl+v", "1", "2", "3", "4", "5"): assert f"[bold]{key}[/bold]" in legend # short labels are present for label in ("quit", "refresh", "detail", "mix", "model", "decisions", "breakdown", "quota", "warnings"): assert label in legend # exactly one quit (q/Q/ctrl+c collapsed) assert legend.count("quit") == 1 def test_app_run_test_renders_keys_legend(): """The app renders a #keys-legend Static (not a Footer) carrying the full compact legend string regardless of terminal width, so no key is truncated away.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): legend = a.query_one("#keys-legend", Static) text = str(legend.content) # full legend is present (wrapped, not truncated) on a narrow window for key in ("q", "r", "e", "2", "5"): assert f"[bold]{key}[/bold]" in text assert text.count("quit") == 1 _run_app(app, _assert) def test_app_run_test_keys_legend_not_truncated_at_narrow_width(): """At a narrow terminal width the legend's content is intact (wraps rather than drops keys), unlike Textual's Footer which would ellipsize.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) async def _go(): async with app.run_test(size=(40, 30)) as pilot: await pilot.pause() legend = app.query_one("#keys-legend", Static) text = str(legend.content) # every short key survives narrow rendering — nothing truncated for key in ("q", "r", "e", "ctrl+v", "1", "2", "3", "4", "5"): assert f"[bold]{key}[/bold]" in text asyncio.run(_go()) # -------------------------------------------------------------------------- # Detail popup and live SSE decision handling (App-level tests). # -------------------------------------------------------------------------- def test_show_decision_detail_pushes_modal_with_full_row(): """Pressing ``e`` on the decisions table opens a modal whose body contains the full JSON of the selected row — not just the table columns.""" stub = _StubFetcher() stub.payload = _fixture() stub.payload["recent_decisions"][0].update( { "required_context_tokens": 50000, "confidence": 0.92, "runner_up_models": '[{"model_id":"kimi-k3"}]', "rejected_reason": None, } ) app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() # Focus the decisions table and move to the first row, then open # the detail popup via the dedicated binding. app.query_one("#decision-table").focus() await pilot.pause() await pilot.press("e") await pilot.pause() # A modal screen is now active and carries the selected decision. from textual.screen import ModalScreen assert isinstance(app.screen, ModalScreen) decision = app.screen.decision assert decision["id"] == 42 assert decision["required_context_tokens"] == 50000 assert "kimi-k3" in decision["runner_up_models"] asyncio.run(_go()) def test_detail_popup_carries_forensics_fields(): """The e-key popup carries the six T2 forensics fields — profile, exploration, pinch tokens, request_id, session_key — both on the live decision dict and in the rendered JSON for a metrics-shaped full row.""" stub = _StubFetcher() stub.payload = _enriched_payload() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() app.query_one("#decision-table").focus() await pilot.pause() await pilot.press("e") await pilot.pause() from textual.screen import ModalScreen assert isinstance(app.screen, ModalScreen) decision = app.screen.decision assert decision["request_id"] == "chatcmpl-fixture-42" assert decision["session_key"] == "sess-42" assert decision["pinch_original_tokens"] == 120000 assert decision["pinch_final_tokens"] == 96122 assert decision["profile"] == "default" assert decision["exploration"] == 0 asyncio.run(_go()) metrics_row = { "id": 43, "observed_at": "2026-09-05T12:34:56+00:00", "kind": "chat", "task_category": "coding_general", "task_tier": 2, "required_context_tokens": 123456, "confidence": 0.91, "classifier_ms": 1800, "classification_source": "classifier", "latency_tolerance": "interactive", "candidates_considered": 7, "selected_model": "fixture-model", "selected_provider": "neuralwatt", "runner_up_models": None, "est_cost_usd": 0.001, "est_proficiency": 0.9, "rejected_reason": None, "session_key": "sess-43", "tools": 0, "images": 0, "json_mode": 0, "streamed": 1, "flex_preference": "auto", "flex_swapped": 0, "flex_forced": 0, "request_id": "chatcmpl-fixture-43", "exploration": 1, "pinch_original_tokens": 120000, "pinch_final_tokens": 96122, "profile": "default", } text = DecisionDetailScreen(tui_model.decision_row(metrics_row))._render_text() for key in ( "profile", "exploration", "pinch_original_tokens", "pinch_final_tokens", "request_id", "session_key", ): assert key in text def test_app_run_test_ctrl_v_opens_verdict_popup(): """Pressing ``ctrl+v`` pushes a VerdictMixScreen modal showing the verdict mix rows from the current model, and ``escape`` dismisses it back to the main screen.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() await pilot.press("ctrl+v") await pilot.pause() from textual.screen import ModalScreen assert isinstance(app.screen, ModalScreen) assert isinstance(app.screen, VerdictMixScreen) # The popup carries the fixture's verdict mix. by_verdict = { row["verdict"]: row["count"] for row in app.screen.verdict_mix } assert by_verdict == {"ok": 5, "unverifiable": 2, "truncated": 1} # The DataTable renders a row per verdict. table = app.screen.query_one("#verdict-table") assert table.row_count == 3 # ``escape`` dismisses back to the main dashboard screen. await pilot.press("escape") await pilot.pause() assert not isinstance(app.screen, VerdictMixScreen) asyncio.run(_go()) def test_verdict_mix_screen_updates_live_after_refresh(): """An open VerdictMixScreen modal tracks the newest verdict mix after a refresh, rather than staying a static snapshot from when it was opened.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() await pilot.press("ctrl+v") await pilot.pause() by_verdict = { row["verdict"]: row["count"] for row in app.screen.verdict_mix } assert by_verdict == {"ok": 5, "unverifiable": 2, "truncated": 1} # The next refresh delivers a new (smaller) verdict mix. stub.payload["verdict_mix"] = {"ok": 100, "failed": 5} app.action_refresh() await pilot.pause() by_verdict = { row["verdict"]: row["count"] for row in app.screen.verdict_mix } assert by_verdict == {"ok": 100, "failed": 5} table = app.screen.query_one("#verdict-table") assert table.row_count == 2 asyncio.run(_go()) def test_live_decision_inserts_row_at_front_and_rerenders(): """A decision delivered via the SSE callback is prepended to the model and re-renders the decisions and category tables without a full re-fetch.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() before = len(app._last_model["recent_decisions"]) # Simulate the SSE consumer handing in a brand-new decision. # Same (category, tier) as the fixture's first row so the bucket # count for coding_general/tier-2 rises to 2. app._handle_live_decision( { "id": 999, "kind": "chat", "task_category": "coding_general", "task_tier": 2, "selected_model": "deepseek-v4-flash", "est_cost_usd": 0.0002, } ) await pilot.pause() after = app._last_model["recent_decisions"] assert len(after) == before + 1 assert after[0]["id"] == 999 # prepended, newest-first # The table was re-rendered: the first row shows the new id. dt = app.query_one("#decision-table") first_row_text = " ".join(str(c) for c in dt.get_row_at(0)) assert "999" in first_row_text asyncio.run(_go()) def test_live_decision_carries_new_fields(): """A live SSE decision keeps the six T2 fields through the shared decision_row projection, so a prompt `e` on a fresh decision shows the same forensics as a /metrics-sourced row.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() app._handle_live_decision( { "id": 1001, "kind": "chat", "task_category": "coding_general", "task_tier": 2, "selected_model": "deepseek-v4-flash", "est_cost_usd": 0.0002, "profile": "default", "exploration": 1, "request_id": "chatcmpl-live-1001", "pinch_original_tokens": 120000, "pinch_final_tokens": 96122, "session_key": "sess-live-1001", } ) await pilot.pause() row = app._last_model["recent_decisions"][0] assert row["id"] == 1001 assert row["profile"] == "default" assert row["exploration"] == 1 assert row["request_id"] == "chatcmpl-live-1001" assert row["pinch_original_tokens"] == 120000 assert row["pinch_final_tokens"] == 96122 assert row["session_key"] == "sess-live-1001" asyncio.run(_go()) def test_live_decision_caps_recent_decisions_at_fifty(): """The live feed never grows the in-memory list past the /metrics cap, so the dashboard's view stays consistent with a /metrics refresh.""" stub = _StubFetcher() # Start with exactly 50 rows so one live addition must evict the oldest. base = _fixture()["recent_decisions"][0] stub.payload = {"recent_decisions": [dict(base, id=i) for i in range(50, 0, -1)]} app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() assert len(app._last_model["recent_decisions"]) == 50 app._handle_live_decision( {"id": 1, "kind": "chat", "task_category": "x", "task_tier": 1} ) await pilot.pause() assert len(app._last_model["recent_decisions"]) == 50 asyncio.run(_go()) def test_live_decision_dedup_skips_duplicate_id(): stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() before = len(app._last_model["recent_decisions"]) app._handle_live_decision( { "id": 42, "kind": "chat", "task_category": "coding_general", "task_tier": 2, "selected_model": "deepseek-v4-flash", } ) await pilot.pause() after = app._last_model["recent_decisions"] assert len(after) == before, "duplicate id did not get skipped" # first row is id=41 (42 was skipped as duplicate; 42 is now at pos 0) assert after[0]["id"] == 42 and after[1]["id"] == 41, ( "first row is id 42 after dedup" ) app._handle_live_decision( { "id": 888, "kind": "route", "task_category": "coding_general", "task_tier": 2, "selected_model": "kimi-k3", } ) await pilot.pause() assert len(app._last_model["recent_decisions"]) == before + 1 assert app._last_model["recent_decisions"][0]["id"] == 888 asyncio.run(_go()) def test_live_decision_new_bucket_rebuilds_breakdown(): stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() buckets_before = set( (r["category"], r["tier"]) for r in app._last_model["category_breakdown"] ) app._handle_live_decision( { "id": 1000, "kind": "route", "task_category": "summarization", "task_tier": 1, "selected_model": "qwen3.6-35b", } ) await pilot.pause() buckets_after = set( (r["category"], r["tier"]) for r in app._last_model["category_breakdown"] ) assert ("summarization", 1) in buckets_after - buckets_before row = next( r for r in app._last_model["category_breakdown"] if r["category"] == "summarization" and r["tier"] == 1 ) assert row["count"] == 1 assert row["majority"] == "qwen3.6-35b" asyncio.run(_go()) # -------------------------------------------------------------------------- # DecisionStream: callback exceptions and stop behavior (tui_sse.py). # -------------------------------------------------------------------------- class _MockResp: def __enter__(self): return self def __exit__(self, *a): pass def raise_for_status(self): pass def iter_lines(self, decode_unicode=False): yield "data: {\"id\": 1}" raise requests.exceptions.ConnectionError("broken pipe") def test_callback_raises_doesnt_break_reconnect(monkeypatch): """A callback that raises RuntimeError is caught; the stream still survives and processes subsequent decisions after a reconnection.""" import tui_sse monkeypatch.setattr(tui_sse.requests, "get", lambda *a, **kw: _MockResp()) call_count = {"n": 0} def failing_callback(decision): call_count["n"] += 1 if call_count["n"] == 1: raise RuntimeError("app loop gone") # second call succeeds — proves reconnect worked s = tui_sse.DecisionStream( "http://127.0.0.1", failing_callback, reconnect_seconds=0.1, ) s.start() time.sleep(0.6) s.stop() s.join(timeout=2) assert not s.is_alive() assert call_count["n"] >= 2, ( f"Expected reconnection after callback failure, got {call_count['n']} call(s)" ) def test_live_decision_existing_bucket_defers_count_update(): """A live decision in an existing bucket does NOT trigger an immediate count rebuild (fixes finding #9 risk), but the count is updated on the next authoritative /metrics poll. """ stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() initial_count = next( r["count"] for r in app._last_model["category_breakdown"] if r["category"] == "coding_general" and r["tier"] == 2 ) assert initial_count == 1 app._handle_live_decision( {"id": 2000, "kind": "chat", "task_category": "coding_general", "task_tier": 2, "selected_model": "deepseek-v4-flash"} ) await pilot.pause() count_after_live = next( r["count"] for r in app._last_model["category_breakdown"] if r["category"] == "coding_general" and r["tier"] == 2 ) assert count_after_live == initial_count, "count grew immediately" stub.payload["recent_decisions"].append( {"id": 2000, "kind": "chat", "task_category": "coding_general", "task_tier": 2, "selected_model": "deepseek-v4-flash"} ) app._on_interval() await pilot.pause() final_count = next( r["count"] for r in app._last_model["category_breakdown"] if r["category"] == "coding_general" and r["tier"] == 2 ) assert final_count == 2 asyncio.run(_go()) def test_stopped_stream_exits_without_reconnect_sleep(monkeypatch): """After stop(), the thread should NOT wait for reconnect_seconds before exiting — the _stopped guard is checked before the sleep.""" import tui_sse class _MockResp: def __enter__(self): return self def __exit__(self, *a): pass def raise_for_status(self): pass def iter_lines(self, decode_unicode=False): raise requests.exceptions.ConnectionError("closed") monkeypatch.setattr(tui_sse.requests, "get", lambda *a, **kw: _MockResp()) stop_at = time.monotonic() s = tui_sse.DecisionStream( "http://127.0.0.1", lambda x: None, reconnect_seconds=5.0, ) s.start() # Let the first request attempt begin. time.sleep(0.2) # Record when stop is called. s.stop() stopped_at = time.monotonic() # Thread should exit BEFORE the 5s reconnect backoff (use 3s as margin). s.join(timeout=3) assert not s.is_alive(), "Thread should exit promptly after stop()" assert stopped_at - stop_at < 3.0, "Thread slept through reconnect_seconds" def test_format_quota_legend_shows_next_reset_date(): """When next_reset_date is available, _render appends the period line. _format_quota_legend itself no longer takes next_reset_date; it is appended in _render.""" app = tui.DashboardApp(fetcher=_StubFetcher(), refresh_seconds=60) legend = app._format_quota_legend( plan=6.25, metered=1.25, frac=0.2, ) assert "6.25" in legend assert "1.25" in legend # -------------------------------------------------------------------------- # T3 — decision table display: time column, profile column, exploration flag. # -------------------------------------------------------------------------- def test_decision_table_columns_and_order(): """The table declares exactly ten columns with `time` first. Time leads because it is the natural scan axis for a live feed; `flex` became `flags` because exploration now shares that cell. """ stub = _StubFetcher() stub.payload = _enriched_payload() app = tui.DashboardApp(fetcher=stub) def _assert(a): dt = a.query_one("#decision-table") labels = [str(c.label) for c in dt.columns.values()] assert labels == [ "time", "id", "kind", "category", "tier", "ctx", "profile", "selected", "est $", "flags", ] _run_app(app, _assert) def test_short_time_renders_hhmmss_and_degrades_quietly(): """`_short_time` is HH:MM:SS local, and never raises on bad input. A malformed timestamp must not take down the whole table render, so the unparseable cases return "" rather than propagating ValueError. """ rendered = tui._short_time("2026-08-23T10:00:00+00:00") assert len(rendered) == 8 and rendered.count(":") == 2 # Local conversion, computed the same way the helper does it. expected = ( datetime.fromisoformat("2026-08-23T10:00:00+00:00") .astimezone() .strftime("%H:%M:%S") ) assert rendered == expected # No date component leaks into the cell. assert "2026" not in rendered for bad in (None, "", "not-a-timestamp", 12345): assert tui._short_time(bad) == "" def test_app_decision_table_profile_column_blank_not_none(): """`profile` renders its value, and renders BLANK when absent. A literal "None" in a dense table reads as a real profile name. """ stub = _StubFetcher() data = _enriched_payload() data["recent_decisions"][1]["profile"] = None stub.payload = data app = tui.DashboardApp(fetcher=stub) def _assert(a): dt = a.query_one("#decision-table") labels = [str(c.label) for c in dt.columns.values()] profile_at = labels.index("profile") assert str(dt.get_row_at(0)[profile_at]) == "default" # Absent profile is an empty cell, never the literal "None". assert str(dt.get_row_at(1)[profile_at]) == "" _run_app(app, _assert) def test_exploration_flag_and_flags_indicator_compose(): """`E` marks an exploratory pick and composes with the flex label. An epsilon-greedy pick is a random sample, not the ranking's judgement; without the marker an operator reads it as the router's answer. """ assert tui._exploration_flag({"exploration": 1}) == "E" assert tui._exploration_flag({"exploration": 0}) == "" assert tui._exploration_flag({}) == "" # Composition: flex label first, then the marker, space separated. both = tui._flags_indicator( {"flex_preference": "auto", "flex_swapped": 0, "exploration": 1} ) assert both == "auto E" # Either alone survives; neither yields an empty cell. assert tui._flags_indicator({"exploration": 1}) == "E" assert tui._flags_indicator({"flex_preference": "auto"}) == "auto" assert tui._flags_indicator({}) == "" def test_app_decision_table_exploration_flag_renders(): """Row 41 is the exploratory pick in the fixture; its flags cell shows E.""" stub = _StubFetcher() stub.payload = _enriched_payload() app = tui.DashboardApp(fetcher=stub) def _assert(a): dt = a.query_one("#decision-table") exploratory = [str(c) for c in dt.get_row_at(1)] assert any("E" == cell or cell.endswith(" E") for cell in exploratory) non_exploratory = [str(c) for c in dt.get_row_at(0)] assert not any(cell == "E" or cell.endswith(" E") for cell in non_exploratory) _run_app(app, _assert) def test_placeholder_row_matches_column_arity(): """The empty-state placeholder must have exactly ten cells. A short placeholder row raises inside Textual at render time — an arity mismatch is a crash, not a cosmetic issue. """ stub = _StubFetcher() data = _enriched_payload() data["recent_decisions"] = [] stub.payload = data app = tui.DashboardApp(fetcher=stub) def _assert(a): dt = a.query_one("#decision-table") assert len(dt.get_row_at(0)) == len(dt.columns) == 10 _run_app(app, _assert) # -------------------------------------------------------------------------- # T4 — quota panel: kWh-only lead, labeled per-provider account rows. # -------------------------------------------------------------------------- def test_format_small_balance_renders_tiny_and_normal_values(): assert tui._format_small_balance(None) == "n/a" assert tui._format_small_balance(0.0) == "$0.00" assert tui._format_small_balance(0.0001) == "<$0.01" assert tui._format_small_balance(-0.004) == "~$0.00 (slight overage)" assert tui._format_small_balance(8.50) == "$8.50" assert tui._format_small_balance(-1.0) == "$-1.00" def test_format_provider_quota_account_labels_billing_shape(): """_format_provider_quota_account renders prepaid_credit and metered_plan.""" prepaid = { "provider": "openrouter", "shape": "prepaid_credit", "spend_usd": {"period": 2.0, "attribution_coverage": 1.0}, "plan": None, "pool": {"total_credits_usd": 5.0, "total_usage_usd": 0.001, "balance_usd": 4.999, "age_seconds": 60}, "burn": {"burn_rate_usd_per_hour": 0.5, "projected_hours_remaining": 10.0}, "credit": {"balance_usd": 4.999, "balance_at": "2026-08-23T10:00:00+00:00", "balance_source": "polled"}, "energy": {"kwh_30d": 0.0, "calls_30d": 0}, } row = tui._format_provider_quota_account(prepaid) assert row.startswith("openrouter") assert "burn $0.50/h" in row assert "runway" in row metered = { "provider": "neuralwatt", "shape": "metered_plan", "spend_usd": {"period": 1.25, "attribution_coverage": 1.0}, "plan": {"kwh_per_period": 6.25, "used_kwh": 1.25, "used_fraction": 0.2}, "pool": None, "burn": None, "credit": {"balance_usd": -0.004, "balance_at": "2026-08-23T09:58:00+00:00", "balance_source": "telemetry"}, "energy": {"kwh_30d": 1.25, "calls_30d": 18}, } row = tui._format_provider_quota_account(metered) assert row.startswith("neuralwatt") assert "kWh plan" in row unmetered = { "provider": "provider-a", "shape": "unmetered", "spend_usd": {"period": 0.0, "attribution_coverage": 1.0}, "plan": None, "pool": None, "burn": None, "credit": None, "energy": None, } row = tui._format_provider_quota_account(unmetered) assert "provider-a" in row assert "unmetered" in row def test_provider_quota_rows_returns_one_row_per_account_sorted(): quota_raw = { "accounts": [ { "provider": "neuralwatt", "shape": "metered_plan", "spend_usd": {"period": 0.001, "attribution_coverage": 0.5}, "plan": {"kwh_per_period": 6.25, "used_kwh": 0.0, "used_fraction": 0.0}, "pool": None, "burn": None, "credit": {"balance_usd": 0.0001, "balance_at": None, "balance_source": "telemetry"}, "energy": {"kwh_30d": 0.0, "calls_30d": 0}, }, { "provider": "openrouter", "shape": "prepaid_credit", "spend_usd": {"period": 0.5, "attribution_coverage": 1.0}, "plan": None, "pool": {"total_credits_usd": 5.0, "total_usage_usd": 0.001, "balance_usd": 4.999, "age_seconds": 60}, "burn": None, "credit": {"balance_usd": 4.999, "balance_at": None, "balance_source": "polled"}, "energy": {"kwh_30d": 0.0, "calls_30d": 0}, }, ], } out = tui._provider_quota_rows(quota_raw) assert len(out) == 2 assert out[0].startswith("neuralwatt") assert out[1].startswith("openrouter") def test_quota_panel_lead_is_kwh_only_and_note_empty(): """The quota lead is the kWh-plan fraction; the note line stays hidden.""" stub = _StubFetcher() data = _fixture() stub.payload = data app = tui.DashboardApp(fetcher=stub) def _assert(a): lead_text = str(a.query_one("#quota-lead", Static).content) assert "20% of 6.25 kWh plan" in lead_text assert "None" not in lead_text assert a.query_one("#quota-note", Static).display is False _run_app(app, _assert) def test_quota_legend_includes_provider_account_rows(): """The TUI quota legend lists one labeled row per provider after the kWh summary line.""" stub = _StubFetcher() data = _fixture() stub.payload = data app = tui.DashboardApp(fetcher=stub) def _assert(a): legend = str(a.query_one("#quota-legend", Static).content) assert "neuralwatt" in legend assert "kWh plan" in legend assert "None" not in legend _run_app(app, _assert) def test_quota_legend_lists_polled_provider_with_credits_label(): """A polled provider row shows the credits label, distinct from a telemetry provider's overage label.""" stub = _StubFetcher() data = _fixture() data["quota"]["accounts"].append({ "provider": "openrouter", "shape": "prepaid_credit", "spend_usd": {"period": 0.5, "attribution_coverage": 1.0}, "plan": None, "pool": {"total_credits_usd": 5.0, "total_usage_usd": 0.001, "balance_usd": 4.999, "age_seconds": 60}, "burn": None, "credit": {"balance_usd": 4.999, "balance_at": "2026-08-23T10:00:00+00:00", "balance_source": "polled"}, "energy": {"kwh_30d": 0.0, "calls_30d": 0}, }) stub.payload = data app = tui.DashboardApp(fetcher=stub) def _assert(a): legend = str(a.query_one("#quota-legend", Static).content) assert "openrouter" in legend _run_app(app, _assert) def test_format_small_balance_handles_boundary_and_zero(): """Exactly ±0.005 renders conventionally (not tiny); zero renders $0.00.""" assert tui._format_small_balance(0.005) == "$0.01" assert tui._format_small_balance(-0.005) == "$-0.01" assert tui._format_small_balance(0.0) == "$0.00" def test_format_provider_quota_account_prepaid_credit(): """prepaid_credit account renders pool balance and burn info.""" account = { "provider": "openrouter", "shape": "prepaid_credit", "spend_usd": {"period": 0.5, "attribution_coverage": 1.0}, "plan": None, "pool": {"total_credits_usd": 5.0, "total_usage_usd": 0.001, "balance_usd": 4.999, "age_seconds": 60}, "burn": None, "credit": {"balance_usd": 4.999, "balance_at": "2026-08-23T10:00:00+00:00", "balance_source": "polled"}, "energy": {"kwh_30d": 0.0, "calls_30d": 0}, } row = tui._format_provider_quota_account(account) assert row.startswith("openrouter") assert "pool" in row def test_format_provider_quota_account_unmetered(): """unmetered account shows minimal info.""" account = { "provider": "provider-x", "shape": "unmetered", "spend_usd": {"period": 3.25, "attribution_coverage": 0.5}, "plan": None, "pool": None, "burn": None, "credit": None, "energy": None, } row = tui._format_provider_quota_account(account) assert "provider-x" in row assert "unmetered" in row assert "$3.25" in row