"""Tests for the Textual monitoring dashboard (tui.py). Offline, no real router / network. The data layer (``build_model`` and ``fetch_metrics``) is importable without a running TUI, so nearly all assertions are on the rendered panel PAYLOADS (plain dicts of display rows), not on pixels. ``App.run_test`` drives the app itself with a stubbed fetcher. This file imports ``tui`` and ``textual`` deliberately — it is the ONE test file that may. No non-tui module imports textual. """ from __future__ import annotations import asyncio import pytest import tui from tui import build_model, fetch_metrics def _fixture() -> dict: """A representative GET /metrics payload (matches dispatcher's shape).""" return { "quota": { "plan_kwh": 6.25, "metered_kwh_30d": 1.25, "metered_fraction_of_plan": 0.2, "metered_calls_30d": 18, "note": "router-metered only", }, "coverage": { "routable_models": 13, "with_energy_data": 10, "with_proficiency_data": 12, "quota": { "plan_kwh": 6.25, "metered_kwh_30d": 1.25, "metered_fraction_of_plan": 0.2, "metered_calls_30d": 18, "note": "router-metered only", }, "warnings": [ "3/13 routable models have no reference-workload observations", "1/13 routable models have no proficiency data", ], }, "recent_decisions": [ { "id": 42, "observed_at": "2026-08-23T10:00:00+00:00", "kind": "chat", "task_category": "coding_general", "task_tier": 2, "selected_model": "deepseek-v4-flash", "selected_provider": "neuralwatt", "est_cost_usd": 0.00016, }, { "id": 41, "observed_at": "2026-08-23T09:59:00+00:00", "kind": "route", "task_category": "docs_writing", "task_tier": 3, "selected_model": "kimi-k2.7-code", "selected_provider": "neuralwatt", "est_cost_usd": 0.0136, }, { "id": 40, "observed_at": "2026-08-23T09:58:00+00:00", "kind": "chat", "task_category": "debugging", "task_tier": 1, "selected_model": None, "selected_provider": None, "est_cost_usd": None, }, ], "per_model": [ { "model_id": "deepseek-v4-flash", "provider": "neuralwatt", "calls": 12, "sum_cost_usd": 0.0012, "sum_energy_kwh": 2.5e-05, "sum_carbon_g_co2eq": 1.2e-04, }, { "model_id": "kimi-k3", "provider": "neuralwatt", "calls": 4, "sum_cost_usd": 0.08, "sum_energy_kwh": 1.0e-03, "sum_carbon_g_co2eq": 3.0e-03, }, ], "verdict_mix": {"ok": 5, "unverifiable": 2, "truncated": 1}, "top_proficiency": [ {"model_id": "deepseek-v4-flash", "provider": "neuralwatt", "blended_score": 1.0, "source": "self_eval_thin", "self_eval_samples": 3} ], "generated_at": "2026-08-23T10:01:00+00:00", } # -------------------------------------------------------------------------- # Direct unit tests of the data layer (no TUI running). # -------------------------------------------------------------------------- def test_build_model_quota_panel(): m = build_model(_fixture()) rows = m["quota"] # plan number is surfaced as a display row joined = " ".join(r["label"] + "=" + str(r["value"]) for r in rows) assert "plan_kwh=6.25" in joined assert "metered_kwh_30d=1.25" in joined assert "fraction=0.2" in joined assert "calls=18" in joined def test_build_model_per_model_lists_seeded_models(): m = build_model(_fixture()) rows = m["per_model"] assert rows[0]["model"] == "deepseek-v4-flash" assert rows[0]["calls"] == 12 assert rows[1]["model"] == "kimi-k3" # every row keeps numeric cost/energy/carbon for display assert rows[0]["cost_usd"] == 0.0012 assert rows[0]["energy_kwh"] == 2.5e-05 assert rows[0]["carbon_g_co2eq"] == 1.2e-04 def test_build_model_verdict_mix(): m = build_model(_fixture()) mix = m["verdict_mix"] by_verdict = {r["verdict"]: r["count"] for r in mix} assert by_verdict == {"ok": 5, "unverifiable": 2, "truncated": 1} def test_build_model_recent_decisions_top_rows(): m = build_model(_fixture()) rows = m["recent_decisions"] # DESC by id: first row is id 42 assert rows[0]["id"] == 42 assert rows[0]["kind"] == "chat" assert rows[0]["category"] == "coding_general" assert rows[0]["tier"] == 2 assert rows[0]["selected"] == "deepseek-v4-flash" assert rows[1]["kind"] == "route" assert rows[2]["selected"] == "none" # no-candidate row renders "none" def test_build_model_warnings_from_coverage(): m = build_model(_fixture()) warnings = m["warnings"] assert len(warnings) == 2 assert "reference-workload" in warnings[0] assert "proficiency data" in warnings[1] def test_build_model_handles_missing_quota(): """Empty DB / null plan: the quota section degrades to a notice.""" data = _fixture() data["quota"] = None data["coverage"]["quota"] = None m = build_model(data) joined = " ".join(str(r) for r in m["quota"]) assert "unset" in joined.lower() or "no plan" in joined.lower() or "n/a" in joined.lower() def test_fetch_metrics_returns_parsed_dict(monkeypatch): """fetch_metrics hits the right URL and returns parsed JSON.""" captured = {} class _FakeResp: def raise_for_status(self): return None def json(self): return {"quota": None, "ok": True} def _fake_get(url, timeout=None): captured["url"] = url captured["timeout"] = timeout return _FakeResp() monkeypatch.setattr(tui.requests, "get", _fake_get) out = fetch_metrics("http://testhost:8081") assert out == {"quota": None, "ok": True} assert captured["url"] == "http://testhost:8081/metrics" def test_fetch_metrics_raises_on_http_error(monkeypatch): class _Err: def raise_for_status(self): raise RuntimeError("500") monkeypatch.setattr(tui.requests, "get", lambda *a, **k: _Err()) with pytest.raises(Exception): fetch_metrics("http://x") def test_fetch_metrics_raises_on_network_error(monkeypatch): def _boom(*a, **k): raise ConnectionError("refused") monkeypatch.setattr(tui.requests, "get", _boom) with pytest.raises(ConnectionError): fetch_metrics("http://x") # -------------------------------------------------------------------------- # App-level tests via App.run_test with a stubbed fetch_metrics. # -------------------------------------------------------------------------- class _StubFetcher: """Swappable fake for fetch_metrics the App calls.""" def __init__(self): self.payload = None self.error = None self.calls = 0 def __call__(self, base_url): self.calls += 1 if self.error is not None: raise self.error return self.payload @pytest.mark.parametrize("fetcher_arg", ["callable", "subclass"]) def test_app_run_test_populates_quota_and_model_panels(fetcher_arg): stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): stash = getattr(a, "_last_model", None) assert stash is not None, "build_model result was not stashed on the app" # quota plan number surfaced quota_text = " ".join( f"{r['label']}={r['value']}" for r in stash["quota"] ) assert "plan_kwh=6.25" in quota_text # per-model lists the seeded models models = {r["model"] for r in stash["per_model"]} assert {"deepseek-v4-flash", "kimi-k3"} <= models assert stub.calls == 1 _run_app(app, _assert) def test_app_run_test_recent_and_warnings_panels(): stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): stash = a._last_model assert stash["recent_decisions"][0]["selected"] == "deepseek-v4-flash" assert len(stash["warnings"]) == 2 _run_app(app, _assert) def test_app_run_test_quota_panel_static_shows_plan(): """The rendered Static widget carries the plan number after a good fetch.""" stub = _StubFetcher() stub.payload = _fixture() app = tui.DashboardApp(fetcher=stub) def _assert(a): quota_widget = a.query_one("#quota-panel") assert "6.25" in str(quota_widget.content) _run_app(app, _assert) def test_app_run_test_failure_shows_error_and_does_not_crash(): """fetch raising -> error panel visible, run_test completes without raising.""" stub = _StubFetcher() stub.error = ConnectionError("cannot reach router") app = tui.DashboardApp(fetcher=stub, base_url="http://127.0.0.1:8080") def _assert(a): error_widget = a.query_one("#error-panel") assert "cannot reach router" in str(error_widget.content) assert a._last_error is not None _run_app(app, _assert) # must not raise def _run_app(app: tui.DashboardApp, body) -> None: """Drive the app via Textual App.run_test synchronously. ``body(app)`` runs while the app is mounted, so queries and the data model are live. Assertion failures inside propagate out of ``asyncio.run`` as normal test failures. """ async def _go(): async with app.run_test() as pilot: await pilot.pause() body(app) asyncio.run(_go()) # -------------------------------------------------------------------------- # Auto-refresh, error resilience and keyboard controls. # # These drive the app through App.run_test with a tiny REFRESH_SECONDS and no # real wall-clock sleep. Textual 8.2.8 schedules set_interval timers on the # asyncio event loop, so repeatedly awaiting ``pilot.pause()`` lets due ticks # fire without the test asserting on elapsed time or calling time.sleep(). # Assertions are on the re-rendered panel payload (``app._last_model``), not # on mock-call counts. # -------------------------------------------------------------------------- class _CountingFetcher: """Hands back a metrics payload whose first per-model call count equals the invocation number, so each fetch produces a distinct, inspectable payload with no script to exhaust.""" def __init__(self): self.calls = 0 def __call__(self, base_url): self.calls += 1 return _variant(self.calls) def _variant(calls: int) -> dict: """Return a metrics payload whose per-model call count is ``calls``.""" data = _fixture() data["per_model"][0]["calls"] = calls return data def _first_model_calls(app) -> int: """Read the re-rendered per-model payload's first row call count.""" return app._last_model["per_model"][0]["calls"] def test_auto_refresh_rerenders_updated_payload(): """Interval ticks re-fetch and re-render: the re-rendered panel tracks the fetcher's latest payload. No real sleep: the interval is tiny and every tick fires while the event loop is pumped through ``pilot.pause()``. The assertion reads the actual re-rendered payload back, not a mock-call count. """ fetcher = _CountingFetcher() app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=0.05) async def _go(): async with app.run_test() as pilot: await pilot.pause() # >1 fetch means an auto-tick fired beyond the on_mount refresh. for _ in range(30): await pilot.pause() assert fetcher.calls > 1 # The displayed panel reflects the fetcher's newest payload. assert _first_model_calls(app) == fetcher.calls assert app._refreshing is False asyncio.run(_go()) def test_error_resilience_keeps_last_good_data_and_recovers(): """A transient fetch failure keeps the app running and the last good data displayed; a later successful refresh re-renders the new payload. A large interval means no stray auto-ticks, so the 2nd fetch is exactly the failing one, driven deterministically through the same ``_on_interval`` callback the timer invokes — no sleep, no timing race. """ calls = {"n": 0} def _scripted(base_url): calls["n"] += 1 if calls["n"] == 2: raise ConnectionError("transient blip") return _variant(calls["n"]) app = tui.DashboardApp(fetcher=_scripted, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() assert _first_model_calls(app) == 1 # initial good fetch displayed # advance one tick (the failing 2nd fetch) app._on_interval() await pilot.pause() assert app._last_error is not None, "transient failure was not seen" assert not app._exit, "app must not exit on a transient failure" err_widget = app.query_one("#error-panel") assert "cannot reach router" in str(err_widget.content) assert "visible" in err_widget.classes # last good data is still the displayed payload assert _first_model_calls(app) == 1 # recover on the next refresh (3rd fetch, now successful) app._on_interval() await pilot.pause() assert _first_model_calls(app) == 3 assert app._last_error is None asyncio.run(_go()) def test_force_refresh_binding_reloads_on_r(): """Pressing ``r`` immediately re-fetches and re-renders a new payload.""" fetcher = _CountingFetcher() # A large interval ensures only the forced refresh advances the payload. app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() baseline = fetcher.calls await pilot.press("r") await pilot.pause() assert fetcher.calls == baseline + 1 assert _first_model_calls(app) == fetcher.calls asyncio.run(_go()) @pytest.mark.parametrize("key", ["q", "Q", "ctrl+c"]) def test_quit_bindings_exit_app(key): """q, Q and Ctrl+C all quit the running app.""" fetcher = _CountingFetcher() app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() assert not app._exit await pilot.press(key) assert app._exit asyncio.run(_go()) @pytest.mark.parametrize( "key,panel", [ ("1", "model-table"), ("2", "verdict-table"), ("3", "decision-table"), ("4", "quota-panel"), ("5", "warnings-panel"), ], ) def test_number_bindings_focus_panel(key, panel): """Number keys 1-5 focus the corresponding panel.""" fetcher = _CountingFetcher() app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60) async def _go(): async with app.run_test() as pilot: await pilot.pause() await pilot.press(key) await pilot.pause() widget = app.query_one(f"#{panel}") assert widget.has_focus, f"{panel} should have focus after {key!r}" asyncio.run(_go()) @pytest.fixture(autouse=True) def _no_real_network(monkeypatch): """Safety net: even if the fetcher is mis-wired, never hit a real router.""" def _guard(base_url): raise AssertionError(f"real fetch_metrics called with {base_url!r}") monkeypatch.setattr(tui, "fetch_metrics", _guard)