1970 lines
70 KiB
Python
1970 lines
70 KiB
Python
"""Tests for the Textual monitoring dashboard (tui.py).
|
|
|
|
Offline, no real router / network. The data layer (``build_model`` and
|
|
``fetch_metrics``) is importable without a running TUI, so nearly all
|
|
assertions are on the rendered panel PAYLOADS (plain dicts of display rows),
|
|
not on pixels. ``App.run_test`` drives the app itself with a stubbed fetcher.
|
|
|
|
This file imports ``tui`` and ``textual`` deliberately — it is the ONE test
|
|
file that may. No non-tui module imports textual.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import copy
|
|
import time
|
|
from datetime import datetime
|
|
|
|
import pytest
|
|
import requests
|
|
from textual.widgets import DataTable, ProgressBar, Static
|
|
|
|
import tui
|
|
import tui_model
|
|
from tui_model import build_model, fetch_metrics
|
|
from tui_screens import DecisionDetailScreen, VerdictMixScreen
|
|
|
|
|
|
def _fixture() -> dict:
|
|
"""A representative /admin/api/snapshot payload (quota_accounts shape)."""
|
|
return {
|
|
"quota": {
|
|
"period": {
|
|
"start": "2026-07-26",
|
|
"next_reset": "2026-08-26",
|
|
"elapsed_fraction": 0.5,
|
|
"source": "billing_reset_day",
|
|
},
|
|
"accounts": [
|
|
{
|
|
"provider": "neuralwatt",
|
|
"shape": "metered_plan",
|
|
"spend_usd": {"period": 1.25, "attribution_coverage": 1.0},
|
|
"plan": {"kwh_per_period": 6.25, "used_kwh": 1.25, "used_fraction": 0.2},
|
|
"pool": None,
|
|
"burn": {
|
|
"burn_rate_usd_per_hour": 1.0,
|
|
"projected_hours_remaining": 8.5,
|
|
},
|
|
"credit": {
|
|
"balance_usd": 8.50,
|
|
"balance_at": "2026-08-23T09:58:00+00:00",
|
|
"balance_source": "telemetry",
|
|
},
|
|
"energy": {"kwh_30d": 1.25, "calls_30d": 18},
|
|
}
|
|
],
|
|
"spend": {
|
|
"by_provider_usd": {"neuralwatt": 1.25},
|
|
"total_usd": 1.25,
|
|
"estimated_usd": 0.0,
|
|
"estimate_ratio": None,
|
|
},
|
|
"alarm": {
|
|
"kind": "none",
|
|
"severity": "info",
|
|
"headline": "$1.25 period spend",
|
|
},
|
|
},
|
|
"coverage": {
|
|
"routable_models": 13,
|
|
"with_energy_data": 10,
|
|
"with_proficiency_data": 12,
|
|
"quota": {
|
|
"plan_kwh": 6.25,
|
|
"metered_kwh_30d": 1.25,
|
|
"metered_fraction_of_plan": 0.2,
|
|
"metered_calls_30d": 18,
|
|
"reset_date": "2026-07-26",
|
|
"note": "router-metered only",
|
|
"total_balance_usd": 8.50,
|
|
"by_provider": {
|
|
"neuralwatt": {
|
|
"balance_usd": 8.50,
|
|
"balance_at": "2026-08-23T09:58:00+00:00",
|
|
"balance_source": "telemetry",
|
|
"burn_window_hours": 24,
|
|
"burn_rate_usd_per_hour": 1.0,
|
|
"projected_hours_remaining": 8.5,
|
|
"runway_low_warning": False,
|
|
"runway_note": None,
|
|
}
|
|
},
|
|
"window_start_30d": "2026-07-24",
|
|
"metered_kwh_period": 0.5,
|
|
},
|
|
"warnings": [
|
|
"3/13 routable models have no reference-workload observations",
|
|
"1/13 routable models have no proficiency data",
|
|
],
|
|
"flex_default": "auto",
|
|
},
|
|
"recent_decisions": [
|
|
{
|
|
"id": 42,
|
|
"observed_at": "2026-08-23T10:00:00+00:00",
|
|
"kind": "chat",
|
|
"task_category": "coding_general",
|
|
"task_tier": 2,
|
|
"selected_model": "deepseek-v4-flash",
|
|
"selected_provider": "neuralwatt",
|
|
"est_cost_usd": 0.00016,
|
|
"flex_preference": "force-flex",
|
|
"flex_swapped": 1,
|
|
"flex_forced": 1,
|
|
},
|
|
{
|
|
"id": 41,
|
|
"observed_at": "2026-08-23T09:59:00+00:00",
|
|
"kind": "route",
|
|
"task_category": "docs_writing",
|
|
"task_tier": 3,
|
|
"selected_model": "kimi-k2.7-code",
|
|
"selected_provider": "neuralwatt",
|
|
"est_cost_usd": 0.0136,
|
|
"flex_preference": "auto",
|
|
"flex_swapped": 0,
|
|
"flex_forced": 0,
|
|
},
|
|
{
|
|
"id": 40,
|
|
"observed_at": "2026-08-23T09:58:00+00:00",
|
|
"kind": "chat",
|
|
"task_category": "debugging",
|
|
"task_tier": 1,
|
|
"selected_model": None,
|
|
"selected_provider": None,
|
|
"est_cost_usd": None,
|
|
},
|
|
],
|
|
"per_model": [
|
|
{
|
|
"model_id": "deepseek-v4-flash",
|
|
"provider": "neuralwatt",
|
|
"calls": 12,
|
|
"sum_cost_usd": 0.0012,
|
|
"sum_energy_kwh": 2.5e-05,
|
|
"sum_carbon_g_co2eq": 1.2e-04,
|
|
},
|
|
{
|
|
"model_id": "kimi-k3",
|
|
"provider": "neuralwatt",
|
|
"calls": 4,
|
|
"sum_cost_usd": 0.08,
|
|
"sum_energy_kwh": 1.0e-03,
|
|
"sum_carbon_g_co2eq": 3.0e-03,
|
|
},
|
|
{
|
|
"model_id": "null-energy-model",
|
|
"provider": "openrouter",
|
|
"calls": 5,
|
|
"sum_cost_usd": None,
|
|
"sum_energy_kwh": None,
|
|
"sum_carbon_g_co2eq": None,
|
|
},
|
|
],
|
|
"verdict_mix": {"ok": 5, "unverifiable": 2, "truncated": 1},
|
|
"top_proficiency": [
|
|
{"model_id": "deepseek-v4-flash", "provider": "neuralwatt",
|
|
"blended_score": 1.0, "source": "self_eval_thin",
|
|
"self_eval_samples": 3}
|
|
],
|
|
"pinch": {
|
|
"calls_30d": 120,
|
|
"pruned_calls_30d": 30,
|
|
"share_pruned": 0.25,
|
|
"total_tokens_saved": 50000,
|
|
"median_tokens_saved": 1200,
|
|
"dollars_saved_usd_30d": 0.0125,
|
|
},
|
|
"generated_at": "2026-08-23T10:01:00+00:00",
|
|
}
|
|
|
|
|
|
def _enriched_payload() -> dict:
|
|
"""A deep copy of _fixture() whose recent_decisions rows carry the six
|
|
T2 data-layer fields, mirroring what /metrics returns after the drift
|
|
catch-up. Row 42 is the popup test's target row (exploration off);
|
|
row 41 is the exploratory pick; row 40 stays a rejection row."""
|
|
data = copy.deepcopy(_fixture())
|
|
enrichments = [
|
|
{
|
|
"profile": "default",
|
|
"exploration": 0,
|
|
"request_id": "chatcmpl-fixture-42",
|
|
"pinch_original_tokens": 120000,
|
|
"pinch_final_tokens": 96122,
|
|
"session_key": "sess-42",
|
|
"observed_at": "2026-08-23T10:00:00+00:00",
|
|
},
|
|
{
|
|
"profile": "default",
|
|
"exploration": 1,
|
|
"request_id": "chatcmpl-fixture-41",
|
|
"pinch_original_tokens": 90000,
|
|
"pinch_final_tokens": 87500,
|
|
"session_key": "sess-41",
|
|
"observed_at": "2026-08-23T09:59:00+00:00",
|
|
},
|
|
{
|
|
"profile": "default",
|
|
"exploration": 0,
|
|
"request_id": None,
|
|
"pinch_original_tokens": None,
|
|
"pinch_final_tokens": None,
|
|
"session_key": "sess-40",
|
|
"observed_at": "2026-08-23T09:58:00+00:00",
|
|
},
|
|
]
|
|
for row, extra in zip(data["recent_decisions"], enrichments):
|
|
row.update(extra)
|
|
return data
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Direct unit tests of the data layer (no TUI running).
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
def test_build_model_quota_panel():
|
|
m = build_model(_fixture())
|
|
rows = m["quota"]
|
|
# Plan info is surfaced via per-account rows
|
|
joined = " ".join(r["label"] + "=" + str(r["value"]) for r in rows)
|
|
assert "period_start=2026-07-26" in joined
|
|
assert "next_reset=2026-08-26" in joined
|
|
assert "spend_total_usd=1.25" in joined
|
|
assert "neuralwatt plan_kwh_per_period=6.25" in joined
|
|
assert "neuralwatt used_kwh=1.25" in joined
|
|
assert "neuralwatt used_fraction=0.2" in joined
|
|
assert "neuralwatt calls_30d=18" in joined
|
|
|
|
|
|
def test_build_model_pinch_panel():
|
|
m = build_model(_fixture())
|
|
rows = m["pinch"]
|
|
by_label = {r["label"]: r["value"] for r in rows}
|
|
assert by_label["share_pruned"] == 0.25
|
|
assert by_label["total_tokens_saved"] == 50000
|
|
assert by_label["median_tokens_saved"] == 1200
|
|
assert by_label["dollars_saved_usd_30d"] == 0.0125
|
|
|
|
|
|
def test_build_model_pinch_panel_empty_when_none():
|
|
data = _fixture()
|
|
data["pinch"] = None
|
|
m = build_model(data)
|
|
assert m["pinch"] == []
|
|
|
|
|
|
def test_build_model_quota_panel_includes_period_and_spend():
|
|
m = build_model(_fixture())
|
|
rows = m["quota"]
|
|
by_label = {r["label"]: r["value"] for r in rows}
|
|
assert by_label["period_start"] == "2026-07-26"
|
|
assert by_label["next_reset"] == "2026-08-26"
|
|
assert by_label["spend_total_usd"] == 1.25
|
|
|
|
|
|
def test_build_model_quota_panel_includes_provider_rows():
|
|
"""The per-provider quota shape reaches the TUI data model."""
|
|
m = build_model(_fixture())
|
|
rows = m["quota"]
|
|
by_label = {r["label"]: r["value"] for r in rows}
|
|
assert by_label["neuralwatt shape"] == "metered_plan"
|
|
assert by_label["neuralwatt plan_kwh_per_period"] == 6.25
|
|
assert by_label["neuralwatt used_kwh"] == 1.25
|
|
assert by_label["neuralwatt used_fraction"] == 0.2
|
|
assert by_label["neuralwatt burn_rate_usd_per_hour"] == 1.0
|
|
assert by_label["neuralwatt projected_hours_remaining"] == 8.5
|
|
|
|
|
|
def test_build_model_quota_panel_drops_flat_balance_keys():
|
|
"""Old flat balance keys must not leak into the TUI row list."""
|
|
m = build_model(_fixture())
|
|
labels = {r["label"] for r in m["quota"]}
|
|
assert "balance_usd" not in labels
|
|
assert "burn_rate_usd_per_hour" not in labels
|
|
assert "total_balance_usd" not in labels
|
|
|
|
|
|
def test_build_model_per_model_lists_seeded_models():
|
|
m = build_model(_fixture())
|
|
rows = m["per_model"]
|
|
assert rows[0]["model"] == "deepseek-v4-flash"
|
|
assert rows[0]["calls"] == 12
|
|
assert rows[1]["model"] == "kimi-k3"
|
|
# every row keeps numeric cost/energy/carbon for display
|
|
assert rows[0]["cost_usd"] == 0.0012
|
|
assert rows[0]["energy_kwh"] == 2.5e-05
|
|
assert rows[0]["carbon_g_co2eq"] == 1.2e-04
|
|
|
|
|
|
def test_build_model_per_model_handles_null_energy():
|
|
"""A per-model row with all-NULL cost/energy/carbon must not crash and
|
|
must preserve the None values for the rendering layer."""
|
|
m = build_model(_fixture())
|
|
rows = m["per_model"]
|
|
null_row = [r for r in rows if r["model"] == "null-energy-model"]
|
|
assert len(null_row) == 1
|
|
nr = null_row[0]
|
|
assert nr["cost_usd"] is None
|
|
assert nr["energy_kwh"] is None
|
|
assert nr["carbon_g_co2eq"] is None
|
|
|
|
|
|
def test_build_model_verdict_mix():
|
|
m = build_model(_fixture())
|
|
mix = m["verdict_mix"]
|
|
by_verdict = {r["verdict"]: r["count"] for r in mix}
|
|
assert by_verdict == {"ok": 5, "unverifiable": 2, "truncated": 1}
|
|
|
|
|
|
def test_build_model_recent_decisions_top_rows():
|
|
m = build_model(_fixture())
|
|
rows = m["recent_decisions"]
|
|
# DESC by id: first row is id 42
|
|
assert rows[0]["id"] == 42
|
|
assert rows[0]["kind"] == "chat"
|
|
assert rows[0]["category"] == "coding_general"
|
|
assert rows[0]["tier"] == 2
|
|
assert rows[0]["selected"] == "deepseek-v4-flash"
|
|
assert rows[1]["kind"] == "route"
|
|
assert rows[2]["selected"] == "none" # no-candidate row renders "none"
|
|
|
|
|
|
def test_build_model_warnings_from_coverage():
|
|
m = build_model(_fixture())
|
|
warnings = m["warnings"]
|
|
assert len(warnings) == 2
|
|
assert "reference-workload" in warnings[0]
|
|
assert "proficiency data" in warnings[1]
|
|
|
|
|
|
def test_build_model_handles_missing_quota():
|
|
"""Empty DB / null plan: the quota section is an empty list (renderer hides it)."""
|
|
data = _fixture()
|
|
data["quota"] = None
|
|
data["coverage"]["quota"] = None
|
|
m = build_model(data)
|
|
assert m["quota"] == []
|
|
|
|
|
|
def test_fetch_metrics_returns_parsed_dict(monkeypatch):
|
|
"""fetch_metrics hits the right URL and returns parsed JSON."""
|
|
captured = {}
|
|
|
|
class _FakeResp:
|
|
def raise_for_status(self):
|
|
return None
|
|
|
|
def json(self):
|
|
return {"quota": None, "ok": True}
|
|
|
|
def _fake_get(url, timeout=None):
|
|
captured["url"] = url
|
|
captured["timeout"] = timeout
|
|
return _FakeResp()
|
|
|
|
monkeypatch.setattr(tui_model.requests, "get", _fake_get)
|
|
out = fetch_metrics("http://testhost:8081")
|
|
assert out == {"quota": None, "ok": True}
|
|
assert captured["url"] == "http://testhost:8081/metrics"
|
|
|
|
|
|
def test_fetch_metrics_raises_on_http_error(monkeypatch):
|
|
class _Err:
|
|
def raise_for_status(self):
|
|
raise RuntimeError("500")
|
|
|
|
monkeypatch.setattr(tui_model.requests, "get", lambda *a, **k: _Err())
|
|
with pytest.raises(Exception):
|
|
fetch_metrics("http://x")
|
|
|
|
|
|
def test_fetch_metrics_raises_on_network_error(monkeypatch):
|
|
def _boom(*a, **k):
|
|
raise ConnectionError("refused")
|
|
|
|
monkeypatch.setattr(tui_model.requests, "get", _boom)
|
|
with pytest.raises(ConnectionError):
|
|
fetch_metrics("http://x")
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# App-level tests via App.run_test with a stubbed fetch_metrics.
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
class _StubFetcher:
|
|
"""Swappable fake for fetch_metrics the App calls."""
|
|
|
|
def __init__(self):
|
|
self.payload = None
|
|
self.error = None
|
|
self.calls = 0
|
|
|
|
def __call__(self, base_url):
|
|
self.calls += 1
|
|
if self.error is not None:
|
|
raise self.error
|
|
return self.payload
|
|
|
|
|
|
@pytest.mark.parametrize("fetcher_arg", ["callable", "subclass"])
|
|
def test_app_run_test_populates_quota_and_model_panels(fetcher_arg):
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
stash = getattr(a, "_last_model", None)
|
|
assert stash is not None, "build_model result was not stashed on the app"
|
|
# quota plan number surfaced
|
|
quota_text = " ".join(
|
|
f"{r['label']}={r['value']}" for r in stash["quota"]
|
|
)
|
|
assert "neuralwatt plan_kwh_per_period=6.25" in quota_text
|
|
# per-model lists the seeded models
|
|
models = {r["model"] for r in stash["per_model"]}
|
|
assert {"deepseek-v4-flash", "kimi-k3"} <= models
|
|
assert stub.calls == 1
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_app_run_test_recent_and_warnings_panels():
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
stash = a._last_model
|
|
assert stash["recent_decisions"][0]["selected"] == "deepseek-v4-flash"
|
|
assert len(stash["warnings"]) == 2
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_app_run_test_duplicate_model_keys_do_not_crash():
|
|
"""Duplicate model ids in the /metrics per_model payload must not raise
|
|
DuplicateKeyError when rendered into #model-table. The source may be a
|
|
poller/anomaly, so the TUI dedups defensively."""
|
|
stub = _StubFetcher()
|
|
payload = _fixture()
|
|
payload["per_model"].append(dict(payload["per_model"][0], calls=99))
|
|
stub.payload = payload
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
mt = a.query_one("#model-table")
|
|
assert mt.row_count == 3, (
|
|
f"expected 3 deduped rows, got {mt.row_count}"
|
|
)
|
|
assert len(payload["per_model"]) == 4
|
|
assert len(a._last_model["per_model"]) == 4
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_app_run_test_null_energy_row_renders_safely():
|
|
"""A per-model row with NULL cost/energy/carbon must render as
|
|
'n/a' for cost and em-dash for energy/carbon without crashing."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
mt = a.query_one("#model-table", DataTable)
|
|
# Find the null-energy-model row by scanning model names (col 0)
|
|
null_row_idx = None
|
|
for idx in range(mt.row_count):
|
|
cell = mt.get_cell_at((idx, 0))
|
|
if "null-energy-model" in str(cell):
|
|
null_row_idx = idx
|
|
break
|
|
assert null_row_idx is not None, "null-energy-model row not found"
|
|
cost_cell = str(mt.get_cell_at((null_row_idx, 2)))
|
|
kwh_cell = str(mt.get_cell_at((null_row_idx, 3)))
|
|
co2_cell = str(mt.get_cell_at((null_row_idx, 4)))
|
|
assert "n/a" in cost_cell, f"expected n/a for cost, got {cost_cell}"
|
|
assert "\u2014" in kwh_cell, f"expected em-dash for kWh, got {kwh_cell}"
|
|
assert "\u2014" in co2_cell, f"expected em-dash for gCO2eq, got {co2_cell}"
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_app_run_test_quota_panel_progress_bar_and_legend():
|
|
"""The quota panel renders a ProgressBar and a legend carrying the plan.
|
|
|
|
``#quota-panel`` is a container holding a ``ProgressBar``
|
|
(``#quota-progress``) and a ``Static`` legend (``#quota-legend``). The bar
|
|
is filled to the metered kWh against the plan kWh total, and the legend
|
|
shows the metered / plan / fraction / calls summary.
|
|
"""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
container = a.query_one("#quota-panel")
|
|
assert container is not None
|
|
|
|
bar = a.query_one("#quota-progress", ProgressBar)
|
|
assert bar.progress == 1.25
|
|
assert bar.total == 6.25
|
|
|
|
legend = a.query_one("#quota-legend", Static)
|
|
text = str(legend.content)
|
|
assert "6.25" in text
|
|
assert "1.25" in text
|
|
assert "period since 2026-07-26" in text
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_format_quota_legend_omits_none_when_unmetered():
|
|
"""With unmetered (None) values the legend never shows the literal "None".
|
|
|
|
The ``metered``/``frac`` rows can be absent (None) for a plan
|
|
that has not been metered yet; they must render as ``n/a`` and the
|
|
returned string must not contain the substring ``"None"``.
|
|
"""
|
|
app = tui.DashboardApp(fetcher=_StubFetcher())
|
|
legend = app._format_quota_legend(
|
|
plan=6.25, metered=None, frac=None,
|
|
flex_default=None,
|
|
)
|
|
assert "None" not in legend
|
|
assert "n/a" in legend
|
|
|
|
|
|
def test_format_quota_legend_happy_path_contains_numbers():
|
|
"""A fully-populated legend carries the metered / plan values."""
|
|
app = tui.DashboardApp(fetcher=_StubFetcher())
|
|
legend = app._format_quota_legend(
|
|
plan=6.25, metered=1.25, frac=0.2,
|
|
)
|
|
assert "6.25" in legend
|
|
assert "1.25" in legend
|
|
assert "20%" in legend
|
|
assert "None" not in legend
|
|
|
|
|
|
def test_format_quota_legend_distinguishes_period_from_window():
|
|
"""The legend shows metered vs plan — period and window labels are
|
|
now appended in _render, not inside _format_quota_legend."""
|
|
app = tui.DashboardApp(fetcher=_StubFetcher())
|
|
legend = app._format_quota_legend(
|
|
plan=6.25, metered=1.25, frac=0.2,
|
|
)
|
|
assert "6.25" in legend
|
|
assert "1.25" in legend
|
|
assert "None" not in legend
|
|
|
|
|
|
def test_format_quota_legend_shows_flex_default():
|
|
"""The configured flex default is surfaced in the quota legend readout."""
|
|
app = tui.DashboardApp(fetcher=_StubFetcher())
|
|
legend = app._format_quota_legend(
|
|
plan=6.25, metered=1.25, frac=0.2, flex_default="auto",
|
|
)
|
|
assert "flex default auto" in legend
|
|
|
|
|
|
def test_format_quota_legend_omits_flex_when_absent():
|
|
"""No flex default in the payload -> the legend does not claim one."""
|
|
app = tui.DashboardApp(fetcher=_StubFetcher())
|
|
legend = app._format_quota_legend(
|
|
plan=6.25, metered=1.25, frac=0.2,
|
|
)
|
|
assert "flex default" not in legend
|
|
|
|
|
|
def test_app_run_test_quota_bar_hidden_when_plan_unconfigured():
|
|
"""No plan configured -> the ProgressBar is hidden (display False).
|
|
|
|
The bar is not removed from the DOM; its ``display`` is toggled so the
|
|
"quota not configured" legend still shows and the widget state is kept
|
|
for when a plan is later configured.
|
|
"""
|
|
stub = _StubFetcher()
|
|
data = _fixture()
|
|
data["quota"]["accounts"] = [] # no metered_plan account → bar hidden
|
|
data["coverage"] = {"quota": None, "warnings": [], "flex_default": None}
|
|
stub.payload = data
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
bar = a.query_one("#quota-progress", ProgressBar)
|
|
assert bar.display is False, "bar should be hidden when plan unconfigured"
|
|
|
|
legend = a.query_one("#quota-legend", Static)
|
|
assert "quota not configured" in str(legend.content)
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_app_run_test_failure_shows_error_and_does_not_crash():
|
|
"""fetch raising -> error panel visible, run_test completes without raising."""
|
|
stub = _StubFetcher()
|
|
stub.error = ConnectionError("cannot reach router")
|
|
|
|
app = tui.DashboardApp(fetcher=stub, base_url="http://127.0.0.1:8080")
|
|
|
|
def _assert(a):
|
|
error_widget = a.query_one("#error-panel")
|
|
assert "cannot reach router" in str(error_widget.content)
|
|
assert a._last_error is not None
|
|
|
|
_run_app(app, _assert) # must not raise
|
|
|
|
|
|
def test_app_run_test_decision_table_focused_on_mount():
|
|
"""#decision-table is focused by default after the app mounts."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
table = app.query_one("#decision-table")
|
|
assert table.has_focus, "#decision-table should have focus on mount"
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def _run_app(app: tui.DashboardApp, body) -> None:
|
|
"""Drive the app via Textual App.run_test synchronously.
|
|
|
|
``body(app)`` runs while the app is mounted, so queries and the data
|
|
model are live. Assertion failures inside propagate out of ``asyncio.run``
|
|
as normal test failures.
|
|
"""
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
body(app)
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Auto-refresh, error resilience and keyboard controls.
|
|
#
|
|
# These drive the app through App.run_test with a tiny REFRESH_SECONDS and no
|
|
# real wall-clock sleep. Textual 8.2.8 schedules set_interval timers on the
|
|
# asyncio event loop, so repeatedly awaiting ``pilot.pause()`` lets due ticks
|
|
# fire without the test asserting on elapsed time or calling time.sleep().
|
|
# Assertions are on the re-rendered panel payload (``app._last_model``), not
|
|
# on mock-call counts.
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
class _CountingFetcher:
|
|
"""Hands back a metrics payload whose first per-model call count equals
|
|
the invocation number, so each fetch produces a distinct, inspectable
|
|
payload with no script to exhaust."""
|
|
|
|
def __init__(self):
|
|
self.calls = 0
|
|
|
|
def __call__(self, base_url):
|
|
self.calls += 1
|
|
return _variant(self.calls)
|
|
|
|
|
|
def _variant(calls: int) -> dict:
|
|
"""Return a metrics payload whose per-model call count is ``calls``."""
|
|
data = _fixture()
|
|
data["per_model"][0]["calls"] = calls
|
|
return data
|
|
|
|
|
|
def _first_model_calls(app) -> int:
|
|
"""Read the re-rendered per-model payload's first row call count."""
|
|
return app._last_model["per_model"][0]["calls"]
|
|
|
|
|
|
def test_auto_refresh_rerenders_updated_payload():
|
|
"""Interval ticks re-fetch and re-render: the re-rendered panel tracks the
|
|
fetcher's latest payload.
|
|
|
|
No real sleep: the interval is tiny and every tick fires while the event
|
|
loop is pumped through ``pilot.pause()``. The assertion reads the actual
|
|
re-rendered payload back, not a mock-call count.
|
|
"""
|
|
fetcher = _CountingFetcher()
|
|
app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=0.05)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
# >1 fetch means an auto-tick fired beyond the on_mount refresh.
|
|
for _ in range(30):
|
|
await pilot.pause()
|
|
assert fetcher.calls > 1
|
|
# The displayed panel reflects the fetcher's newest payload.
|
|
assert _first_model_calls(app) == fetcher.calls
|
|
assert app._refreshing is False
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_error_resilience_keeps_last_good_data_and_recovers():
|
|
"""A transient fetch failure keeps the app running and the last good data
|
|
displayed; a later successful refresh re-renders the new payload.
|
|
|
|
A large interval means no stray auto-ticks, so the 2nd fetch is exactly the
|
|
failing one, driven deterministically through the same ``_on_interval``
|
|
callback the timer invokes — no sleep, no timing race.
|
|
"""
|
|
calls = {"n": 0}
|
|
|
|
def _scripted(base_url):
|
|
calls["n"] += 1
|
|
if calls["n"] == 2:
|
|
raise ConnectionError("transient blip")
|
|
return _variant(calls["n"])
|
|
|
|
app = tui.DashboardApp(fetcher=_scripted, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
assert _first_model_calls(app) == 1 # initial good fetch displayed
|
|
# advance one tick (the failing 2nd fetch)
|
|
app._on_interval()
|
|
await pilot.pause()
|
|
assert app._last_error is not None, "transient failure was not seen"
|
|
assert not app._exit, "app must not exit on a transient failure"
|
|
err_widget = app.query_one("#error-panel")
|
|
assert "cannot reach router" in str(err_widget.content)
|
|
assert "visible" in err_widget.classes
|
|
# last good data is still the displayed payload
|
|
assert _first_model_calls(app) == 1
|
|
# recover on the next refresh (3rd fetch, now successful)
|
|
app._on_interval()
|
|
await pilot.pause()
|
|
assert _first_model_calls(app) == 3
|
|
assert app._last_error is None
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_force_refresh_binding_reloads_on_r():
|
|
"""Pressing ``r`` immediately re-fetches and re-renders a new payload."""
|
|
fetcher = _CountingFetcher()
|
|
# A large interval ensures only the forced refresh advances the payload.
|
|
app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
baseline = fetcher.calls
|
|
await pilot.press("r")
|
|
await pilot.pause()
|
|
assert fetcher.calls == baseline + 1
|
|
assert _first_model_calls(app) == fetcher.calls
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
@pytest.mark.parametrize("key", ["q", "Q", "ctrl+c"])
|
|
def test_quit_bindings_exit_app(key):
|
|
"""q, Q and Ctrl+C all quit the running app."""
|
|
fetcher = _CountingFetcher()
|
|
app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
assert not app._exit
|
|
await pilot.press(key)
|
|
assert app._exit
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"key,panel",
|
|
[
|
|
("1", "model-table"),
|
|
("2", "decision-table"),
|
|
("3", "category-table"),
|
|
("4", "quota-panel"),
|
|
("5", "warnings-panel"),
|
|
],
|
|
)
|
|
def test_number_bindings_focus_panel(key, panel):
|
|
"""Number keys 1-5 focus the corresponding panel."""
|
|
fetcher = _CountingFetcher()
|
|
app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
await pilot.press(key)
|
|
await pilot.pause()
|
|
widget = app.query_one(f"#{panel}")
|
|
assert widget.has_focus, f"{panel} should have focus after {key!r}"
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Cursor persistence across refreshes (decision table).
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
def _decisions_payload(*ids: int) -> dict:
|
|
"""A /metrics payload whose recent_decisions carry the given ids (newest
|
|
first), all otherwise identical to the fixture's first row shape."""
|
|
base = _fixture()["recent_decisions"][0]
|
|
return {"recent_decisions": [dict(base, id=i) for i in ids]}
|
|
|
|
|
|
def test_cursor_persistence_across_refresh():
|
|
"""Highlighting a row survives a refresh that shifts it: after a new
|
|
decision is prepended, the cursor follows the same logical decision
|
|
(by stable row key ``str(id)``) instead of resetting to row 0."""
|
|
payloads = [_decisions_payload(42, 41, 40), _decisions_payload(43, 42, 41, 40)]
|
|
|
|
class _Scripted:
|
|
def __init__(self):
|
|
self.calls = 0
|
|
|
|
def __call__(self, base_url):
|
|
p = payloads[self.calls]
|
|
self.calls += 1
|
|
return p
|
|
|
|
fetcher = _Scripted()
|
|
app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
dt = app.query_one("#decision-table")
|
|
# Row 1 is id 41; highlight it.
|
|
dt.focus()
|
|
dt.move_cursor(row=1)
|
|
await pilot.pause()
|
|
assert dt.cursor_coordinate.row == 1
|
|
# Logically the highlighted decision id.
|
|
highlighted_id = app._last_model["recent_decisions"][1]["id"]
|
|
assert highlighted_id == 41
|
|
|
|
# Refresh to a payload with a new decision prepended.
|
|
app.action_refresh()
|
|
await pilot.pause()
|
|
assert fetcher.calls == 2
|
|
# id 41 now sits at index 2; the cursor must have followed it.
|
|
assert app._last_model["recent_decisions"][2]["id"] == 41
|
|
assert dt.cursor_coordinate.row == 2, (
|
|
f"cursor reset to {dt.cursor_coordinate.row}; expected 2 "
|
|
"(same logical decision across the refresh)"
|
|
)
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_cursor_persistence_clamps_when_selected_row_removed():
|
|
"""When the highlighted decision disappears from the payload on refresh,
|
|
the cursor rests at row 0 (the default ``clear()`` already set) instead of
|
|
clamping to a stale anchor."""
|
|
payloads = [_decisions_payload(42, 41, 40), _decisions_payload(40, 39)]
|
|
|
|
class _Scripted:
|
|
def __init__(self):
|
|
self.calls = 0
|
|
|
|
def __call__(self, base_url):
|
|
p = payloads[self.calls]
|
|
self.calls += 1
|
|
return p
|
|
|
|
fetcher = _Scripted()
|
|
app = tui.DashboardApp(fetcher=fetcher, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
dt = app.query_one("#decision-table")
|
|
dt.focus()
|
|
dt.move_cursor(row=1) # id 41
|
|
await pilot.pause()
|
|
assert dt.cursor_coordinate.row == 1
|
|
|
|
app.action_refresh() # id 41 is dropped in the new payload
|
|
await pilot.pause()
|
|
# The cursor rests at the cleared default row 0.
|
|
assert dt.cursor_coordinate.row == 0, (
|
|
f"cursor at {dt.cursor_coordinate.row}; expected 0 "
|
|
"(dropped row should rest at the cleared default)"
|
|
)
|
|
# The table reflects the two-row payload.
|
|
assert app._last_model["recent_decisions"][0]["id"] == 40
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _no_real_network(monkeypatch):
|
|
"""Safety net: even if the fetcher is mis-wired, never hit a real router."""
|
|
def _guard(base_url):
|
|
raise AssertionError(f"real fetch_metrics called with {base_url!r}")
|
|
|
|
monkeypatch.setattr(tui, "fetch_metrics", _guard)
|
|
# Also guard the SSE consumer's requests so a stray DecisionStream never
|
|
# reaches the network even if a test forgets to disable live_events.
|
|
import tui_sse
|
|
|
|
def _sse_guard(*args, **kwargs):
|
|
raise AssertionError(
|
|
f"real requests.get called from tui_sse with {args!r} {kwargs!r}"
|
|
)
|
|
|
|
monkeypatch.setattr(tui_sse.requests, "get", _sse_guard)
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Category breakdown and enriched decision fields (pure data-layer tests).
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
def test_build_category_breakdown_majority_and_share():
|
|
"""One category, two different winners: majority is the most common and
|
|
the share is its fraction of the count."""
|
|
decisions = [
|
|
{"id": 3, "category": "coding_general", "tier": 2, "selected": "a"},
|
|
{"id": 2, "category": "coding_general", "tier": 2, "selected": "a"},
|
|
{"id": 1, "category": "coding_general", "tier": 2, "selected": "b"},
|
|
]
|
|
rows = tui_model.build_category_breakdown(decisions)
|
|
assert len(rows) == 1
|
|
row = rows[0]
|
|
assert row["category"] == "coding_general"
|
|
assert row["tier"] == 2
|
|
assert row["count"] == 3
|
|
assert row["majority"] == "a"
|
|
assert row["share"] == round(2 / 3, 2)
|
|
|
|
|
|
def test_build_category_breakdown_separates_tiers():
|
|
"""Same category, different tiers are separate buckets."""
|
|
decisions = [
|
|
{"id": 2, "category": "coding_general", "tier": 1, "selected": "tiny"},
|
|
{"id": 1, "category": "coding_general", "tier": 3, "selected": "big"},
|
|
]
|
|
rows = tui_model.build_category_breakdown(decisions)
|
|
assert len(rows) == 2
|
|
tiers = {r["tier"] for r in rows}
|
|
assert tiers == {1, 3}
|
|
|
|
|
|
def test_build_category_breakdown_handles_empty_and_none_selected():
|
|
"""No decisions: empty list. Decisions with no selected model get
|
|
majority 'none' and share 1.0 (they all count toward the bucket)."""
|
|
assert tui_model.build_category_breakdown([]) == []
|
|
|
|
rows = tui_model.build_category_breakdown(
|
|
[{"id": 1, "category": "x", "tier": 1, "selected": None}]
|
|
)
|
|
assert rows[0]["majority"] == "none"
|
|
|
|
|
|
def test_build_model_recent_decisions_carry_enriched_fields():
|
|
"""The enriched /metrics row fields must reach the TUI data model so the
|
|
detail popup can render the full decision in full."""
|
|
data = _fixture()
|
|
# Ensure the fixture's first row has the fields the new model surfaces.
|
|
data["recent_decisions"][0].update(
|
|
{
|
|
"required_context_tokens": 50000,
|
|
"confidence": 0.92,
|
|
"classifier_ms": 1800,
|
|
"classification_source": "classifier",
|
|
"latency_tolerance": "interactive",
|
|
"candidates_considered": 8,
|
|
"runner_up_models": '[{"model_id":"kimi-k3","provider":"neuralwatt"}]',
|
|
"est_proficiency": 0.9,
|
|
"rejected_reason": None,
|
|
"tools": 0,
|
|
"images": 0,
|
|
"json_mode": 0,
|
|
"streamed": 1,
|
|
}
|
|
)
|
|
m = build_model(data)
|
|
row = m["recent_decisions"][0]
|
|
assert row["required_context_tokens"] == 50000
|
|
assert row["confidence"] == 0.92
|
|
assert row["runner_up_models"].startswith("[{")
|
|
assert row["streamed"] == 1
|
|
# The breakdown is always present (even an empty list proves the key).
|
|
assert "category_breakdown" in m
|
|
|
|
|
|
def test_decision_row_projects_flex_fields():
|
|
"""decision_row must project the three flex telemetry fields so they reach
|
|
both the /metrics path and the live SSE path (shared by build_model)."""
|
|
row = tui_model.decision_row(
|
|
{
|
|
"flex_preference": "prefer-flex",
|
|
"flex_swapped": 1,
|
|
"flex_forced": 0,
|
|
}
|
|
)
|
|
assert row["flex_preference"] == "prefer-flex"
|
|
assert row["flex_swapped"] == 1
|
|
assert row["flex_forced"] == 0
|
|
|
|
|
|
def test_decision_row_missing_flex_fields_default_to_none():
|
|
"""Rows without flex columns (older payloads) yield None, not a crash."""
|
|
row = tui_model.decision_row({"id": 1})
|
|
assert row["flex_preference"] is None
|
|
assert row["flex_swapped"] is None
|
|
assert row["flex_forced"] is None
|
|
|
|
|
|
def test_build_model_threads_flex_default():
|
|
"""build_model carries the configured flex default through from coverage."""
|
|
m = build_model(_fixture())
|
|
assert m["flex_default"] == "auto"
|
|
|
|
|
|
def test_flex_indicator_labels():
|
|
"""The compact flex cell: plain preference by default, markers on a swap."""
|
|
assert tui._flex_indicator({"flex_preference": "auto",
|
|
"flex_swapped": 0, "flex_forced": 0}) == "auto"
|
|
assert tui._flex_indicator({"flex_preference": "prefer-flex",
|
|
"flex_swapped": 1, "flex_forced": 0}) == "prefer-flex!"
|
|
assert tui._flex_indicator({"flex_preference": "force-flex",
|
|
"flex_swapped": 1, "flex_forced": 1}) == "force!"
|
|
assert tui._flex_indicator({"id": 1}) == ""
|
|
|
|
|
|
def test_app_decision_table_renders_flex_column():
|
|
"""The decision table renders a flex indicator per row and the configured
|
|
flex default appears in the UI."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
dt = a.query_one("#decision-table")
|
|
first_row_text = " ".join(str(c) for c in dt.get_row_at(0))
|
|
assert "force!" in first_row_text
|
|
second_row_text = " ".join(str(c) for c in dt.get_row_at(1))
|
|
assert "auto" in second_row_text
|
|
legend = a.query_one("#quota-legend", Static)
|
|
assert "flex default auto" in str(legend.content)
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_keys_legend_built_from_short_pairs_dedups_quit():
|
|
"""The bottom keys legend is built from compact short pairs and collapses
|
|
the three quit bindings (q/Q/ctrl+c) to a single 'quit'."""
|
|
legend = tui._keys_legend()
|
|
assert isinstance(legend, str)
|
|
# every short key is present
|
|
for key in ("q", "r", "e", "ctrl+v", "1", "2", "3", "4", "5"):
|
|
assert f"[bold]{key}[/bold]" in legend
|
|
# short labels are present
|
|
for label in ("quit", "refresh", "detail", "mix", "model", "decisions",
|
|
"breakdown", "quota", "warnings"):
|
|
assert label in legend
|
|
# exactly one quit (q/Q/ctrl+c collapsed)
|
|
assert legend.count("quit") == 1
|
|
|
|
|
|
def test_app_run_test_renders_keys_legend():
|
|
"""The app renders a #keys-legend Static (not a Footer) carrying the
|
|
full compact legend string regardless of terminal width, so no key is
|
|
truncated away."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
legend = a.query_one("#keys-legend", Static)
|
|
text = str(legend.content)
|
|
# full legend is present (wrapped, not truncated) on a narrow window
|
|
for key in ("q", "r", "e", "2", "5"):
|
|
assert f"[bold]{key}[/bold]" in text
|
|
assert text.count("quit") == 1
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_app_run_test_keys_legend_not_truncated_at_narrow_width():
|
|
"""At a narrow terminal width the legend's content is intact (wraps rather
|
|
than drops keys), unlike Textual's Footer which would ellipsize."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
async def _go():
|
|
async with app.run_test(size=(40, 30)) as pilot:
|
|
await pilot.pause()
|
|
legend = app.query_one("#keys-legend", Static)
|
|
text = str(legend.content)
|
|
# every short key survives narrow rendering — nothing truncated
|
|
for key in ("q", "r", "e", "ctrl+v", "1", "2", "3", "4", "5"):
|
|
assert f"[bold]{key}[/bold]" in text
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Detail popup and live SSE decision handling (App-level tests).
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
def test_show_decision_detail_pushes_modal_with_full_row():
|
|
"""Pressing ``e`` on the decisions table opens a modal whose body contains
|
|
the full JSON of the selected row — not just the table columns."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
stub.payload["recent_decisions"][0].update(
|
|
{
|
|
"required_context_tokens": 50000,
|
|
"confidence": 0.92,
|
|
"runner_up_models": '[{"model_id":"kimi-k3"}]',
|
|
"rejected_reason": None,
|
|
}
|
|
)
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
# Focus the decisions table and move to the first row, then open
|
|
# the detail popup via the dedicated binding.
|
|
app.query_one("#decision-table").focus()
|
|
await pilot.pause()
|
|
await pilot.press("e")
|
|
await pilot.pause()
|
|
# A modal screen is now active and carries the selected decision.
|
|
from textual.screen import ModalScreen
|
|
|
|
assert isinstance(app.screen, ModalScreen)
|
|
decision = app.screen.decision
|
|
assert decision["id"] == 42
|
|
assert decision["required_context_tokens"] == 50000
|
|
assert "kimi-k3" in decision["runner_up_models"]
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_detail_popup_carries_forensics_fields():
|
|
"""The e-key popup carries the six T2 forensics fields — profile,
|
|
exploration, pinch tokens, request_id, session_key — both on the live
|
|
decision dict and in the rendered JSON for a metrics-shaped full row."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _enriched_payload()
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
app.query_one("#decision-table").focus()
|
|
await pilot.pause()
|
|
await pilot.press("e")
|
|
await pilot.pause()
|
|
from textual.screen import ModalScreen
|
|
|
|
assert isinstance(app.screen, ModalScreen)
|
|
decision = app.screen.decision
|
|
assert decision["request_id"] == "chatcmpl-fixture-42"
|
|
assert decision["session_key"] == "sess-42"
|
|
assert decision["pinch_original_tokens"] == 120000
|
|
assert decision["pinch_final_tokens"] == 96122
|
|
assert decision["profile"] == "default"
|
|
assert decision["exploration"] == 0
|
|
|
|
asyncio.run(_go())
|
|
|
|
metrics_row = {
|
|
"id": 43,
|
|
"observed_at": "2026-09-05T12:34:56+00:00",
|
|
"kind": "chat",
|
|
"task_category": "coding_general",
|
|
"task_tier": 2,
|
|
"required_context_tokens": 123456,
|
|
"confidence": 0.91,
|
|
"classifier_ms": 1800,
|
|
"classification_source": "classifier",
|
|
"latency_tolerance": "interactive",
|
|
"candidates_considered": 7,
|
|
"selected_model": "fixture-model",
|
|
"selected_provider": "neuralwatt",
|
|
"runner_up_models": None,
|
|
"est_cost_usd": 0.001,
|
|
"est_proficiency": 0.9,
|
|
"rejected_reason": None,
|
|
"session_key": "sess-43",
|
|
"tools": 0,
|
|
"images": 0,
|
|
"json_mode": 0,
|
|
"streamed": 1,
|
|
"flex_preference": "auto",
|
|
"flex_swapped": 0,
|
|
"flex_forced": 0,
|
|
"request_id": "chatcmpl-fixture-43",
|
|
"exploration": 1,
|
|
"pinch_original_tokens": 120000,
|
|
"pinch_final_tokens": 96122,
|
|
"profile": "default",
|
|
}
|
|
text = DecisionDetailScreen(tui_model.decision_row(metrics_row))._render_text()
|
|
for key in (
|
|
"profile",
|
|
"exploration",
|
|
"pinch_original_tokens",
|
|
"pinch_final_tokens",
|
|
"request_id",
|
|
"session_key",
|
|
):
|
|
assert key in text
|
|
|
|
|
|
def test_app_run_test_ctrl_v_opens_verdict_popup():
|
|
"""Pressing ``ctrl+v`` pushes a VerdictMixScreen modal showing the verdict
|
|
mix rows from the current model, and ``escape`` dismisses it back to the
|
|
main screen."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
await pilot.press("ctrl+v")
|
|
await pilot.pause()
|
|
|
|
from textual.screen import ModalScreen
|
|
|
|
assert isinstance(app.screen, ModalScreen)
|
|
assert isinstance(app.screen, VerdictMixScreen)
|
|
# The popup carries the fixture's verdict mix.
|
|
by_verdict = {
|
|
row["verdict"]: row["count"] for row in app.screen.verdict_mix
|
|
}
|
|
assert by_verdict == {"ok": 5, "unverifiable": 2, "truncated": 1}
|
|
# The DataTable renders a row per verdict.
|
|
table = app.screen.query_one("#verdict-table")
|
|
assert table.row_count == 3
|
|
|
|
# ``escape`` dismisses back to the main dashboard screen.
|
|
await pilot.press("escape")
|
|
await pilot.pause()
|
|
assert not isinstance(app.screen, VerdictMixScreen)
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_verdict_mix_screen_updates_live_after_refresh():
|
|
"""An open VerdictMixScreen modal tracks the newest verdict mix after a
|
|
refresh, rather than staying a static snapshot from when it was opened."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
await pilot.press("ctrl+v")
|
|
await pilot.pause()
|
|
by_verdict = {
|
|
row["verdict"]: row["count"] for row in app.screen.verdict_mix
|
|
}
|
|
assert by_verdict == {"ok": 5, "unverifiable": 2, "truncated": 1}
|
|
|
|
# The next refresh delivers a new (smaller) verdict mix.
|
|
stub.payload["verdict_mix"] = {"ok": 100, "failed": 5}
|
|
app.action_refresh()
|
|
await pilot.pause()
|
|
|
|
by_verdict = {
|
|
row["verdict"]: row["count"] for row in app.screen.verdict_mix
|
|
}
|
|
assert by_verdict == {"ok": 100, "failed": 5}
|
|
table = app.screen.query_one("#verdict-table")
|
|
assert table.row_count == 2
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_live_decision_inserts_row_at_front_and_rerenders():
|
|
"""A decision delivered via the SSE callback is prepended to the model and
|
|
re-renders the decisions and category tables without a full re-fetch."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
before = len(app._last_model["recent_decisions"])
|
|
# Simulate the SSE consumer handing in a brand-new decision.
|
|
# Same (category, tier) as the fixture's first row so the bucket
|
|
# count for coding_general/tier-2 rises to 2.
|
|
app._handle_live_decision(
|
|
{
|
|
"id": 999,
|
|
"kind": "chat",
|
|
"task_category": "coding_general",
|
|
"task_tier": 2,
|
|
"selected_model": "deepseek-v4-flash",
|
|
"est_cost_usd": 0.0002,
|
|
}
|
|
)
|
|
await pilot.pause()
|
|
after = app._last_model["recent_decisions"]
|
|
assert len(after) == before + 1
|
|
assert after[0]["id"] == 999 # prepended, newest-first
|
|
# The table was re-rendered: the first row shows the new id.
|
|
dt = app.query_one("#decision-table")
|
|
first_row_text = " ".join(str(c) for c in dt.get_row_at(0))
|
|
assert "999" in first_row_text
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_live_decision_carries_new_fields():
|
|
"""A live SSE decision keeps the six T2 fields through the shared
|
|
decision_row projection, so a prompt `e` on a fresh decision shows the
|
|
same forensics as a /metrics-sourced row."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
app._handle_live_decision(
|
|
{
|
|
"id": 1001,
|
|
"kind": "chat",
|
|
"task_category": "coding_general",
|
|
"task_tier": 2,
|
|
"selected_model": "deepseek-v4-flash",
|
|
"est_cost_usd": 0.0002,
|
|
"profile": "default",
|
|
"exploration": 1,
|
|
"request_id": "chatcmpl-live-1001",
|
|
"pinch_original_tokens": 120000,
|
|
"pinch_final_tokens": 96122,
|
|
"session_key": "sess-live-1001",
|
|
}
|
|
)
|
|
await pilot.pause()
|
|
row = app._last_model["recent_decisions"][0]
|
|
assert row["id"] == 1001
|
|
assert row["profile"] == "default"
|
|
assert row["exploration"] == 1
|
|
assert row["request_id"] == "chatcmpl-live-1001"
|
|
assert row["pinch_original_tokens"] == 120000
|
|
assert row["pinch_final_tokens"] == 96122
|
|
assert row["session_key"] == "sess-live-1001"
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_live_decision_caps_recent_decisions_at_fifty():
|
|
"""The live feed never grows the in-memory list past the /metrics cap, so
|
|
the dashboard's view stays consistent with a /metrics refresh."""
|
|
stub = _StubFetcher()
|
|
# Start with exactly 50 rows so one live addition must evict the oldest.
|
|
base = _fixture()["recent_decisions"][0]
|
|
stub.payload = {"recent_decisions": [dict(base, id=i) for i in range(50, 0, -1)]}
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
assert len(app._last_model["recent_decisions"]) == 50
|
|
app._handle_live_decision(
|
|
{"id": 1, "kind": "chat", "task_category": "x", "task_tier": 1}
|
|
)
|
|
await pilot.pause()
|
|
assert len(app._last_model["recent_decisions"]) == 50
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_live_decision_dedup_skips_duplicate_id():
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
before = len(app._last_model["recent_decisions"])
|
|
app._handle_live_decision(
|
|
{
|
|
"id": 42,
|
|
"kind": "chat",
|
|
"task_category": "coding_general",
|
|
"task_tier": 2,
|
|
"selected_model": "deepseek-v4-flash",
|
|
}
|
|
)
|
|
await pilot.pause()
|
|
after = app._last_model["recent_decisions"]
|
|
assert len(after) == before, "duplicate id did not get skipped"
|
|
# first row is id=41 (42 was skipped as duplicate; 42 is now at pos 0)
|
|
assert after[0]["id"] == 42 and after[1]["id"] == 41, (
|
|
"first row is id 42 after dedup"
|
|
)
|
|
app._handle_live_decision(
|
|
{
|
|
"id": 888,
|
|
"kind": "route",
|
|
"task_category": "coding_general",
|
|
"task_tier": 2,
|
|
"selected_model": "kimi-k3",
|
|
}
|
|
)
|
|
await pilot.pause()
|
|
assert len(app._last_model["recent_decisions"]) == before + 1
|
|
assert app._last_model["recent_decisions"][0]["id"] == 888
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_live_decision_new_bucket_rebuilds_breakdown():
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
buckets_before = set(
|
|
(r["category"], r["tier"]) for r in app._last_model["category_breakdown"]
|
|
)
|
|
app._handle_live_decision(
|
|
{
|
|
"id": 1000,
|
|
"kind": "route",
|
|
"task_category": "summarization",
|
|
"task_tier": 1,
|
|
"selected_model": "qwen3.6-35b",
|
|
}
|
|
)
|
|
await pilot.pause()
|
|
buckets_after = set(
|
|
(r["category"], r["tier"]) for r in app._last_model["category_breakdown"]
|
|
)
|
|
assert ("summarization", 1) in buckets_after - buckets_before
|
|
row = next(
|
|
r
|
|
for r in app._last_model["category_breakdown"]
|
|
if r["category"] == "summarization" and r["tier"] == 1
|
|
)
|
|
assert row["count"] == 1
|
|
assert row["majority"] == "qwen3.6-35b"
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# DecisionStream: callback exceptions and stop behavior (tui_sse.py).
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
class _MockResp:
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *a):
|
|
pass
|
|
|
|
def raise_for_status(self):
|
|
pass
|
|
|
|
def iter_lines(self, decode_unicode=False):
|
|
yield "data: {\"id\": 1}"
|
|
raise requests.exceptions.ConnectionError("broken pipe")
|
|
|
|
|
|
def test_callback_raises_doesnt_break_reconnect(monkeypatch):
|
|
"""A callback that raises RuntimeError is caught; the stream still
|
|
survives and processes subsequent decisions after a reconnection."""
|
|
|
|
import tui_sse
|
|
|
|
monkeypatch.setattr(tui_sse.requests, "get", lambda *a, **kw: _MockResp())
|
|
|
|
call_count = {"n": 0}
|
|
|
|
def failing_callback(decision):
|
|
call_count["n"] += 1
|
|
if call_count["n"] == 1:
|
|
raise RuntimeError("app loop gone")
|
|
# second call succeeds — proves reconnect worked
|
|
|
|
s = tui_sse.DecisionStream(
|
|
"http://127.0.0.1",
|
|
failing_callback,
|
|
reconnect_seconds=0.1,
|
|
)
|
|
s.start()
|
|
time.sleep(0.6)
|
|
s.stop()
|
|
s.join(timeout=2)
|
|
assert not s.is_alive()
|
|
assert call_count["n"] >= 2, (
|
|
f"Expected reconnection after callback failure, got {call_count['n']} call(s)"
|
|
)
|
|
|
|
def test_live_decision_existing_bucket_defers_count_update():
|
|
"""A live decision in an existing bucket does NOT trigger an immediate
|
|
count rebuild (fixes finding #9 risk), but the count is updated on the
|
|
next authoritative /metrics poll.
|
|
"""
|
|
stub = _StubFetcher()
|
|
stub.payload = _fixture()
|
|
app = tui.DashboardApp(fetcher=stub, refresh_seconds=60)
|
|
|
|
async def _go():
|
|
async with app.run_test() as pilot:
|
|
await pilot.pause()
|
|
initial_count = next(
|
|
r["count"] for r in app._last_model["category_breakdown"]
|
|
if r["category"] == "coding_general" and r["tier"] == 2
|
|
)
|
|
assert initial_count == 1
|
|
app._handle_live_decision(
|
|
{"id": 2000, "kind": "chat", "task_category": "coding_general",
|
|
"task_tier": 2, "selected_model": "deepseek-v4-flash"}
|
|
)
|
|
await pilot.pause()
|
|
count_after_live = next(
|
|
r["count"] for r in app._last_model["category_breakdown"]
|
|
if r["category"] == "coding_general" and r["tier"] == 2
|
|
)
|
|
assert count_after_live == initial_count, "count grew immediately"
|
|
stub.payload["recent_decisions"].append(
|
|
{"id": 2000, "kind": "chat", "task_category": "coding_general",
|
|
"task_tier": 2, "selected_model": "deepseek-v4-flash"}
|
|
)
|
|
app._on_interval()
|
|
await pilot.pause()
|
|
final_count = next(
|
|
r["count"] for r in app._last_model["category_breakdown"]
|
|
if r["category"] == "coding_general" and r["tier"] == 2
|
|
)
|
|
assert final_count == 2
|
|
|
|
asyncio.run(_go())
|
|
|
|
|
|
def test_stopped_stream_exits_without_reconnect_sleep(monkeypatch):
|
|
"""After stop(), the thread should NOT wait for reconnect_seconds
|
|
before exiting — the _stopped guard is checked before the sleep."""
|
|
import tui_sse
|
|
|
|
class _MockResp:
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *a):
|
|
pass
|
|
|
|
def raise_for_status(self):
|
|
pass
|
|
|
|
def iter_lines(self, decode_unicode=False):
|
|
raise requests.exceptions.ConnectionError("closed")
|
|
|
|
monkeypatch.setattr(tui_sse.requests, "get", lambda *a, **kw: _MockResp())
|
|
|
|
stop_at = time.monotonic()
|
|
s = tui_sse.DecisionStream(
|
|
"http://127.0.0.1",
|
|
lambda x: None,
|
|
reconnect_seconds=5.0,
|
|
)
|
|
s.start()
|
|
# Let the first request attempt begin.
|
|
time.sleep(0.2)
|
|
# Record when stop is called.
|
|
s.stop()
|
|
stopped_at = time.monotonic()
|
|
# Thread should exit BEFORE the 5s reconnect backoff (use 3s as margin).
|
|
s.join(timeout=3)
|
|
assert not s.is_alive(), "Thread should exit promptly after stop()"
|
|
assert stopped_at - stop_at < 3.0, "Thread slept through reconnect_seconds"
|
|
|
|
|
|
def test_format_quota_legend_shows_next_reset_date():
|
|
"""When next_reset_date is available, _render appends the period line.
|
|
_format_quota_legend itself no longer takes next_reset_date; it is
|
|
appended in _render."""
|
|
app = tui.DashboardApp(fetcher=_StubFetcher(), refresh_seconds=60)
|
|
legend = app._format_quota_legend(
|
|
plan=6.25,
|
|
metered=1.25,
|
|
frac=0.2,
|
|
)
|
|
assert "6.25" in legend
|
|
assert "1.25" in legend
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# T3 — decision table display: time column, profile column, exploration flag.
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
def test_decision_table_columns_and_order():
|
|
"""The table declares exactly ten columns with `time` first.
|
|
|
|
Time leads because it is the natural scan axis for a live feed; `flex`
|
|
became `flags` because exploration now shares that cell.
|
|
"""
|
|
stub = _StubFetcher()
|
|
stub.payload = _enriched_payload()
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
dt = a.query_one("#decision-table")
|
|
labels = [str(c.label) for c in dt.columns.values()]
|
|
assert labels == [
|
|
"time",
|
|
"id",
|
|
"kind",
|
|
"category",
|
|
"tier",
|
|
"ctx",
|
|
"profile",
|
|
"selected",
|
|
"est $",
|
|
"flags",
|
|
]
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_short_time_renders_hhmmss_and_degrades_quietly():
|
|
"""`_short_time` is HH:MM:SS local, and never raises on bad input.
|
|
|
|
A malformed timestamp must not take down the whole table render, so the
|
|
unparseable cases return "" rather than propagating ValueError.
|
|
"""
|
|
rendered = tui._short_time("2026-08-23T10:00:00+00:00")
|
|
assert len(rendered) == 8 and rendered.count(":") == 2
|
|
# Local conversion, computed the same way the helper does it.
|
|
expected = (
|
|
datetime.fromisoformat("2026-08-23T10:00:00+00:00")
|
|
.astimezone()
|
|
.strftime("%H:%M:%S")
|
|
)
|
|
assert rendered == expected
|
|
# No date component leaks into the cell.
|
|
assert "2026" not in rendered
|
|
for bad in (None, "", "not-a-timestamp", 12345):
|
|
assert tui._short_time(bad) == ""
|
|
|
|
|
|
def test_app_decision_table_profile_column_blank_not_none():
|
|
"""`profile` renders its value, and renders BLANK when absent.
|
|
|
|
A literal "None" in a dense table reads as a real profile name.
|
|
"""
|
|
stub = _StubFetcher()
|
|
data = _enriched_payload()
|
|
data["recent_decisions"][1]["profile"] = None
|
|
stub.payload = data
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
dt = a.query_one("#decision-table")
|
|
labels = [str(c.label) for c in dt.columns.values()]
|
|
profile_at = labels.index("profile")
|
|
assert str(dt.get_row_at(0)[profile_at]) == "default"
|
|
# Absent profile is an empty cell, never the literal "None".
|
|
assert str(dt.get_row_at(1)[profile_at]) == ""
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_app_decision_table_ctx_column_blank_not_none():
|
|
"""`ctx` renders its value, and renders BLANK when absent.
|
|
|
|
A literal "None" in a dense table reads as a real number.
|
|
"""
|
|
stub = _StubFetcher()
|
|
data = _enriched_payload()
|
|
data["recent_decisions"][0]["required_context_tokens"] = 50000
|
|
data["recent_decisions"][1]["required_context_tokens"] = None
|
|
stub.payload = data
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
dt = a.query_one("#decision-table")
|
|
labels = [str(c.label) for c in dt.columns.values()]
|
|
ctx_at = labels.index("ctx")
|
|
assert str(dt.get_row_at(0)[ctx_at]) == "50000"
|
|
# Absent ctx is an empty cell, never the literal "None".
|
|
assert str(dt.get_row_at(1)[ctx_at]) == ""
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_exploration_flag_and_flags_indicator_compose():
|
|
"""`E` marks an exploratory pick and composes with the flex label.
|
|
|
|
An epsilon-greedy pick is a random sample, not the ranking's judgement;
|
|
without the marker an operator reads it as the router's answer.
|
|
"""
|
|
assert tui._exploration_flag({"exploration": 1}) == "E"
|
|
assert tui._exploration_flag({"exploration": 0}) == ""
|
|
assert tui._exploration_flag({}) == ""
|
|
# Composition: flex label first, then the marker, space separated.
|
|
both = tui._flags_indicator(
|
|
{"flex_preference": "auto", "flex_swapped": 0, "exploration": 1}
|
|
)
|
|
assert both == "auto E"
|
|
# Either alone survives; neither yields an empty cell.
|
|
assert tui._flags_indicator({"exploration": 1}) == "E"
|
|
assert tui._flags_indicator({"flex_preference": "auto"}) == "auto"
|
|
assert tui._flags_indicator({}) == ""
|
|
|
|
|
|
def test_app_decision_table_exploration_flag_renders():
|
|
"""Row 41 is the exploratory pick in the fixture; its flags cell shows E."""
|
|
stub = _StubFetcher()
|
|
stub.payload = _enriched_payload()
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
dt = a.query_one("#decision-table")
|
|
exploratory = [str(c) for c in dt.get_row_at(1)]
|
|
assert any("E" == cell or cell.endswith(" E") for cell in exploratory)
|
|
non_exploratory = [str(c) for c in dt.get_row_at(0)]
|
|
assert not any(cell == "E" or cell.endswith(" E") for cell in non_exploratory)
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_placeholder_row_matches_column_arity():
|
|
"""The empty-state placeholder must have exactly ten cells.
|
|
|
|
A short placeholder row raises inside Textual at render time — an arity
|
|
mismatch is a crash, not a cosmetic issue.
|
|
"""
|
|
stub = _StubFetcher()
|
|
data = _enriched_payload()
|
|
data["recent_decisions"] = []
|
|
stub.payload = data
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
dt = a.query_one("#decision-table")
|
|
assert len(dt.get_row_at(0)) == len(dt.columns) == 10
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# T4 — quota panel: kWh-only lead, labeled per-provider account rows.
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
def test_format_small_balance_renders_tiny_and_normal_values():
|
|
assert tui._format_small_balance(None) == "n/a"
|
|
assert tui._format_small_balance(0.0) == "$0.00"
|
|
assert tui._format_small_balance(0.0001) == "<$0.01"
|
|
assert tui._format_small_balance(-0.004) == "~$0.00 (slight overage)"
|
|
assert tui._format_small_balance(8.50) == "$8.50"
|
|
assert tui._format_small_balance(-1.0) == "$-1.00"
|
|
|
|
|
|
def test_format_provider_quota_account_labels_billing_shape():
|
|
"""_format_provider_quota_account renders prepaid_credit and metered_plan."""
|
|
prepaid = {
|
|
"provider": "openrouter",
|
|
"shape": "prepaid_credit",
|
|
"spend_usd": {"period": 2.0, "attribution_coverage": 1.0},
|
|
"plan": None,
|
|
"pool": {"total_credits_usd": 5.0, "total_usage_usd": 0.001, "balance_usd": 4.999, "age_seconds": 60},
|
|
"burn": {"burn_rate_usd_per_hour": 0.5, "projected_hours_remaining": 10.0},
|
|
"credit": {"balance_usd": 4.999, "balance_at": "2026-08-23T10:00:00+00:00", "balance_source": "polled"},
|
|
"energy": {"kwh_30d": 0.0, "calls_30d": 0},
|
|
}
|
|
row = tui._format_provider_quota_account(prepaid)
|
|
assert row.startswith("openrouter")
|
|
assert "burn $0.50/h" in row
|
|
assert "runway" in row
|
|
|
|
metered = {
|
|
"provider": "neuralwatt",
|
|
"shape": "metered_plan",
|
|
"spend_usd": {"period": 1.25, "attribution_coverage": 1.0},
|
|
"plan": {"kwh_per_period": 6.25, "used_kwh": 1.25, "used_fraction": 0.2},
|
|
"pool": None,
|
|
"burn": None,
|
|
"credit": {"balance_usd": -0.004, "balance_at": "2026-08-23T09:58:00+00:00", "balance_source": "telemetry"},
|
|
"energy": {"kwh_30d": 1.25, "calls_30d": 18},
|
|
}
|
|
row = tui._format_provider_quota_account(metered)
|
|
assert row.startswith("neuralwatt")
|
|
assert "kWh plan" in row
|
|
|
|
unmetered = {
|
|
"provider": "provider-a",
|
|
"shape": "unmetered",
|
|
"spend_usd": {"period": 0.0, "attribution_coverage": 1.0},
|
|
"plan": None, "pool": None, "burn": None, "credit": None, "energy": None,
|
|
}
|
|
row = tui._format_provider_quota_account(unmetered)
|
|
assert "provider-a" in row
|
|
assert "unmetered" in row
|
|
|
|
|
|
def test_provider_quota_rows_returns_one_row_per_account_sorted():
|
|
quota_raw = {
|
|
"accounts": [
|
|
{
|
|
"provider": "neuralwatt",
|
|
"shape": "metered_plan",
|
|
"spend_usd": {"period": 0.001, "attribution_coverage": 0.5},
|
|
"plan": {"kwh_per_period": 6.25, "used_kwh": 0.0, "used_fraction": 0.0},
|
|
"pool": None,
|
|
"burn": None,
|
|
"credit": {"balance_usd": 0.0001, "balance_at": None, "balance_source": "telemetry"},
|
|
"energy": {"kwh_30d": 0.0, "calls_30d": 0},
|
|
},
|
|
{
|
|
"provider": "openrouter",
|
|
"shape": "prepaid_credit",
|
|
"spend_usd": {"period": 0.5, "attribution_coverage": 1.0},
|
|
"plan": None,
|
|
"pool": {"total_credits_usd": 5.0, "total_usage_usd": 0.001, "balance_usd": 4.999, "age_seconds": 60},
|
|
"burn": None,
|
|
"credit": {"balance_usd": 4.999, "balance_at": None, "balance_source": "polled"},
|
|
"energy": {"kwh_30d": 0.0, "calls_30d": 0},
|
|
},
|
|
],
|
|
}
|
|
out = tui._provider_quota_rows(quota_raw)
|
|
assert len(out) == 2
|
|
assert out[0].startswith("neuralwatt")
|
|
assert out[1].startswith("openrouter")
|
|
|
|
|
|
def test_quota_panel_lead_is_kwh_only_and_note_empty():
|
|
"""The quota lead is the kWh-plan fraction; the note line stays hidden."""
|
|
stub = _StubFetcher()
|
|
data = _fixture()
|
|
stub.payload = data
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
lead_text = str(a.query_one("#quota-lead", Static).content)
|
|
assert "20% of 6.25 kWh plan" in lead_text
|
|
assert "None" not in lead_text
|
|
assert a.query_one("#quota-note", Static).display is False
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_quota_legend_includes_provider_account_rows():
|
|
"""The TUI quota legend lists one labeled row per provider after the kWh
|
|
summary line."""
|
|
stub = _StubFetcher()
|
|
data = _fixture()
|
|
stub.payload = data
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
legend = str(a.query_one("#quota-legend", Static).content)
|
|
assert "neuralwatt" in legend
|
|
assert "kWh plan" in legend
|
|
assert "None" not in legend
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_quota_legend_lists_polled_provider_with_credits_label():
|
|
"""A polled provider row shows the credits label, distinct from a
|
|
telemetry provider's overage label."""
|
|
stub = _StubFetcher()
|
|
data = _fixture()
|
|
data["quota"]["accounts"].append({
|
|
"provider": "openrouter",
|
|
"shape": "prepaid_credit",
|
|
"spend_usd": {"period": 0.5, "attribution_coverage": 1.0},
|
|
"plan": None,
|
|
"pool": {"total_credits_usd": 5.0, "total_usage_usd": 0.001, "balance_usd": 4.999, "age_seconds": 60},
|
|
"burn": None,
|
|
"credit": {"balance_usd": 4.999, "balance_at": "2026-08-23T10:00:00+00:00", "balance_source": "polled"},
|
|
"energy": {"kwh_30d": 0.0, "calls_30d": 0},
|
|
})
|
|
stub.payload = data
|
|
app = tui.DashboardApp(fetcher=stub)
|
|
|
|
def _assert(a):
|
|
legend = str(a.query_one("#quota-legend", Static).content)
|
|
assert "openrouter" in legend
|
|
|
|
_run_app(app, _assert)
|
|
|
|
|
|
def test_format_small_balance_handles_boundary_and_zero():
|
|
"""Exactly ±0.005 renders conventionally (not tiny); zero renders $0.00."""
|
|
assert tui._format_small_balance(0.005) == "$0.01"
|
|
assert tui._format_small_balance(-0.005) == "$-0.01"
|
|
assert tui._format_small_balance(0.0) == "$0.00"
|
|
|
|
|
|
def test_format_provider_quota_account_prepaid_credit():
|
|
"""prepaid_credit account renders pool balance and burn info."""
|
|
account = {
|
|
"provider": "openrouter",
|
|
"shape": "prepaid_credit",
|
|
"spend_usd": {"period": 0.5, "attribution_coverage": 1.0},
|
|
"plan": None,
|
|
"pool": {"total_credits_usd": 5.0, "total_usage_usd": 0.001, "balance_usd": 4.999, "age_seconds": 60},
|
|
"burn": None,
|
|
"credit": {"balance_usd": 4.999, "balance_at": "2026-08-23T10:00:00+00:00", "balance_source": "polled"},
|
|
"energy": {"kwh_30d": 0.0, "calls_30d": 0},
|
|
}
|
|
row = tui._format_provider_quota_account(account)
|
|
assert row.startswith("openrouter")
|
|
assert "pool" in row
|
|
|
|
|
|
def test_format_provider_quota_account_unmetered():
|
|
"""unmetered account shows minimal info."""
|
|
account = {
|
|
"provider": "provider-x",
|
|
"shape": "unmetered",
|
|
"spend_usd": {"period": 3.25, "attribution_coverage": 0.5},
|
|
"plan": None, "pool": None, "burn": None, "credit": None, "energy": None,
|
|
}
|
|
row = tui._format_provider_quota_account(account)
|
|
assert "provider-x" in row
|
|
assert "unmetered" in row
|
|
assert "$3.25" in row
|