"""Tripwire: every warning class /metrics can emit must reach the panel. The 2026-09-04 vision-ceiling incident was invisible for ~19 hours because a whole warning family was computed correctly and never displayed. The fix there was surfacing, not computing — so the guard has to be a test that fails when a warning class exists but nothing renders it. Two directions, both load-bearing: * every REGISTERED class actually fires against a fixture that provokes all of them — catches the fixture rotting or a message shape changing underneath; * every EMITTED warning is registered — catches a NEW class being added to ``metrics.scoring_coverage`` without anyone deciding to surface it. This one fails with the raw warning text, because the whole point is that nobody knew the class existed. Deliberately no assertions on warning COUNT: the numbers shift as fixture semantics evolve, and item 4 of the spec asks for class coverage, not arity. """ from __future__ import annotations import asyncio import copy import re import sqlite3 from datetime import datetime, timedelta, timezone from pathlib import Path from types import SimpleNamespace import pytest from textual.widgets import Static import metrics import tui ROOT = Path(__file__).resolve().parent.parent SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text() # The registry. A class here without a matching emitted warning means the # fixture rotted; an emitted warning without a class here means someone added # a warning nobody decided to display. WARN_CLASS_MATCHERS = { # Was r"metered usage is \d+% of the" -- a bare percentage, emitted on its # own >80%-of-plan threshold while the quota chip alarmed on pace, so the # bell and the chip could contradict each other. Both now read # quota_accounts' alarm. # # ONE class, matching the phrase both variants share, because the two # cannot co-occur: a deployment either has a billing_reset_day (pace) or # does not (absolute usage over a rolling 30d). Registering them # separately would make one of them permanently unfireable in this # fixture, which is precisely the rot this registry exists to catch. "quota-plan": r"of the [\d.]+ kWh plan", "models-missing-energy": r"\d+/\d+ routable models have no reference-workload", "models-missing-proficiency": r"\d+/\d+ routable models have no proficiency", "catalog-stale": r"catalog last polled", # Fixed-width lookbehind so this cannot swallow the capability warnings, # which share the "tier N context ceiling" tail. "demand-ceiling": r"(? None: """Seed a catalog and traffic that provoke ALL ten warning classes at once. Named for what it is: a deliberately maximally-broken deployment. Every timestamp is relative to now so the rejection windows (1h alert, 24h baseline) stay valid whenever the suite runs. """ now = datetime.now(timezone.utc) stale = (now - timedelta(days=3)).isoformat() rows = [ # Covered model: has proficiency AND energy, so the "missing" counts # come out as 2/3 rather than 3/3. _model_row("cov-t1", 1, 20000), # Uncovered: drives both missing- classes; its 5000 window is also the # tier-2 ceiling behind the escalation hazard. _model_row("gap-t2", 2, 5000), # Vision-capable but small: the capability sub-ceiling. _model_row("vis-t1", 1, 5000, vision=1), # Deprecated: excluded from routable, so tier 3 has zero eligible. _model_row("dead-t3", 3, 1000, availability="deprecated"), # Zero effective context: active, never rejected, never selectable. # Tier 1 on purpose -- a tier-3 row here would refill the "tier 3 has # 0 eligible models" class and silence it. Contributes 0 to every # ceiling max(), which is exactly why no other detector sees it. _model_row("zero-ctx-t1", 1, 0), ] for row in rows: conn.execute( """ INSERT INTO models ( model_id, provider, base_model_id, tier, context_window, effective_context_window, max_output_tokens, cost_per_1m_prompt, cost_per_1m_completion, supports_vision, supports_json_mode, latency_class, reasoning_mode, context_variant, access_level, availability, last_updated ) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?) """, (*row, stale), ) conn.execute( "INSERT INTO proficiency " "(model_id, provider, category, blended_score, last_updated) " "VALUES ('cov-t1','neuralwatt','coding_general',0.9,?)", (now.isoformat(),), ) # 6.0 kWh against a 6.25 plan = 96% > 80%, firing the quota class. The # 'seed_reference' task_category is also what marks a model as having # reference-workload data, so this one row serves both purposes. conn.execute( "INSERT INTO energy_observations " "(model_id, provider, energy_kwh, task_category, observed_at) " "VALUES ('cov-t1','neuralwatt',6.0,'seed_reference',?)", ((now - timedelta(days=1)).isoformat(),), ) def _decision(**kw): cols = { "kind": "chat", "task_category": "coding_general", "task_tier": 1, "required_context_tokens": 100, "latency_tolerance": "interactive", "selected_model": "cov-t1", "selected_provider": "neuralwatt", "observed_at": now.isoformat(), "images": 0, "json_mode": 0, "rejected_reason": None, } cols.update(kw) conn.execute( f"INSERT INTO route_decisions ({','.join(cols)}) " f"VALUES ({','.join('?' * len(cols))})", tuple(cols.values()), ) # One huge vision request: demand-ceiling (20000 < 999999), capability # sub-ceiling (vision 5000 < 999999), and the escalation hazard all at once. _decision(required_context_tokens=999999, images=1) # Two rejections of a pattern with no baseline -> "new rejection pattern:". for _ in range(2): _decision( selected_model=None, selected_provider=None, rejected_reason="context >= 99999 tokens; interactive", ) # A familiar group: one in the baseline window, six in the alert window, # crossing the min-count threshold -> "rejection rate:". familiar = "tier >= 2; context >= 30000 tokens; interactive" _decision( task_tier=2, selected_model=None, selected_provider=None, rejected_reason=familiar, observed_at=(now - timedelta(hours=2)).isoformat(), ) for _ in range(6): _decision( task_tier=2, selected_model=None, selected_provider=None, rejected_reason=familiar, ) # 20 fallback-classified decisions: 100% degraded over the minimum count, # firing the classifier-degradation class. Two details keep them from # disturbing the classes seeded above. selected_model is set, so they do # not read as rejections. And their context MATCHES the demand row rather # than being small — 20 small rows drag the tier-1 p95 down far enough to # silence the escalation hazard, which is how this fixture first broke. for _ in range(20): _decision(classification_source="fallback", required_context_tokens=999999) # --- content-fault warning seeds --- # Novel pattern: "novel-cf" has no baseline rows, only alert-window rows. # Count >= NOVEL_GROUP_MIN_COUNT (2) fires "new content fault:". for i in range(3): conn.execute( "INSERT INTO verifications " "(model_id, provider, task_category, request_id, kind, verdict, detail, " " completion_tokens, model_attributable, observed_at) " "VALUES (?, 'neuralwatt', 'coding_general', ?, 'structural', " "'malformed', 'empty response', 0, 1, ?)", ( "novel-cf", f"req-novel-{i}", (now - timedelta(minutes=5 * i)).isoformat(), ), ) # Known pattern: "known-cf" has rows in BOTH alert and baseline windows. # Alert count >= min_count (6) fires "content fault:". # Baseline rows prove this is not novel. baseline_hours = 24 conn.execute( "INSERT INTO verifications " "(model_id, provider, task_category, request_id, kind, verdict, detail, " " completion_tokens, model_attributable, observed_at) " "VALUES (?, 'neuralwatt', 'coding_general', 'req-known-bl', " "'structural', 'malformed', 'empty response', 0, 1, ?)", ("known-cf", (now - timedelta(hours=baseline_hours)).isoformat()), ) for i in range(6): conn.execute( "INSERT INTO verifications " "(model_id, provider, task_category, request_id, kind, verdict, detail, " " completion_tokens, model_attributable, observed_at) " "VALUES (?, 'neuralwatt', 'coding_general', ?, 'structural', " "'malformed', 'empty response', 0, 1, ?)", ( "known-cf", f"req-known-{i}", (now - timedelta(minutes=5 * i)).isoformat(), ), ) # --- cache-rate warning seeds --- # 30 reported-cache observations on one (provider, model) at 0.400 against # an assumed 0.917: past the 25-observation floor, 0.517 outside the 0.10 # margin, so the aggregate class and the per-group outlier class both fire. # # Three properties keep this from disturbing the classes above, and all # three are load-bearing in a fixture whose seeds are known to silence each # other. energy_kwh is 0.0, so the 6.0-against-6.25 quota margin is # untouched. task_category is NOT 'seed_reference', so this does not count # as reference-workload coverage for 'cache-cold' (which is not in `models` # anyway) and is not filtered out of the cache query. And cost_usd stays # NULL, so no spend series acquires a flat-zero segment. for i in range(30): conn.execute( "INSERT INTO energy_observations " "(model_id, provider, task_category, prompt_tokens, " " completion_tokens, cached_prompt_tokens, cached_tokens_source, " " energy_kwh, observed_at) " "VALUES ('cache-cold','neuralwatt','coding_general',1000,100," "400,'reported',0.0,?)", ((now - timedelta(minutes=i)).isoformat(),), ) conn.commit() @pytest.fixture() def chernobyl(tmp_path): conn = sqlite3.connect(tmp_path / "warn.db") conn.executescript(SCHEMA_SQL) conn.row_factory = sqlite3.Row _seed_chernobyl(conn) yield conn conn.close() @pytest.fixture(autouse=True) def _no_real_network(monkeypatch): """Even if a fetcher is mis-wired, never reach a real router.""" def _guard(base_url): raise AssertionError(f"real fetch_metrics called with {base_url!r}") monkeypatch.setattr(tui, "fetch_metrics", _guard) def _run_app(app: tui.DashboardApp, body) -> None: async def _go(): async with app.run_test() as pilot: await pilot.pause() body(app) asyncio.run(_go()) def test_every_registered_warning_class_fires(chernobyl): """Each registered class is provoked by the fixture. A failure here means either the fixture rotted or a warning's message shape changed under the matcher — both silently disable the tripwire. """ warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] for name, pattern in WARN_CLASS_MATCHERS.items(): assert any(re.search(pattern, w) for w in warnings), ( f"registered warning class {name!r} did not fire.\n" f"Either the chernobyl fixture no longer provokes it, or the " f"message shape changed and the matcher needs updating.\n" f"Pattern: {pattern}\nEmitted:\n " + "\n ".join(warnings) ) def test_every_emitted_warning_belongs_to_a_registered_class(chernobyl): """No warning class exists without a decision about surfacing it. This is the direction that catches the vision-ceiling failure mode: a new class added to scoring_coverage that nothing displays. It fails with the RAW text because the whole point is that nobody knew it existed. """ warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] unmatched = [ w for w in warnings if not any(re.search(p, w) for p in WARN_CLASS_MATCHERS.values()) ] assert not unmatched, ( "a warning class was added to metrics.scoring_coverage without a " "surfacing decision. Register a matcher in WARN_CLASS_MATCHERS (and a " "chernobyl seed if it needs one).\nUNMATCHED:\n " + "\n ".join(unmatched) ) def test_registered_warnings_render_in_the_panel(chernobyl): """The panel actually shows them — computing is not surfacing. Drives the real app with the chernobyl warnings so the assertion covers the render path, not just the metrics call. """ warnings = metrics.scoring_coverage(chernobyl, CFG)["warnings"] from tests.test_tui import _fixture # noqa: PLC0415 — shared payload shape payload = copy.deepcopy(_fixture()) payload["coverage"]["warnings"] = warnings class _Stub: def __call__(self, base_url): return payload app = tui.DashboardApp(fetcher=_Stub()) def _assert(a): panel = str(a.query_one("#warnings-panel", Static).content) for name, pattern in WARN_CLASS_MATCHERS.items(): assert re.search(pattern, panel), ( f"warning class {name!r} is emitted by /metrics but does not " f"render in #warnings-panel — computed but not surfaced, which " f"is exactly the 2026-09-04 failure.\nPanel:\n{panel}" ) _run_app(app, _assert)