`kind` is 'chat' on 99.79% of rows -- 25,337 of 25,390, with 52 'route' and a single 'dispatch' in the whole table -- so on both the dashboard's Recent Decisions and the Decisions page it spent a column's width rendering a constant, and a colored badge drawing the eye to it on every row. Removed from both tables. The dashboard drops it outright: it is the glanceable surface, and the freed width goes to the model cell, which was the column actually being truncated. The Decisions page keeps the information, because 0.2% of 25k rows is still 53 rows someone will want to find. It moves into the Flags cell, which already follows the right rule -- flagChips renders only what is ON, so nothing shows on a chat row and a chip appears on the exceptions. The Kind filter stays too; with the column gone it is now the way to isolate those rows. Caught in the browser, not in review: `.flag-chip` is an 18px square sized for one letter (T/I/J/S), so a word chip rendered with the pill behind the middle few characters and the rest spilling out -- 'passthrough' was unreadable. Added a `.flag-chip.kind` modifier whose width follows its text, in purple so it does not read as another single-letter flag. Verified on the 8081 sandbox against 23,846 real decisions: both tables render, the non-chat chips are legible, console is clean. Full suite 1895 passed. test_empty_states_span_every_column now DERIVES the expected count from the <thead> instead of hardcoding it. That literal has been stale twice already (colspan 11, then 13) and this change would have been its third: hand-updating a number in a test that exists to catch stale numbers fails the build without telling you which value is right. Also pins the intent of this commit, so the column cannot quietly come back and the chip cannot quietly go. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
170 lines
6.9 KiB
Python
170 lines
6.9 KiB
Python
"""The decisions table's model cell used to carry two different slashes.
|
|
|
|
``selected_model`` and ``selected_provider`` were joined into one cell as
|
|
``"<model> / <provider>"``. On a NeuralWatt row that reads fine. On an
|
|
OpenRouter row the model id is itself namespaced, so the cell rendered
|
|
|
|
xiaomi/mimo-v2.5 / openrouter
|
|
|
|
putting a vendor separator and a provider separator in one string, with
|
|
nothing to say they mean different things. Provider is its own column now.
|
|
|
|
``required_context_tokens`` was in the API payload all along and shown
|
|
nowhere, despite being the value the hard context filter compares against --
|
|
the one that produces a 422.
|
|
|
|
Static checks on the source, as with tests/test_models_page_structure.py --
|
|
this repo has no DOM harness. Behaviour was verified in a browser.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
SRC = (Path(__file__).resolve().parent.parent
|
|
/ "admin" / "frontend" / "decisions.html").read_text(encoding="utf-8")
|
|
|
|
|
|
def _function_body(name: str) -> str:
|
|
match = re.search(rf"^(?:async )?function {re.escape(name)}\(.*?^\}}",
|
|
SRC, re.S | re.M)
|
|
assert match, f"{name}() not found in decisions.html"
|
|
return match.group(0)
|
|
|
|
|
|
def _strip_comments(text: str) -> str:
|
|
text = re.sub(r"/\*.*?\*/", "", text, flags=re.S)
|
|
return "\n".join(ln for ln in text.split("\n")
|
|
if not ln.strip().startswith("//"))
|
|
|
|
|
|
@pytest.mark.parametrize("column", ["Provider", "Ctx"])
|
|
def test_the_new_columns_are_present(column):
|
|
assert f">{column}<" in SRC
|
|
|
|
|
|
def test_the_model_cell_no_longer_carries_the_provider():
|
|
"""Two slashes meaning different things in one cell, on every
|
|
OpenRouter row."""
|
|
body = _strip_comments(_function_body("renderTable"))
|
|
assert "' / '" not in body
|
|
assert "/ ${d.selected_provider" not in body
|
|
|
|
|
|
def test_model_and_provider_are_separate_cells():
|
|
body = _function_body("renderTable")
|
|
assert "d.selected_model" in body
|
|
assert "d.selected_provider" in body
|
|
# ...and they reach the row as two independent values, not one string.
|
|
assert "const model =" in body
|
|
assert "const provider =" in body
|
|
|
|
|
|
def test_null_and_zero_context_render_differently():
|
|
"""They are different facts. NULL means the column was never written --
|
|
it is nullable, and a NULL there already has one known latent
|
|
consequence (a TypeError in metrics.py's demand-ceiling comparison).
|
|
0 means the classifier declined to estimate: local_encoder mode always
|
|
returns required_context_tokens=0.
|
|
"""
|
|
body = _function_body("ctxCell")
|
|
assert "null" in body and "undefined" in body
|
|
assert "=== 0" in body, "zero needs its own branch, not the NULL branch"
|
|
# Distinct output for each.
|
|
assert body.count("text-muted") >= 2
|
|
|
|
|
|
def test_the_ctx_cell_never_renders_a_literal_none():
|
|
"""The defect the TUI's ctx cell still has, and the one the profile cell
|
|
was written to avoid: String(null) is 'null', and Python's str(None) --
|
|
which is what put the literal 'None' in the TUI -- is 'None'."""
|
|
body = _function_body("ctxCell")
|
|
assert "String(n)" not in body.split("=== 0")[0], (
|
|
"an unguarded String(n) above the zero branch would stringify null"
|
|
)
|
|
assert "'None'" not in body
|
|
|
|
|
|
def test_context_is_humanized_but_the_exact_value_stays_reachable():
|
|
"""153970 as '154k' is scannable; the exact figure is what you need when
|
|
checking a 422 against a ceiling, so it stays in the title."""
|
|
body = _function_body("ctxCell")
|
|
assert "toFixed(1)}M" in body or "'M'" in body
|
|
assert "Math.round(v / 1000)" in body
|
|
render = _function_body("renderTable")
|
|
assert "toLocaleString() + ' tokens'" in render
|
|
|
|
|
|
def test_ctx_is_a_right_aligned_numeric_column():
|
|
"""A magnitude column read by scanning: 153k and 9k must line up on the
|
|
digit or the column buys nothing."""
|
|
assert '<th class="num">Ctx</th>' in SRC
|
|
assert "#dec-tbody td.num, thead th.num{text-align:right" in SRC
|
|
assert "font-variant-numeric:tabular-nums" in SRC
|
|
|
|
|
|
def test_a_long_model_id_cannot_wrap_the_row_to_two_lines():
|
|
"""`nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free` is 44 characters
|
|
and wrapped, doubling the height of exactly the rows worth scanning.
|
|
Truncated rather than wrapped because this is a scanning surface, and the
|
|
full id stays in the cell's title."""
|
|
assert 'class="model-cell"' in SRC
|
|
assert "#dec-tbody td.model-cell{max-width:" in SRC
|
|
assert "text-overflow:ellipsis" in SRC
|
|
body = _function_body("renderTable")
|
|
assert 'title="${escapeHtml(model)}"' in body, "truncation needs the title"
|
|
|
|
|
|
def _header_column_count() -> int:
|
|
"""How many <th> cells the decisions table declares.
|
|
|
|
Derived rather than hardcoded: this assertion has now been stale twice
|
|
(colspan 11, then 13) because the literal had to be hand-updated every
|
|
time a column came or went, and a stale literal fails the build without
|
|
saying which number is the correct one. Note '<thead>' itself contains
|
|
the substring '<th', so match the cell tag specifically.
|
|
"""
|
|
thead = re.search(r"<thead>.*?</thead>", SRC, re.S)
|
|
assert thead, "decisions table <thead> not found"
|
|
return len(re.findall(r"<th[ >]", thead.group(0)))
|
|
|
|
|
|
def test_empty_states_span_every_column():
|
|
"""A stale colspan boxes the empty-state message into part of the row,
|
|
which reads as a broken table rather than an empty one."""
|
|
expected = _header_column_count()
|
|
colspans = {int(n) for n in re.findall(r'colspan="(\d+)"', SRC)}
|
|
assert colspans, "no colspan found — the empty states lost their span"
|
|
assert colspans == {expected}, (
|
|
f"empty-state colspan(s) {sorted(colspans)} do not match the "
|
|
f"{expected} columns the header declares"
|
|
)
|
|
assert len(re.findall(r'colspan="\d+"', SRC)) >= 2, (
|
|
"both the loading and the no-rows empty state need a colspan"
|
|
)
|
|
|
|
|
|
def test_kind_is_a_chip_on_the_exceptions_not_a_column():
|
|
"""`kind` was 'chat' on 99.8% of rows (25,337 of 25,390 on 2026-09-11), so
|
|
it spent a column's width on a constant. It is a flag chip on the
|
|
remaining 0.2% instead, and the Kind FILTER stays — that filter is the
|
|
only way to isolate those rows now."""
|
|
assert ">Kind<" not in SRC, "the Kind column is back"
|
|
assert "badgeForKind" not in SRC, "dead kind-badge helper left behind"
|
|
chips = _function_body("flagChips")
|
|
assert "d.kind !== 'chat'" in chips, "non-chat rows are no longer marked"
|
|
# A word does not fit the 18px single-letter chip; it needs the modifier.
|
|
assert "flag-chip on kind" in chips
|
|
assert ".flag-chip.kind{" in SRC
|
|
assert 'id="f-kind"' in SRC, "the Kind filter was removed with the column"
|
|
|
|
|
|
def test_search_still_matches_both_model_and_provider():
|
|
"""Splitting the cell must not narrow what the filter searches."""
|
|
body = _function_body("matchesFilters")
|
|
assert "d.selected_model" in body
|
|
assert "d.selected_provider" in body
|