fix: NULL required_context_tokens handling in TUI ctx cell and metrics demand-ceiling path #96

Merged
alee merged 2 commits from fix/required-context-null-guards into main 2026-09-19 15:56:48 +00:00
4 changed files with 154 additions and 5 deletions

View File

@@ -1424,8 +1424,8 @@ def demand_ceiling_warnings(
" AND kind = 'chat' AND task_tier = ?",
(tier,),
).fetchone()
observed_max = row["mx"] if row else 0
if observed_max > ceiling:
observed_max = row["mx"] if row and row["mx"] is not None else None
if observed_max is not None and observed_max > ceiling:
warnings.append(
f"tier {tier} context ceiling ({ceiling}) is below observed max "
f"demand ({observed_max}) — requests above this may return 422"
@@ -1454,6 +1454,7 @@ def demand_ceiling_warnings(
" AND kind = 'chat' AND task_tier = ?",
(t,),
).fetchall()
if r["required_context_tokens"] is not None
]
p95 = _percentile(tokens, 95)
if p95 is not None and ceiling_next < p95:
@@ -1688,9 +1689,9 @@ def capability_demand_warnings(
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
(tier,),
).fetchone()
observed_max = row["mx"] if row else 0
observed_max = row["mx"] if row and row["mx"] is not None else None
ceiling = bucket["ceiling"]
if observed_max > ceiling:
if observed_max is not None and observed_max > ceiling:
warnings.append(
f"{dimension}-capable tier {tier} context ceiling "
f"({ceiling}) is below observed max demand "

View File

@@ -766,7 +766,7 @@ class DashboardApp(App):
str(r.get("kind")),
str(r.get("category")),
str(r.get("tier")),
str(r.get("required_context_tokens")),
str(r.get("required_context_tokens") or ""),
# `or ""` so an absent profile is blank, never the literal "None".
str(r.get("profile") or ""),
str(r.get("selected")),

View File

@@ -1041,6 +1041,131 @@ def test_capability_json_mode_demand_warning(tmp_path):
assert "200000" in w
# --- NULL required_context_tokens guards (regression) ---------------------------
#
# route_decisions.required_context_tokens is a nullable column (rows written
# before the column existed, or decisions persisted without a measured token
# count). SQLite MAX() over an all-NULL group returns NULL, not 0, so a
# window whose every row lacks a token count yields observed_max=None —
# which used to raise ``None > ceiling`` TypeError inside the demand-ceiling
# checks, and crashed _percentile on the escalation path. Semantics: NULL is
# "no demand recorded for this row", not zero — so a NULL row is excluded
# from comparisons, never fabricated into a warning.
def test_demand_ceiling_all_null_7d_window_no_crash_no_warning(tmp_path):
"""A tier whose 7d rows are all NULL does not crash demand_ceiling_warnings.
Site A regression: ``MAX(required_context_tokens)`` over an all-NULL
window returns NULL, so ``observed_max`` used to become None and the
``None > ceiling`` comparison raised TypeError. Seeding the tier's
demand proof in the 30d window (one non-NULL row at 20 days ago) plus
12 NULL rows inside 7d puts the tier in ``tiers_with_demand`` while
keeping the observed-max window at 7d — exactly the shape that used to
crash. With the guard, the comparison is skipped: no crash, and no
false "0 > ceiling" warning either.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
old = (_now() - timedelta(days=20)).isoformat()
for _ in range(12):
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens",
observed_at=ts,
tokens=None,
)
# Non-NULL demand inside 30d (outside 7d) proves the tier has demand,
# while every row inside the 7d observed-max window stays NULL.
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens",
observed_at=old,
tokens=200000,
)
ctx = context_ceilings(conn, CFG)
assert ctx[(2, "interactive")]["ceiling"] > 0
# Old code: None > ceiling -> TypeError. New code: comparison skipped.
assert demand_ceiling_warnings(conn, CFG, ctx, [1, 2]) == []
def test_demand_ceiling_escalation_percentile_skips_null_tokens(tmp_path):
"""NULL tokens in the escalation p95 query are excluded, not sorted.
Site B regression: the escalation-hazard probe fetched every tier's 7d
``required_context_tokens`` raw and fed the list to _percentile, whose
``sorted()`` raises TypeError the moment a None meets an int. A mixed
NULL/non-NULL 7d window for tier 1 used to crash; after the fix the
NULL row is filtered out and the percentile is computed over the
remaining values. 100000 < the tier-2 ceiling, so no escalation
warning fires either — the assertion proves both.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
_insert_rejected_decision(
conn, tier=1, reason="context >= 999999 tokens", observed_at=ts, tokens=None
)
_insert_rejected_decision(
conn, tier=1, reason="context >= 999999 tokens", observed_at=ts, tokens=100000
)
ctx = context_ceilings(conn, CFG)
# Escalation enabled in the default CFG; tier-2 bucket must exist and
# carry a non-zero ceiling or the percentile probe would be skipped.
assert getattr(CFG.escalation, "enabled", False)
assert ctx[(2, "interactive")]["ceiling"] > 0
# Old code: _percentile([None, 100000], 95) -> TypeError on sorted().
# New code: None filtered out, p95 = 100000 < ceiling, no warning.
assert demand_ceiling_warnings(conn, CFG, ctx, [1, 2]) == []
def test_capability_demand_all_null_7d_window_no_crash_no_warning(tmp_path):
"""A capability-gated tier whose 7d rows are all NULL does not crash.
Site C regression: same all-NULL MAX shape as the tier-wide demand
check, but inside capability_demand_warnings' vision dimension. One
vision-carrying row with a non-NULL token count at 20 days ago puts
the tier's vision bucket past the unexercised-silence guard, and 12
NULL-token vision rows inside 7d keep the observed-max window at 7d —
the exact shape that used to raise ``None > ceiling`` TypeError. With
the guard, no crash and no fabricated warning.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
old = (_now() - timedelta(days=20)).isoformat()
for _ in range(12):
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens; vision-capable model",
observed_at=ts,
tokens=None,
images=1,
)
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens; vision-capable model",
observed_at=old,
tokens=200000,
images=1,
)
cap_ctx = capability_ceilings(conn, CFG)
assert cap_ctx["vision"][(2, "interactive")]["ceiling"] > 0
# Old code: None > ceiling -> TypeError. New code: comparison skipped.
assert capability_demand_warnings(conn, CFG, cap_ctx) == []
# --- rejection warnings (reactive rejection detector) --------------------------

View File

@@ -1708,6 +1708,29 @@ def test_app_decision_table_profile_column_blank_not_none():
_run_app(app, _assert)
def test_app_decision_table_ctx_column_blank_not_none():
"""`ctx` renders its value, and renders BLANK when absent.
A literal "None" in a dense table reads as a real number.
"""
stub = _StubFetcher()
data = _enriched_payload()
data["recent_decisions"][0]["required_context_tokens"] = 50000
data["recent_decisions"][1]["required_context_tokens"] = None
stub.payload = data
app = tui.DashboardApp(fetcher=stub)
def _assert(a):
dt = a.query_one("#decision-table")
labels = [str(c.label) for c in dt.columns.values()]
ctx_at = labels.index("ctx")
assert str(dt.get_row_at(0)[ctx_at]) == "50000"
# Absent ctx is an empty cell, never the literal "None".
assert str(dt.get_row_at(1)[ctx_at]) == ""
_run_app(app, _assert)
def test_exploration_flag_and_flags_indicator_compose():
"""`E` marks an exploratory pick and composes with the flex label.