feat: capability-aware ceiling warnings + reactive rejection detector #31

Merged
alee merged 2 commits from feat/capability-aware-ceiling-warnings into main 2026-09-05 05:42:54 +00:00
5 changed files with 893 additions and 5 deletions

View File

@@ -84,6 +84,26 @@ objective:
# 29-31 to avoid shorter-month edge cases).
billing_reset_day: 6
# Rejection-rate detection over route_decisions rows that selected no model
# (422 "no model satisfies the hard filters"). The signal is novelty or rate —
# NEVER mere presence: this deployment routinely has ~3 rejections/hour of
# ordinary over-large tier-3 requests that are behaving as designed (6 in the
# last 24h, 17 all-time at 2026-09-05), so a count-only tripwire would be
# permanently on.
#
# alert window (hours): rejections this recent are counted per
# (task_tier, digit-normalized reason) group.
rejection_warning_window_hours: 1
# baseline window (hours) BEFORE the alert window: groups absent here are
# "new". A new group warns from 2 occurrences — the 2026-09-04 vision
# incident produced exactly 2 and no rate threshold can sit below routine
# noise yet above that.
rejection_warning_baseline_hours: 24
# a group already present in the baseline warns only at this many
# occurrences inside the alert window: 6 = 2x the observed routine hourly
# peak, far below the dozens/hour a deprecation flare produces.
rejection_warning_min_count: 6
context:
safety_factor: 0.75 # fraction of advertised context treated as usable
default_output_reserve_tokens: 4096

View File

@@ -54,6 +54,9 @@ class Objective(StrictModel):
quota_runway_warning_hours: Optional[int] = None
quota_burn_min_segment_samples: Optional[int] = None
quota_burn_min_segment_hours: Optional[float] = None
rejection_warning_window_hours: Optional[int] = None
rejection_warning_baseline_hours: Optional[int] = None
rejection_warning_min_count: Optional[int] = None
@field_validator("quality_tolerance")
@classmethod
@@ -115,6 +118,33 @@ class Objective(StrictModel):
)
return v
@field_validator("rejection_warning_window_hours")
@classmethod
def rejection_window_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.rejection_warning_window_hours must be > 0"
)
return v
@field_validator("rejection_warning_baseline_hours")
@classmethod
def rejection_baseline_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.rejection_warning_baseline_hours must be > 0"
)
return v
@field_validator("rejection_warning_min_count")
@classmethod
def rejection_min_count_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.rejection_warning_min_count must be > 0"
)
return v
class ContextOverride(StrictModel):
"""Per-model context handling, for a row whose real limits are known.

View File

@@ -14,6 +14,9 @@ quota_burn — kWh metered in the last 30 d and in the current billing
credit balance, burn rate and runway from
allowance_remaining_usd
scoring_coverage — which scoring axes actually have data
capability_ceilings — vision / json_mode context-window sub-ceilings
capability_demand_warnings — demand-relative warnings for those sub-ceilings
rejection_warnings — grouped rejection-rate / new-pattern warnings
recent_decisions — last N rows from the route_decisions observability table
per_model — per-model aggregates over energy_observations (last 30 d)
verdict_mix — counts by verdict from verifications (last N days)
@@ -23,9 +26,10 @@ pinch_summary — context-pruning savings over the last 30 d
from __future__ import annotations
import re
import sqlite3
from datetime import date, datetime, timedelta, timezone
from typing import Any, List, Optional
from typing import Any, Final, List, Optional
import routing
@@ -406,6 +410,10 @@ def scoring_coverage(
f"tier {tier} has 0 eligible models for latency_tolerance={lat_tol}"
)
cap_ctx = capability_ceilings(conn, cfg)
warnings.extend(capability_demand_warnings(conn, cfg, cap_ctx))
warnings.extend(rejection_warnings(conn, cfg))
return {
"routable_models": total,
"with_energy_data": total - len(missing_energy),
@@ -568,6 +576,8 @@ def context_ceilings_with_rows(
cfg: Any,
rows: List[dict],
exclude_models: Optional[set[str]] = None,
require_vision: bool = False,
require_json_mode: bool = False,
) -> dict:
"""Like ``context_ceilings`` but over caller-supplied model rows.
@@ -575,6 +585,9 @@ def context_ceilings_with_rows(
proficiency and because the default ``exclude_models`` set is read from
``admin_model_overrides``. Callers that already know the exclusion set
(e.g. admin.py applying a just-committed override) may pass it explicitly.
``require_vision`` / ``require_json_mode`` gate the candidate set the
same way live routing gates it, so a capability-gated sub-ceiling goes
through this one path rather than a second copy of the filters.
"""
if exclude_models is None:
exclude_models = _admin_deprecated_models(conn)
@@ -600,8 +613,8 @@ def context_ceilings_with_rows(
exclude_deprecated=True,
exclude_models=exclude_models,
min_tool_proficiency=None,
require_vision=False,
require_json_mode=False,
require_vision=require_vision,
require_json_mode=require_json_mode,
)
ceiling = max((r["effective_context_window"] for r in candidates), default=0)
result[(tier, latency_tolerance)] = {
@@ -611,6 +624,255 @@ def context_ceilings_with_rows(
return result
def capability_ceilings(
conn: sqlite3.Connection,
cfg: Any,
) -> dict:
"""Context-window ceiling per capability-gated subset, by tier.
The plain ``(tier, latency_tolerance)`` buckets from
``context_ceilings`` are blind to capability gates: during the
2026-09-04 incident the tier-1 interactive ceiling stayed at 782,324
while the vision-capable subset had collapsed to 192,500 (the only
vision rows large enough had been deprecated through admin overrides),
so every existing demand check was silent on requests that 422ed.
Vision and JSON mode are the two flags that can independently empty
the candidate set — both fail CLOSED on an unknown catalog flag, which
is what lets them shrink the set sharply — so each gets its own series
here. Deliberately NOT one bucket per capability combination; that is
a combinatorial explosion over dimensions that do not interact.
Delegates to ``context_ceilings_with_rows`` once per dimension, so every
sub-ceiling comes from the same ``routing.select_candidates`` path
live routing uses, admin deprecations included.
Returns
``{"vision": {(tier, latency_tolerance): {"ceiling", "count"}}, "json_mode":
{...}}``, one sub-dict per capability dimension, each keyed by the
``(tier, latency_tolerance)`` pair present in the DB.
"""
rows = [dict(r) for r in conn.execute("SELECT * FROM models")]
exclude_models = _admin_deprecated_models(conn)
result: dict[str, dict[tuple, dict[str, int]]] = {}
for dimension in ("vision", "json_mode"):
sub_ceilings = context_ceilings_with_rows(
conn,
cfg,
rows,
exclude_models=exclude_models,
require_vision=dimension == "vision",
require_json_mode=dimension == "json_mode",
)
result[dimension] = sub_ceilings
return result
def capability_demand_warnings(
conn: sqlite3.Connection,
cfg: Any,
cap_ctx: dict,
) -> List[str]:
"""Demand-relative warnings for the capability-gated sub-ceilings.
Mirrors ``demand_ceiling_warnings`` for each dimension — 7d MAX probe
first with a 30d fallback, silence for a bucket no such request ever
hit, else window = 7 when the 7d row count is >= 10 (30 otherwise),
warning only when that window's observed max exceeds the sub-ceiling.
The one difference is what "demand" means: request rows must carry the
capability (``images`` / ``json_mode`` columns on route_decisions), so
a vision sub-ceiling is compared only against requests that actually
carry images, for the ``(tier, "interactive")`` bucket. ``cfg`` is
accepted for signature symmetry with ``demand_ceiling_warnings``; this
detector has no escalation counterpart to read from it.
The demand read is trailing-window, so a warning stays alive until a
big historical request ages out of it; the ``window: last {n}d``
annotation in the message is what keeps that self-explaining.
"""
warnings: List[str] = []
dimension_columns = {
"vision": "images",
"json_mode": "json_mode",
}
capability_nouns = {
"vision": "for requests carrying images",
"json_mode": "for requests requesting JSON mode",
}
for dimension in ("vision", "json_mode"):
buckets = cap_ctx.get(dimension) or {}
for tier in sorted({t for (t, _) in buckets}):
bucket = buckets.get((tier, "interactive"))
if bucket is None:
continue
column = dimension_columns.get(dimension)
if column is None:
continue
row = conn.execute(
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
" WHERE julianday(observed_at) > julianday('now', '-7 days')"
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
(tier,),
).fetchone()
mx_7d = row["mx"] if row else None
if mx_7d is None:
row = conn.execute(
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
" WHERE julianday(observed_at) > julianday('now', '-30 days')"
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
(tier,),
).fetchone()
mx_7d = row["mx"] if row else None
if mx_7d is None:
# No request carrying this capability was ever routed for this
# tier — an unexercised sub-ceiling warns about nothing.
continue
row_cnt = conn.execute(
"SELECT COUNT(*) AS n FROM route_decisions"
" WHERE julianday(observed_at) > julianday('now', '-7 days')"
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
(tier,),
).fetchone()["n"]
window = 7 if row_cnt >= 10 else 30
row = conn.execute(
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
f" WHERE julianday(observed_at) > julianday('now', '-{window} days')"
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
(tier,),
).fetchone()
observed_max = row["mx"] if row else 0
ceiling = bucket["ceiling"]
if observed_max > ceiling:
warnings.append(
f"{dimension}-capable tier {tier} context ceiling "
f"({ceiling}) is below observed max demand "
f"({observed_max}, window: last {window}d) "
f"{capability_nouns[dimension]} — requests above this may "
"return 422"
)
return warnings
# A novel (tier, normalized_reason) group must reach this count inside the
# alert window before warning. The 2026-09-04 incident produced exactly 2
# image rejections, so the floor must be <= 2; 1 would fire on any one-off
# genuinely-impossible request that correctly 422s.
NOVEL_GROUP_MIN_COUNT: Final = 2
# How many rejections a group needs inside the alert window before
# "new rejection pattern:" is reported. A single rejection is
# indistinguishable from a genuinely impossible request, which should 422 —
# the signal is a rate, not the existence of a rejection.
#
# Caveat: "novel" means ZERO occurrences in the prior baseline window, so
# a familiar-but-intermittent pattern can read as novel after a quiet gap
# and fire one spurious "new rejection pattern:" warning. It is
# self-clearing as the pattern's own rejections age into the baseline
# window; widen objective.rejection_warning_baseline_hours to suppress it.
def rejection_warnings(
conn: sqlite3.Connection,
cfg: Any,
) -> List[str]:
"""Reactive safety net over the rejections the router already logged.
The ceiling checks above only catch dimensions someone modelled; this
one catches everything, at the cost of firing after the first failures
rather than before. Rejections (``selected_model IS NULL AND
rejected_reason IS NOT NULL``) are pulled once across the alert plus
baseline window and split by exact age — alert bucket at
``rejection_warning_window_hours`` or younger, baseline bucket older
than the alert window but within ``rejection_warning_baseline_hours``
of it — then grouped by *normalized* reason plus ``task_tier``.
Two signals, one line per group:
* ``rejection rate:`` — the group reached
``objective.rejection_warning_min_count`` rejections in the alert
window.
* ``new rejection pattern:`` — the group is ABSENT from the baseline
window and reached ``NOVEL_GROUP_MIN_COUNT`` in the alert window.
This prefix wins when both would fire.
Zero alert-window rejections produce nothing at all: a rejection is
not inherently an error, and a genuinely impossible request should
422 — the signal is a rate.
"""
window = getattr(cfg.objective, "rejection_warning_window_hours", None)
if window is None:
window = 1
baseline = getattr(cfg.objective, "rejection_warning_baseline_hours", None)
if baseline is None:
baseline = 24
min_count = getattr(cfg.objective, "rejection_warning_min_count", None)
if min_count is None:
min_count = 6
now = datetime.now(timezone.utc)
rows = conn.execute(
"""
SELECT observed_at, task_tier, rejected_reason
FROM route_decisions
WHERE selected_model IS NULL
AND rejected_reason IS NOT NULL
AND julianday(observed_at) > julianday('now', '-' || ? || ' hours')
ORDER BY observed_at DESC
""",
(str(window + baseline),),
).fetchall()
alert_groups: dict[tuple[Optional[int], str], dict] = {}
baseline_keys: set[tuple[Optional[int], str]] = set()
for row in rows:
try:
observed = datetime.fromisoformat(row["observed_at"])
except ValueError:
# Defensive, same shape as quota_balance_and_burn: one malformed
# timestamp must not take the whole report down.
continue
# Group on digit-normalized reasons: the raw strings embed the
# request's own numbers ("context >= 242486 tokens"), so literal
# grouping yields N groups of one and hides the pattern entirely.
key = (row["task_tier"], re.sub(r"\d+", "N", row["rejected_reason"]))
age_hours = (now - observed).total_seconds() / 3600.0
if age_hours <= window:
group = alert_groups.get(key)
if group is None:
# Rows arrive newest-first, so the first touch per group is
# its most recent occurrence.
alert_groups[key] = {"count": 1, "most_recent": observed}
else:
group["count"] += 1
elif age_hours <= window + baseline:
baseline_keys.add(key)
if not alert_groups:
return []
warnings: List[str] = []
for (tier, normalized_reason), group in alert_groups.items():
count = group["count"]
is_novel = (tier, normalized_reason) not in baseline_keys
prefix: Optional[str] = None
if is_novel and count >= NOVEL_GROUP_MIN_COUNT:
prefix = "new rejection pattern:"
elif count >= min_count:
prefix = "rejection rate:"
if prefix is None:
continue
tier_label = "unknown" if tier is None else str(tier)
warnings.append(
f"{prefix} {count} rejection(s) in the last {window}h: "
f"{normalized_reason} (tier {tier_label}; "
f"most recent: {group['most_recent'].isoformat()})"
)
return warnings
def recent_decisions(
conn: sqlite3.Connection,
limit: int = 50,

View File

@@ -30,12 +30,16 @@ import dispatcher
from config import load_config
from metrics import (
_next_reset_date,
capability_ceilings,
capability_demand_warnings,
context_ceilings,
demand_ceiling_warnings,
local_energy_summary,
per_model,
pinch_summary,
quota_burn,
recent_decisions,
rejection_warnings,
scoring_coverage,
top_proficiency,
verdict_mix,
@@ -794,6 +798,482 @@ def test_context_ceilings_excludes_admin_deprecated_models(tmp_path):
assert ctx[key]["count"] == 1
# --- capability sub-ceiling demand warnings (2026-09-04 incident) ---------------
def _insert_model_row(
conn: sqlite3.Connection,
*,
model_id: str,
tier: int,
effective_context_window: int,
supports_vision: int,
supports_json_mode: int = 1,
) -> None:
"""Insert one active, routable catalog row with explicit capability flags.
``_seed_models`` covers the common case but hardcodes its capability
flags and effective window; the incident-reproduction tests need both
per-row.
"""
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, ?, ?, ?, 16384, 0.30, 0.10,
?, ?, 'standard', 'default', 'full', 'public', 'active',
?)
""",
(
model_id,
model_id,
tier,
effective_context_window,
effective_context_window,
supports_vision,
supports_json_mode,
_now().isoformat(),
),
)
conn.commit()
def _deprecate_model(conn: sqlite3.Connection, model_id: str) -> None:
"""Insert an admin override marking *model_id* deprecated (reason: cost)."""
conn.execute(
"INSERT INTO admin_model_overrides "
"(model_id, provider, availability, reason, updated_at) "
"VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)",
(model_id, _now().isoformat()),
)
conn.commit()
def _insert_rejected_decision(
conn: sqlite3.Connection,
*,
tier: int,
reason: str,
observed_at: str,
tokens: int = 200000,
images: int = 0,
json_mode: int = 0,
) -> None:
"""Insert a route_decisions row that selected no model (a 422 rejection).
``kind='chat'`` so the demand queries (which filter on kind) see it;
``latency_tolerance='interactive'`` to match the tier-1 bucket key the
ceiling checks use.
"""
conn.execute(
"""
INSERT INTO route_decisions (
observed_at, kind, task_category, task_tier,
required_context_tokens, confidence, classifier_ms,
classification_source, latency_tolerance, candidates_considered,
selected_model, selected_provider, rejected_reason,
session_key, tools, images, json_mode, streamed
) VALUES (?, 'chat', 'coding_general', ?, ?, NULL, NULL,
'override', 'interactive', 0, NULL, NULL, ?,
'sess', 0, ?, ?, 0)
""",
(observed_at, tier, tokens, reason, images, json_mode),
)
conn.commit()
def _seed_incident_state(conn: sqlite3.Connection) -> str:
"""Seed the 2026-09-04 incident model state; returns the shared 'now' stamp.
Three tier-1 rows: kimi-k3 and kimi-k3-fast — the only vision-capable
large-context models — deprecated through admin overrides, and
deepseek-v4-flash (no vision, 262,128 effective tokens) left active.
The overall tier-1 interactive ceiling stays high while the vision
sub-ceiling collapses to zero, which is exactly the shape every
pre-2026-09-05 ceiling check was blind to.
"""
for model_id, vision, eff_ctx in (
("kimi-k3", 1, 782324),
("kimi-k3-fast", 1, 782324),
("deepseek-v4-flash", 0, 262128),
):
_insert_model_row(
conn,
model_id=model_id,
tier=1,
effective_context_window=eff_ctx,
supports_vision=vision,
)
for model_id in ("kimi-k3", "kimi-k3-fast"):
_deprecate_model(conn, model_id)
return _now().isoformat()
def test_capability_demand_warning_incident_reproduction(tmp_path):
"""The 2026-09-04 incident state warns on the vision sub-ceiling while
every existing ``(tier, latency_tolerance)`` check stays silent.
In the incident, kimi-k3 and kimi-k3-fast — the only vision-capable
tier-1 models with a large context — were deprecated through admin
overrides, so a 200,000-token image request was unservable even though
the overall tier-1 ceiling (262,128, from non-vision deepseek-v4-flash)
comfortably covered that demand. The new capability check must fire;
the existing demand_ceiling_warnings must NOT — that silence is what
hid the incident for 19 hours.
"""
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
ts = _seed_incident_state(conn)
_insert_rejected_decision(
conn,
tier=1,
reason=(
"context >= 242486 tokens; "
"vision-capable model (request carries image(s))"
),
observed_at=ts,
tokens=200000,
images=1,
)
# The overall tier-1 interactive ceiling is untouched by the deprecations.
ctx = context_ceilings(conn, CFG)
assert ctx[(1, "interactive")]["ceiling"] == 262128
# The vision-gated subset collapsed: no candidate survives the gate.
cap_ctx = capability_ceilings(conn, CFG)
assert cap_ctx["vision"][(1, "interactive")]["ceiling"] == 0
assert cap_ctx["vision"][(1, "interactive")]["count"] == 0
# deepseek-v4-flash still serves the json_mode subset.
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 262128
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
assert len(warnings) == 1
w = warnings[0]
assert "vision" in w
assert "tier 1" in w
assert "ceiling (0)" in w
assert "200000" in w
# The pre-existing check must stay silent on the same state: the overall
# tier-1 ceiling (262128) still exceeds the observed max demand (200000).
assert demand_ceiling_warnings(conn, CFG, ctx, [1]) == []
def test_capability_demand_no_demand_no_warning(tmp_path):
"""A state with demand but no capability-flagged demand warns about nothing.
Requests exist at the tier — far above every ceiling — but none carry
images or ask for JSON mode, so both capability subsets are unexercised:
an unexercised sub-ceiling is not a warning.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
_insert_rejected_decision(
conn,
tier=2,
reason="context >= 999999 tokens",
observed_at=_now().isoformat(),
tokens=999999,
)
assert capability_demand_warnings(conn, CFG, capability_ceilings(conn, CFG)) == []
def test_capability_json_mode_demand_warning(tmp_path):
"""Deprecating the only json_mode-capable model trips the json_mode warning.
Same incident shape as the vision case, different capability dimension:
the json_mode sub-ceiling collapses to zero while the overall ceiling
stays fine, and a json_mode request that cannot be served produces a
warning naming json_mode.
"""
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
_insert_model_row(
conn,
model_id="jm-big",
tier=1,
effective_context_window=782324,
supports_vision=0,
supports_json_mode=1,
)
_insert_model_row(
conn,
model_id="plain-fallback",
tier=1,
effective_context_window=262128,
supports_vision=0,
supports_json_mode=0,
)
_deprecate_model(conn, "jm-big")
_insert_rejected_decision(
conn,
tier=1,
reason=(
"context >= 242486 tokens; json-mode-capable model "
"(json_mode requested)"
),
observed_at=_now().isoformat(),
tokens=200000,
json_mode=1,
)
cap_ctx = capability_ceilings(conn, CFG)
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 0
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
assert len(warnings) == 1
w = warnings[0]
assert "json_mode" in w
assert "tier 1" in w
assert "ceiling (0)" in w
assert "200000" in w
# --- rejection warnings (reactive rejection detector) --------------------------
def test_rejection_warnings_zero_rejections(tmp_path):
"""Zero NULL-selection rows produce no rejection warnings.
The absence of rejections is not a signal worth reporting — the
detector watches rates and new patterns, not mere existence.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
assert rejection_warnings(conn, CFG) == []
def test_rejection_warnings_novel_group_at_incident_volume_fires(tmp_path):
"""The 2026-09-04 incident shape — a novel group of only 2 — must fire.
The incident produced exactly 2 image rejections, below the routine
noise floor (about 3/hour on this deployment), so no rate threshold can
catch it; the novelty prong — absent from the 24h baseline, count
>= 2 — is what must fire here. Both rows share one timestamp so the
rendered most-recent value is deterministic.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
for reason in (
"tier >= 1; context >= 242486 tokens; interactive; "
"vision-capable model (request carries image(s))",
"tier >= 1; context >= 195000 tokens; interactive; "
"vision-capable model (request carries image(s))",
):
_insert_rejected_decision(conn, tier=1, reason=reason, observed_at=ts)
warnings = rejection_warnings(conn, CFG)
assert len(warnings) == 1
w = warnings[0]
assert w.startswith("new rejection pattern:")
# Token counts collapse: both raw reasons normalize to "context >= N tokens".
assert "2 rejection(s)" in w
assert "context >= N tokens" in w
# The tier is rendered from the structured task_tier column, not the string.
assert "(tier 1;" in w
assert ts in w
def test_rejection_warnings_familiar_group_below_threshold_silent(tmp_path):
"""Routine over-large tier-3 traffic — 3/hour and familiar — stays silent.
Regression for the measured routine floor: this deployment routinely
sees about 3 rejections per hour of ordinary over-large tier-3 requests
that behave as designed. The group is familiar (present in the 24h
baseline window) and below ``rejection_warning_min_count`` (6), so
neither the novelty prong nor the rate prong fires.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
now = _now()
# Baseline occurrence: 2h ago — inside the 24h baseline, outside the 1h
# alert window.
_insert_rejected_decision(
conn,
tier=3,
reason="tier >= 3; context >= 40000 tokens; interactive",
observed_at=(now - timedelta(hours=2)).isoformat(),
)
# The live-DB-measured routine hourly volume, with distinct token counts
# so digit normalization does real grouping work.
for tokens in (30000, 31000, 32000):
_insert_rejected_decision(
conn,
tier=3,
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
observed_at=now.isoformat(),
)
assert rejection_warnings(conn, CFG) == []
def test_rejection_warnings_familiar_group_at_threshold_fires(tmp_path):
"""A familiar group at ``rejection_warning_min_count`` (6) fires as a rate."""
conn = _make_db(tmp_path)
_seed_models(conn)
now = _now()
_insert_rejected_decision(
conn,
tier=3,
reason="tier >= 3; context >= 40000 tokens; interactive",
observed_at=(now - timedelta(hours=2)).isoformat(),
)
for tokens in (30000, 31000, 32000, 33000, 34000, 35000):
_insert_rejected_decision(
conn,
tier=3,
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
observed_at=now.isoformat(),
)
warnings = rejection_warnings(conn, CFG)
assert len(warnings) == 1
w = warnings[0]
assert w.startswith("rejection rate:")
assert "6 rejection(s)" in w
def test_rejection_warnings_novel_single_rejection_silent(tmp_path):
"""A single novel rejection stays silent.
A genuinely impossible one-off request SHOULD 422; presence of one
rejection is not a signal. n=1 is below NOVEL_GROUP_MIN_COUNT (2).
"""
conn = _make_db(tmp_path)
_seed_models(conn)
_insert_rejected_decision(
conn,
tier=1,
reason=(
"tier >= 1; context >= 242486 tokens; interactive; "
"vision-capable model (request carries image(s))"
),
observed_at=_now().isoformat(),
)
assert rejection_warnings(conn, CFG) == []
def test_rejection_warnings_groups_by_structured_tier(tmp_path):
"""Tier-1 and tier-3 rejections of the same normalized shape never merge.
Regression for the sibling finding of the incident review: digit
normalization erases the tier embedded in the reason string ("tier >= 1"
and "tier >= 3" both become "tier >= N"), so grouping by normalized
string alone would produce ONE merged count-4 group, hiding which
tier's candidate set went empty. The discriminator is the structured
``task_tier`` column, rendered back into the body as "(tier N; ...)".
"""
conn = _make_db(tmp_path)
_seed_models(conn)
ts = _now().isoformat()
for tier, tokens in ((1, 242486), (1, 50000), (3, 229702), (3, 30000)):
_insert_rejected_decision(
conn,
tier=tier,
reason=f"tier >= {tier}; context >= {tokens} tokens; interactive",
observed_at=ts,
)
warnings = rejection_warnings(conn, CFG)
assert len(warnings) == 2
tier1 = [w for w in warnings if "(tier 1;" in w]
tier3 = [w for w in warnings if "(tier 3;" in w]
assert len(tier1) == 1
assert len(tier3) == 1
assert "2 rejection(s)" in tier1[0]
assert "2 rejection(s)" in tier3[0]
def test_rejection_warnings_old_rejections_not_counted(tmp_path):
"""A rejection outside both windows is not counted at all.
Two same-shape rows 3 days old reach the novelty floor of 2, so the
only thing that keeps them silent is their age — outside the 1h alert
window and beyond the 25h total pull window they are not even read.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
for tokens in (242486, 195000):
_insert_rejected_decision(
conn,
tier=1,
reason=f"tier >= 1; context >= {tokens} tokens; interactive",
observed_at=(_now() - timedelta(days=3)).isoformat(),
)
assert rejection_warnings(conn, CFG) == []
def test_rejection_warnings_custom_window(tmp_path):
"""``rejection_warning_window_hours`` widens the alert window.
Two same-shape rows 2h old sit outside the default 1h window but
inside a 3h window; with the knob set they form a novel group at the
incident volume and must fire.
"""
conn = _make_db(tmp_path)
_seed_models(conn)
two_hours_ago = (_now() - timedelta(hours=2)).isoformat()
for tokens in (242486, 195000):
_insert_rejected_decision(
conn,
tier=1,
reason=(
f"tier >= 1; context >= {tokens} tokens; interactive; "
"vision-capable model (request carries image(s))"
),
observed_at=two_hours_ago,
)
wide_cfg = SimpleNamespace(
objective=SimpleNamespace(
rejection_warning_window_hours=3,
rejection_warning_baseline_hours=24,
rejection_warning_min_count=6,
)
)
warnings = rejection_warnings(conn, wide_cfg)
assert len(warnings) == 1
assert warnings[0].startswith("new rejection pattern:")
assert "2 rejection(s)" in warnings[0]
def test_scoring_coverage_includes_new_warnings(tmp_path):
"""``scoring_coverage`` surfaces both new warning classes on the incident state.
The 2026-09-04 incident produced two image rejections against a catalog
whose vision sub-ceiling had collapsed. Both the capability demand
warning and the novel rejection pattern must reach the same warnings
list that /metrics, the admin portal and the TUI read.
"""
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
ts = _seed_incident_state(conn)
for tokens in (242486, 195000):
_insert_rejected_decision(
conn,
tier=1,
reason=(
f"context >= {tokens} tokens; "
"vision-capable model (request carries image(s))"
),
observed_at=ts,
images=1,
)
result = scoring_coverage(conn, CFG)
warnings = result["warnings"]
assert any("vision" in w for w in warnings)
assert any(w.startswith("new rejection pattern:") for w in warnings)
# --- recent_decisions tests ---------------------------------------------------

View File

@@ -31,10 +31,10 @@ def _now() -> datetime:
return datetime.now(timezone.utc)
def _make_db(tmp_path: Path) -> sqlite3.Connection:
def _make_db(tmp_path: Path, extra_sql: str = "") -> sqlite3.Connection:
conn = sqlite3.connect(str(tmp_path / "test.db"))
conn.row_factory = sqlite3.Row
conn.executescript(SCHEMA_SQL)
conn.executescript(SCHEMA_SQL + extra_sql)
return conn
@@ -358,6 +358,102 @@ def test_metrics_empty_db_returns_200(monkeypatch, tmp_path):
assert "generated_at" in data
# admin_model_overrides is NOT in schema.sql — dispatcher creates it at
# startup via ensure_route_decisions-style migration, so endpoint tests that
# need it build it with extra_sql (pattern from tests/test_metrics.py).
_ADMIN_TABLE_SQL = """
CREATE TABLE IF NOT EXISTS admin_model_overrides (
model_id TEXT NOT NULL,
provider TEXT NOT NULL,
availability TEXT NOT NULL,
reason TEXT,
updated_at TEXT NOT NULL,
PRIMARY KEY (model_id, provider)
);
CREATE INDEX IF NOT EXISTS idx_admin_model_overrides_availability
ON admin_model_overrides (availability);
"""
def test_metrics_coverage_warnings_include_rejection_and_capability(
tmp_path, monkeypatch
):
"""GET /metrics surfaces the new rejection and capability warnings.
Seeds the 2026-09-04 incident state — kimi-k3 and kimi-k3-fast (the only
vision-capable tier-1 models) deprecated via admin overrides, plus two
image rejections — into the endpoint's temp DB, and asserts the warnings
reach the same coverage.warnings list the admin bell and TUI render.
The rejection warning is the deterministic assertion; on this exact seed
the capability warning fires too.
"""
now = _now().isoformat()
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
for model_id, vision, eff_ctx in (
("kimi-k3", 1, 782324),
("kimi-k3-fast", 1, 782324),
("deepseek-v4-flash", 0, 262128),
):
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, last_updated
) VALUES (?, 'neuralwatt', ?, 1, ?, ?, 16384, 0.30, 0.10,
?, 1, 'standard', 'default', 'full', 'public', 'active',
?)
""",
(model_id, model_id, eff_ctx, eff_ctx, vision, now),
)
for model_id in ("kimi-k3", "kimi-k3-fast"):
conn.execute(
"INSERT INTO admin_model_overrides "
"(model_id, provider, availability, reason, updated_at) "
"VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)",
(model_id, now),
)
for tokens in (242486, 195000):
conn.execute(
"""
INSERT INTO route_decisions (
observed_at, kind, task_category, task_tier,
required_context_tokens, latency_tolerance,
selected_model, selected_provider, rejected_reason,
tools, images, json_mode, streamed
) VALUES (?, 'chat', 'coding_general', 1, 200000,
'interactive', NULL, NULL, ?, 0, 1, 0, 0)
""",
(
now,
f"context >= {tokens} tokens; "
f"vision-capable model (request carries image(s))",
),
)
conn.commit()
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", False)
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
with TestClient(dispatcher.app) as client:
resp = client.get("/metrics")
assert resp.status_code == 200
data = resp.json()
warnings = data["coverage"]["warnings"]
assert len(warnings) > 0
assert any("rejection" in w for w in warnings)
# The capability warning is secondary per the plan (it depends on the
# seeded models) but this seed reproduces the incident, so it fires.
assert any("vision" in w for w in warnings)
def _sse_frame(line: str | bytes) -> dict:
"""Parse one ``data: <json>`` SSE line and return the JSON payload."""
if isinstance(line, bytes):