feat: capability-aware ceiling warnings + reactive rejection detector #31
@@ -84,6 +84,26 @@ objective:
|
||||
# 29-31 to avoid shorter-month edge cases).
|
||||
billing_reset_day: 6
|
||||
|
||||
# Rejection-rate detection over route_decisions rows that selected no model
|
||||
# (422 "no model satisfies the hard filters"). The signal is novelty or rate —
|
||||
# NEVER mere presence: this deployment routinely has ~3 rejections/hour of
|
||||
# ordinary over-large tier-3 requests that are behaving as designed (6 in the
|
||||
# last 24h, 17 all-time at 2026-09-05), so a count-only tripwire would be
|
||||
# permanently on.
|
||||
#
|
||||
# alert window (hours): rejections this recent are counted per
|
||||
# (task_tier, digit-normalized reason) group.
|
||||
rejection_warning_window_hours: 1
|
||||
# baseline window (hours) BEFORE the alert window: groups absent here are
|
||||
# "new". A new group warns from 2 occurrences — the 2026-09-04 vision
|
||||
# incident produced exactly 2 and no rate threshold can sit below routine
|
||||
# noise yet above that.
|
||||
rejection_warning_baseline_hours: 24
|
||||
# a group already present in the baseline warns only at this many
|
||||
# occurrences inside the alert window: 6 = 2x the observed routine hourly
|
||||
# peak, far below the dozens/hour a deprecation flare produces.
|
||||
rejection_warning_min_count: 6
|
||||
|
||||
context:
|
||||
safety_factor: 0.75 # fraction of advertised context treated as usable
|
||||
default_output_reserve_tokens: 4096
|
||||
|
||||
@@ -54,6 +54,9 @@ class Objective(StrictModel):
|
||||
quota_runway_warning_hours: Optional[int] = None
|
||||
quota_burn_min_segment_samples: Optional[int] = None
|
||||
quota_burn_min_segment_hours: Optional[float] = None
|
||||
rejection_warning_window_hours: Optional[int] = None
|
||||
rejection_warning_baseline_hours: Optional[int] = None
|
||||
rejection_warning_min_count: Optional[int] = None
|
||||
|
||||
@field_validator("quality_tolerance")
|
||||
@classmethod
|
||||
@@ -115,6 +118,33 @@ class Objective(StrictModel):
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("rejection_warning_window_hours")
|
||||
@classmethod
|
||||
def rejection_window_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||
if v is not None and v <= 0:
|
||||
raise ValueError(
|
||||
"objective.rejection_warning_window_hours must be > 0"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("rejection_warning_baseline_hours")
|
||||
@classmethod
|
||||
def rejection_baseline_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||
if v is not None and v <= 0:
|
||||
raise ValueError(
|
||||
"objective.rejection_warning_baseline_hours must be > 0"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("rejection_warning_min_count")
|
||||
@classmethod
|
||||
def rejection_min_count_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||
if v is not None and v <= 0:
|
||||
raise ValueError(
|
||||
"objective.rejection_warning_min_count must be > 0"
|
||||
)
|
||||
return v
|
||||
|
||||
|
||||
class ContextOverride(StrictModel):
|
||||
"""Per-model context handling, for a row whose real limits are known.
|
||||
|
||||
268
src/metrics.py
268
src/metrics.py
@@ -14,6 +14,9 @@ quota_burn — kWh metered in the last 30 d and in the current billing
|
||||
credit balance, burn rate and runway from
|
||||
allowance_remaining_usd
|
||||
scoring_coverage — which scoring axes actually have data
|
||||
capability_ceilings — vision / json_mode context-window sub-ceilings
|
||||
capability_demand_warnings — demand-relative warnings for those sub-ceilings
|
||||
rejection_warnings — grouped rejection-rate / new-pattern warnings
|
||||
recent_decisions — last N rows from the route_decisions observability table
|
||||
per_model — per-model aggregates over energy_observations (last 30 d)
|
||||
verdict_mix — counts by verdict from verifications (last N days)
|
||||
@@ -23,9 +26,10 @@ pinch_summary — context-pruning savings over the last 30 d
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import sqlite3
|
||||
from datetime import date, datetime, timedelta, timezone
|
||||
from typing import Any, List, Optional
|
||||
from typing import Any, Final, List, Optional
|
||||
|
||||
import routing
|
||||
|
||||
@@ -406,6 +410,10 @@ def scoring_coverage(
|
||||
f"tier {tier} has 0 eligible models for latency_tolerance={lat_tol}"
|
||||
)
|
||||
|
||||
cap_ctx = capability_ceilings(conn, cfg)
|
||||
warnings.extend(capability_demand_warnings(conn, cfg, cap_ctx))
|
||||
warnings.extend(rejection_warnings(conn, cfg))
|
||||
|
||||
return {
|
||||
"routable_models": total,
|
||||
"with_energy_data": total - len(missing_energy),
|
||||
@@ -568,6 +576,8 @@ def context_ceilings_with_rows(
|
||||
cfg: Any,
|
||||
rows: List[dict],
|
||||
exclude_models: Optional[set[str]] = None,
|
||||
require_vision: bool = False,
|
||||
require_json_mode: bool = False,
|
||||
) -> dict:
|
||||
"""Like ``context_ceilings`` but over caller-supplied model rows.
|
||||
|
||||
@@ -575,6 +585,9 @@ def context_ceilings_with_rows(
|
||||
proficiency and because the default ``exclude_models`` set is read from
|
||||
``admin_model_overrides``. Callers that already know the exclusion set
|
||||
(e.g. admin.py applying a just-committed override) may pass it explicitly.
|
||||
``require_vision`` / ``require_json_mode`` gate the candidate set the
|
||||
same way live routing gates it, so a capability-gated sub-ceiling goes
|
||||
through this one path rather than a second copy of the filters.
|
||||
"""
|
||||
if exclude_models is None:
|
||||
exclude_models = _admin_deprecated_models(conn)
|
||||
@@ -600,8 +613,8 @@ def context_ceilings_with_rows(
|
||||
exclude_deprecated=True,
|
||||
exclude_models=exclude_models,
|
||||
min_tool_proficiency=None,
|
||||
require_vision=False,
|
||||
require_json_mode=False,
|
||||
require_vision=require_vision,
|
||||
require_json_mode=require_json_mode,
|
||||
)
|
||||
ceiling = max((r["effective_context_window"] for r in candidates), default=0)
|
||||
result[(tier, latency_tolerance)] = {
|
||||
@@ -611,6 +624,255 @@ def context_ceilings_with_rows(
|
||||
return result
|
||||
|
||||
|
||||
def capability_ceilings(
|
||||
conn: sqlite3.Connection,
|
||||
cfg: Any,
|
||||
) -> dict:
|
||||
"""Context-window ceiling per capability-gated subset, by tier.
|
||||
|
||||
The plain ``(tier, latency_tolerance)`` buckets from
|
||||
``context_ceilings`` are blind to capability gates: during the
|
||||
2026-09-04 incident the tier-1 interactive ceiling stayed at 782,324
|
||||
while the vision-capable subset had collapsed to 192,500 (the only
|
||||
vision rows large enough had been deprecated through admin overrides),
|
||||
so every existing demand check was silent on requests that 422ed.
|
||||
Vision and JSON mode are the two flags that can independently empty
|
||||
the candidate set — both fail CLOSED on an unknown catalog flag, which
|
||||
is what lets them shrink the set sharply — so each gets its own series
|
||||
here. Deliberately NOT one bucket per capability combination; that is
|
||||
a combinatorial explosion over dimensions that do not interact.
|
||||
|
||||
Delegates to ``context_ceilings_with_rows`` once per dimension, so every
|
||||
sub-ceiling comes from the same ``routing.select_candidates`` path
|
||||
live routing uses, admin deprecations included.
|
||||
|
||||
Returns
|
||||
``{"vision": {(tier, latency_tolerance): {"ceiling", "count"}}, "json_mode":
|
||||
{...}}``, one sub-dict per capability dimension, each keyed by the
|
||||
``(tier, latency_tolerance)`` pair present in the DB.
|
||||
"""
|
||||
rows = [dict(r) for r in conn.execute("SELECT * FROM models")]
|
||||
exclude_models = _admin_deprecated_models(conn)
|
||||
result: dict[str, dict[tuple, dict[str, int]]] = {}
|
||||
for dimension in ("vision", "json_mode"):
|
||||
sub_ceilings = context_ceilings_with_rows(
|
||||
conn,
|
||||
cfg,
|
||||
rows,
|
||||
exclude_models=exclude_models,
|
||||
require_vision=dimension == "vision",
|
||||
require_json_mode=dimension == "json_mode",
|
||||
)
|
||||
result[dimension] = sub_ceilings
|
||||
return result
|
||||
|
||||
|
||||
def capability_demand_warnings(
|
||||
conn: sqlite3.Connection,
|
||||
cfg: Any,
|
||||
cap_ctx: dict,
|
||||
) -> List[str]:
|
||||
"""Demand-relative warnings for the capability-gated sub-ceilings.
|
||||
|
||||
Mirrors ``demand_ceiling_warnings`` for each dimension — 7d MAX probe
|
||||
first with a 30d fallback, silence for a bucket no such request ever
|
||||
hit, else window = 7 when the 7d row count is >= 10 (30 otherwise),
|
||||
warning only when that window's observed max exceeds the sub-ceiling.
|
||||
The one difference is what "demand" means: request rows must carry the
|
||||
capability (``images`` / ``json_mode`` columns on route_decisions), so
|
||||
a vision sub-ceiling is compared only against requests that actually
|
||||
carry images, for the ``(tier, "interactive")`` bucket. ``cfg`` is
|
||||
accepted for signature symmetry with ``demand_ceiling_warnings``; this
|
||||
detector has no escalation counterpart to read from it.
|
||||
|
||||
The demand read is trailing-window, so a warning stays alive until a
|
||||
big historical request ages out of it; the ``window: last {n}d``
|
||||
annotation in the message is what keeps that self-explaining.
|
||||
"""
|
||||
warnings: List[str] = []
|
||||
|
||||
dimension_columns = {
|
||||
"vision": "images",
|
||||
"json_mode": "json_mode",
|
||||
}
|
||||
capability_nouns = {
|
||||
"vision": "for requests carrying images",
|
||||
"json_mode": "for requests requesting JSON mode",
|
||||
}
|
||||
|
||||
for dimension in ("vision", "json_mode"):
|
||||
buckets = cap_ctx.get(dimension) or {}
|
||||
for tier in sorted({t for (t, _) in buckets}):
|
||||
bucket = buckets.get((tier, "interactive"))
|
||||
if bucket is None:
|
||||
continue
|
||||
column = dimension_columns.get(dimension)
|
||||
if column is None:
|
||||
continue
|
||||
row = conn.execute(
|
||||
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
|
||||
" WHERE julianday(observed_at) > julianday('now', '-7 days')"
|
||||
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
|
||||
(tier,),
|
||||
).fetchone()
|
||||
mx_7d = row["mx"] if row else None
|
||||
if mx_7d is None:
|
||||
row = conn.execute(
|
||||
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
|
||||
" WHERE julianday(observed_at) > julianday('now', '-30 days')"
|
||||
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
|
||||
(tier,),
|
||||
).fetchone()
|
||||
mx_7d = row["mx"] if row else None
|
||||
if mx_7d is None:
|
||||
# No request carrying this capability was ever routed for this
|
||||
# tier — an unexercised sub-ceiling warns about nothing.
|
||||
continue
|
||||
row_cnt = conn.execute(
|
||||
"SELECT COUNT(*) AS n FROM route_decisions"
|
||||
" WHERE julianday(observed_at) > julianday('now', '-7 days')"
|
||||
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
|
||||
(tier,),
|
||||
).fetchone()["n"]
|
||||
window = 7 if row_cnt >= 10 else 30
|
||||
row = conn.execute(
|
||||
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
|
||||
f" WHERE julianday(observed_at) > julianday('now', '-{window} days')"
|
||||
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
|
||||
(tier,),
|
||||
).fetchone()
|
||||
observed_max = row["mx"] if row else 0
|
||||
ceiling = bucket["ceiling"]
|
||||
if observed_max > ceiling:
|
||||
warnings.append(
|
||||
f"{dimension}-capable tier {tier} context ceiling "
|
||||
f"({ceiling}) is below observed max demand "
|
||||
f"({observed_max}, window: last {window}d) "
|
||||
f"{capability_nouns[dimension]} — requests above this may "
|
||||
"return 422"
|
||||
)
|
||||
|
||||
return warnings
|
||||
|
||||
|
||||
# A novel (tier, normalized_reason) group must reach this count inside the
|
||||
# alert window before warning. The 2026-09-04 incident produced exactly 2
|
||||
# image rejections, so the floor must be <= 2; 1 would fire on any one-off
|
||||
# genuinely-impossible request that correctly 422s.
|
||||
NOVEL_GROUP_MIN_COUNT: Final = 2
|
||||
|
||||
# How many rejections a group needs inside the alert window before
|
||||
# "new rejection pattern:" is reported. A single rejection is
|
||||
# indistinguishable from a genuinely impossible request, which should 422 —
|
||||
# the signal is a rate, not the existence of a rejection.
|
||||
#
|
||||
# Caveat: "novel" means ZERO occurrences in the prior baseline window, so
|
||||
# a familiar-but-intermittent pattern can read as novel after a quiet gap
|
||||
# and fire one spurious "new rejection pattern:" warning. It is
|
||||
# self-clearing as the pattern's own rejections age into the baseline
|
||||
# window; widen objective.rejection_warning_baseline_hours to suppress it.
|
||||
|
||||
|
||||
def rejection_warnings(
|
||||
conn: sqlite3.Connection,
|
||||
cfg: Any,
|
||||
) -> List[str]:
|
||||
"""Reactive safety net over the rejections the router already logged.
|
||||
|
||||
The ceiling checks above only catch dimensions someone modelled; this
|
||||
one catches everything, at the cost of firing after the first failures
|
||||
rather than before. Rejections (``selected_model IS NULL AND
|
||||
rejected_reason IS NOT NULL``) are pulled once across the alert plus
|
||||
baseline window and split by exact age — alert bucket at
|
||||
``rejection_warning_window_hours`` or younger, baseline bucket older
|
||||
than the alert window but within ``rejection_warning_baseline_hours``
|
||||
of it — then grouped by *normalized* reason plus ``task_tier``.
|
||||
|
||||
Two signals, one line per group:
|
||||
|
||||
* ``rejection rate:`` — the group reached
|
||||
``objective.rejection_warning_min_count`` rejections in the alert
|
||||
window.
|
||||
* ``new rejection pattern:`` — the group is ABSENT from the baseline
|
||||
window and reached ``NOVEL_GROUP_MIN_COUNT`` in the alert window.
|
||||
This prefix wins when both would fire.
|
||||
|
||||
Zero alert-window rejections produce nothing at all: a rejection is
|
||||
not inherently an error, and a genuinely impossible request should
|
||||
422 — the signal is a rate.
|
||||
"""
|
||||
window = getattr(cfg.objective, "rejection_warning_window_hours", None)
|
||||
if window is None:
|
||||
window = 1
|
||||
baseline = getattr(cfg.objective, "rejection_warning_baseline_hours", None)
|
||||
if baseline is None:
|
||||
baseline = 24
|
||||
min_count = getattr(cfg.objective, "rejection_warning_min_count", None)
|
||||
if min_count is None:
|
||||
min_count = 6
|
||||
|
||||
now = datetime.now(timezone.utc)
|
||||
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT observed_at, task_tier, rejected_reason
|
||||
FROM route_decisions
|
||||
WHERE selected_model IS NULL
|
||||
AND rejected_reason IS NOT NULL
|
||||
AND julianday(observed_at) > julianday('now', '-' || ? || ' hours')
|
||||
ORDER BY observed_at DESC
|
||||
""",
|
||||
(str(window + baseline),),
|
||||
).fetchall()
|
||||
|
||||
alert_groups: dict[tuple[Optional[int], str], dict] = {}
|
||||
baseline_keys: set[tuple[Optional[int], str]] = set()
|
||||
for row in rows:
|
||||
try:
|
||||
observed = datetime.fromisoformat(row["observed_at"])
|
||||
except ValueError:
|
||||
# Defensive, same shape as quota_balance_and_burn: one malformed
|
||||
# timestamp must not take the whole report down.
|
||||
continue
|
||||
# Group on digit-normalized reasons: the raw strings embed the
|
||||
# request's own numbers ("context >= 242486 tokens"), so literal
|
||||
# grouping yields N groups of one and hides the pattern entirely.
|
||||
key = (row["task_tier"], re.sub(r"\d+", "N", row["rejected_reason"]))
|
||||
age_hours = (now - observed).total_seconds() / 3600.0
|
||||
if age_hours <= window:
|
||||
group = alert_groups.get(key)
|
||||
if group is None:
|
||||
# Rows arrive newest-first, so the first touch per group is
|
||||
# its most recent occurrence.
|
||||
alert_groups[key] = {"count": 1, "most_recent": observed}
|
||||
else:
|
||||
group["count"] += 1
|
||||
elif age_hours <= window + baseline:
|
||||
baseline_keys.add(key)
|
||||
|
||||
if not alert_groups:
|
||||
return []
|
||||
|
||||
warnings: List[str] = []
|
||||
for (tier, normalized_reason), group in alert_groups.items():
|
||||
count = group["count"]
|
||||
is_novel = (tier, normalized_reason) not in baseline_keys
|
||||
prefix: Optional[str] = None
|
||||
if is_novel and count >= NOVEL_GROUP_MIN_COUNT:
|
||||
prefix = "new rejection pattern:"
|
||||
elif count >= min_count:
|
||||
prefix = "rejection rate:"
|
||||
if prefix is None:
|
||||
continue
|
||||
tier_label = "unknown" if tier is None else str(tier)
|
||||
warnings.append(
|
||||
f"{prefix} {count} rejection(s) in the last {window}h: "
|
||||
f"{normalized_reason} (tier {tier_label}; "
|
||||
f"most recent: {group['most_recent'].isoformat()})"
|
||||
)
|
||||
return warnings
|
||||
|
||||
|
||||
def recent_decisions(
|
||||
conn: sqlite3.Connection,
|
||||
limit: int = 50,
|
||||
|
||||
@@ -30,12 +30,16 @@ import dispatcher
|
||||
from config import load_config
|
||||
from metrics import (
|
||||
_next_reset_date,
|
||||
capability_ceilings,
|
||||
capability_demand_warnings,
|
||||
context_ceilings,
|
||||
demand_ceiling_warnings,
|
||||
local_energy_summary,
|
||||
per_model,
|
||||
pinch_summary,
|
||||
quota_burn,
|
||||
recent_decisions,
|
||||
rejection_warnings,
|
||||
scoring_coverage,
|
||||
top_proficiency,
|
||||
verdict_mix,
|
||||
@@ -794,6 +798,482 @@ def test_context_ceilings_excludes_admin_deprecated_models(tmp_path):
|
||||
assert ctx[key]["count"] == 1
|
||||
|
||||
|
||||
# --- capability sub-ceiling demand warnings (2026-09-04 incident) ---------------
|
||||
|
||||
|
||||
def _insert_model_row(
|
||||
conn: sqlite3.Connection,
|
||||
*,
|
||||
model_id: str,
|
||||
tier: int,
|
||||
effective_context_window: int,
|
||||
supports_vision: int,
|
||||
supports_json_mode: int = 1,
|
||||
) -> None:
|
||||
"""Insert one active, routable catalog row with explicit capability flags.
|
||||
|
||||
``_seed_models`` covers the common case but hardcodes its capability
|
||||
flags and effective window; the incident-reproduction tests need both
|
||||
per-row.
|
||||
"""
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO models (
|
||||
model_id, provider, base_model_id, tier, context_window,
|
||||
effective_context_window, max_output_tokens,
|
||||
cost_per_1m_prompt, cost_per_1m_completion,
|
||||
supports_vision, supports_json_mode,
|
||||
latency_class, reasoning_mode, context_variant,
|
||||
access_level, availability, last_updated
|
||||
) VALUES (?, 'neuralwatt', ?, ?, ?, ?, 16384, 0.30, 0.10,
|
||||
?, ?, 'standard', 'default', 'full', 'public', 'active',
|
||||
?)
|
||||
""",
|
||||
(
|
||||
model_id,
|
||||
model_id,
|
||||
tier,
|
||||
effective_context_window,
|
||||
effective_context_window,
|
||||
supports_vision,
|
||||
supports_json_mode,
|
||||
_now().isoformat(),
|
||||
),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _deprecate_model(conn: sqlite3.Connection, model_id: str) -> None:
|
||||
"""Insert an admin override marking *model_id* deprecated (reason: cost)."""
|
||||
conn.execute(
|
||||
"INSERT INTO admin_model_overrides "
|
||||
"(model_id, provider, availability, reason, updated_at) "
|
||||
"VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)",
|
||||
(model_id, _now().isoformat()),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _insert_rejected_decision(
|
||||
conn: sqlite3.Connection,
|
||||
*,
|
||||
tier: int,
|
||||
reason: str,
|
||||
observed_at: str,
|
||||
tokens: int = 200000,
|
||||
images: int = 0,
|
||||
json_mode: int = 0,
|
||||
) -> None:
|
||||
"""Insert a route_decisions row that selected no model (a 422 rejection).
|
||||
|
||||
``kind='chat'`` so the demand queries (which filter on kind) see it;
|
||||
``latency_tolerance='interactive'`` to match the tier-1 bucket key the
|
||||
ceiling checks use.
|
||||
"""
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO route_decisions (
|
||||
observed_at, kind, task_category, task_tier,
|
||||
required_context_tokens, confidence, classifier_ms,
|
||||
classification_source, latency_tolerance, candidates_considered,
|
||||
selected_model, selected_provider, rejected_reason,
|
||||
session_key, tools, images, json_mode, streamed
|
||||
) VALUES (?, 'chat', 'coding_general', ?, ?, NULL, NULL,
|
||||
'override', 'interactive', 0, NULL, NULL, ?,
|
||||
'sess', 0, ?, ?, 0)
|
||||
""",
|
||||
(observed_at, tier, tokens, reason, images, json_mode),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _seed_incident_state(conn: sqlite3.Connection) -> str:
|
||||
"""Seed the 2026-09-04 incident model state; returns the shared 'now' stamp.
|
||||
|
||||
Three tier-1 rows: kimi-k3 and kimi-k3-fast — the only vision-capable
|
||||
large-context models — deprecated through admin overrides, and
|
||||
deepseek-v4-flash (no vision, 262,128 effective tokens) left active.
|
||||
The overall tier-1 interactive ceiling stays high while the vision
|
||||
sub-ceiling collapses to zero, which is exactly the shape every
|
||||
pre-2026-09-05 ceiling check was blind to.
|
||||
"""
|
||||
for model_id, vision, eff_ctx in (
|
||||
("kimi-k3", 1, 782324),
|
||||
("kimi-k3-fast", 1, 782324),
|
||||
("deepseek-v4-flash", 0, 262128),
|
||||
):
|
||||
_insert_model_row(
|
||||
conn,
|
||||
model_id=model_id,
|
||||
tier=1,
|
||||
effective_context_window=eff_ctx,
|
||||
supports_vision=vision,
|
||||
)
|
||||
for model_id in ("kimi-k3", "kimi-k3-fast"):
|
||||
_deprecate_model(conn, model_id)
|
||||
return _now().isoformat()
|
||||
|
||||
|
||||
def test_capability_demand_warning_incident_reproduction(tmp_path):
|
||||
"""The 2026-09-04 incident state warns on the vision sub-ceiling while
|
||||
every existing ``(tier, latency_tolerance)`` check stays silent.
|
||||
|
||||
In the incident, kimi-k3 and kimi-k3-fast — the only vision-capable
|
||||
tier-1 models with a large context — were deprecated through admin
|
||||
overrides, so a 200,000-token image request was unservable even though
|
||||
the overall tier-1 ceiling (262,128, from non-vision deepseek-v4-flash)
|
||||
comfortably covered that demand. The new capability check must fire;
|
||||
the existing demand_ceiling_warnings must NOT — that silence is what
|
||||
hid the incident for 19 hours.
|
||||
"""
|
||||
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
||||
ts = _seed_incident_state(conn)
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=1,
|
||||
reason=(
|
||||
"context >= 242486 tokens; "
|
||||
"vision-capable model (request carries image(s))"
|
||||
),
|
||||
observed_at=ts,
|
||||
tokens=200000,
|
||||
images=1,
|
||||
)
|
||||
|
||||
# The overall tier-1 interactive ceiling is untouched by the deprecations.
|
||||
ctx = context_ceilings(conn, CFG)
|
||||
assert ctx[(1, "interactive")]["ceiling"] == 262128
|
||||
|
||||
# The vision-gated subset collapsed: no candidate survives the gate.
|
||||
cap_ctx = capability_ceilings(conn, CFG)
|
||||
assert cap_ctx["vision"][(1, "interactive")]["ceiling"] == 0
|
||||
assert cap_ctx["vision"][(1, "interactive")]["count"] == 0
|
||||
# deepseek-v4-flash still serves the json_mode subset.
|
||||
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 262128
|
||||
|
||||
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
|
||||
assert len(warnings) == 1
|
||||
w = warnings[0]
|
||||
assert "vision" in w
|
||||
assert "tier 1" in w
|
||||
assert "ceiling (0)" in w
|
||||
assert "200000" in w
|
||||
|
||||
# The pre-existing check must stay silent on the same state: the overall
|
||||
# tier-1 ceiling (262128) still exceeds the observed max demand (200000).
|
||||
assert demand_ceiling_warnings(conn, CFG, ctx, [1]) == []
|
||||
|
||||
|
||||
def test_capability_demand_no_demand_no_warning(tmp_path):
|
||||
"""A state with demand but no capability-flagged demand warns about nothing.
|
||||
|
||||
Requests exist at the tier — far above every ceiling — but none carry
|
||||
images or ask for JSON mode, so both capability subsets are unexercised:
|
||||
an unexercised sub-ceiling is not a warning.
|
||||
"""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=2,
|
||||
reason="context >= 999999 tokens",
|
||||
observed_at=_now().isoformat(),
|
||||
tokens=999999,
|
||||
)
|
||||
|
||||
assert capability_demand_warnings(conn, CFG, capability_ceilings(conn, CFG)) == []
|
||||
|
||||
|
||||
def test_capability_json_mode_demand_warning(tmp_path):
|
||||
"""Deprecating the only json_mode-capable model trips the json_mode warning.
|
||||
|
||||
Same incident shape as the vision case, different capability dimension:
|
||||
the json_mode sub-ceiling collapses to zero while the overall ceiling
|
||||
stays fine, and a json_mode request that cannot be served produces a
|
||||
warning naming json_mode.
|
||||
"""
|
||||
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
||||
_insert_model_row(
|
||||
conn,
|
||||
model_id="jm-big",
|
||||
tier=1,
|
||||
effective_context_window=782324,
|
||||
supports_vision=0,
|
||||
supports_json_mode=1,
|
||||
)
|
||||
_insert_model_row(
|
||||
conn,
|
||||
model_id="plain-fallback",
|
||||
tier=1,
|
||||
effective_context_window=262128,
|
||||
supports_vision=0,
|
||||
supports_json_mode=0,
|
||||
)
|
||||
_deprecate_model(conn, "jm-big")
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=1,
|
||||
reason=(
|
||||
"context >= 242486 tokens; json-mode-capable model "
|
||||
"(json_mode requested)"
|
||||
),
|
||||
observed_at=_now().isoformat(),
|
||||
tokens=200000,
|
||||
json_mode=1,
|
||||
)
|
||||
|
||||
cap_ctx = capability_ceilings(conn, CFG)
|
||||
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 0
|
||||
|
||||
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
|
||||
assert len(warnings) == 1
|
||||
w = warnings[0]
|
||||
assert "json_mode" in w
|
||||
assert "tier 1" in w
|
||||
assert "ceiling (0)" in w
|
||||
assert "200000" in w
|
||||
|
||||
|
||||
# --- rejection warnings (reactive rejection detector) --------------------------
|
||||
|
||||
|
||||
def test_rejection_warnings_zero_rejections(tmp_path):
|
||||
"""Zero NULL-selection rows produce no rejection warnings.
|
||||
|
||||
The absence of rejections is not a signal worth reporting — the
|
||||
detector watches rates and new patterns, not mere existence.
|
||||
"""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
assert rejection_warnings(conn, CFG) == []
|
||||
|
||||
|
||||
def test_rejection_warnings_novel_group_at_incident_volume_fires(tmp_path):
|
||||
"""The 2026-09-04 incident shape — a novel group of only 2 — must fire.
|
||||
|
||||
The incident produced exactly 2 image rejections, below the routine
|
||||
noise floor (about 3/hour on this deployment), so no rate threshold can
|
||||
catch it; the novelty prong — absent from the 24h baseline, count
|
||||
>= 2 — is what must fire here. Both rows share one timestamp so the
|
||||
rendered most-recent value is deterministic.
|
||||
"""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
ts = _now().isoformat()
|
||||
for reason in (
|
||||
"tier >= 1; context >= 242486 tokens; interactive; "
|
||||
"vision-capable model (request carries image(s))",
|
||||
"tier >= 1; context >= 195000 tokens; interactive; "
|
||||
"vision-capable model (request carries image(s))",
|
||||
):
|
||||
_insert_rejected_decision(conn, tier=1, reason=reason, observed_at=ts)
|
||||
|
||||
warnings = rejection_warnings(conn, CFG)
|
||||
assert len(warnings) == 1
|
||||
w = warnings[0]
|
||||
assert w.startswith("new rejection pattern:")
|
||||
# Token counts collapse: both raw reasons normalize to "context >= N tokens".
|
||||
assert "2 rejection(s)" in w
|
||||
assert "context >= N tokens" in w
|
||||
# The tier is rendered from the structured task_tier column, not the string.
|
||||
assert "(tier 1;" in w
|
||||
assert ts in w
|
||||
|
||||
|
||||
def test_rejection_warnings_familiar_group_below_threshold_silent(tmp_path):
|
||||
"""Routine over-large tier-3 traffic — 3/hour and familiar — stays silent.
|
||||
|
||||
Regression for the measured routine floor: this deployment routinely
|
||||
sees about 3 rejections per hour of ordinary over-large tier-3 requests
|
||||
that behave as designed. The group is familiar (present in the 24h
|
||||
baseline window) and below ``rejection_warning_min_count`` (6), so
|
||||
neither the novelty prong nor the rate prong fires.
|
||||
"""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
now = _now()
|
||||
# Baseline occurrence: 2h ago — inside the 24h baseline, outside the 1h
|
||||
# alert window.
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=3,
|
||||
reason="tier >= 3; context >= 40000 tokens; interactive",
|
||||
observed_at=(now - timedelta(hours=2)).isoformat(),
|
||||
)
|
||||
# The live-DB-measured routine hourly volume, with distinct token counts
|
||||
# so digit normalization does real grouping work.
|
||||
for tokens in (30000, 31000, 32000):
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=3,
|
||||
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
|
||||
observed_at=now.isoformat(),
|
||||
)
|
||||
|
||||
assert rejection_warnings(conn, CFG) == []
|
||||
|
||||
|
||||
def test_rejection_warnings_familiar_group_at_threshold_fires(tmp_path):
|
||||
"""A familiar group at ``rejection_warning_min_count`` (6) fires as a rate."""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
now = _now()
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=3,
|
||||
reason="tier >= 3; context >= 40000 tokens; interactive",
|
||||
observed_at=(now - timedelta(hours=2)).isoformat(),
|
||||
)
|
||||
for tokens in (30000, 31000, 32000, 33000, 34000, 35000):
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=3,
|
||||
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
|
||||
observed_at=now.isoformat(),
|
||||
)
|
||||
|
||||
warnings = rejection_warnings(conn, CFG)
|
||||
assert len(warnings) == 1
|
||||
w = warnings[0]
|
||||
assert w.startswith("rejection rate:")
|
||||
assert "6 rejection(s)" in w
|
||||
|
||||
|
||||
def test_rejection_warnings_novel_single_rejection_silent(tmp_path):
|
||||
"""A single novel rejection stays silent.
|
||||
|
||||
A genuinely impossible one-off request SHOULD 422; presence of one
|
||||
rejection is not a signal. n=1 is below NOVEL_GROUP_MIN_COUNT (2).
|
||||
"""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=1,
|
||||
reason=(
|
||||
"tier >= 1; context >= 242486 tokens; interactive; "
|
||||
"vision-capable model (request carries image(s))"
|
||||
),
|
||||
observed_at=_now().isoformat(),
|
||||
)
|
||||
|
||||
assert rejection_warnings(conn, CFG) == []
|
||||
|
||||
|
||||
def test_rejection_warnings_groups_by_structured_tier(tmp_path):
|
||||
"""Tier-1 and tier-3 rejections of the same normalized shape never merge.
|
||||
|
||||
Regression for the sibling finding of the incident review: digit
|
||||
normalization erases the tier embedded in the reason string ("tier >= 1"
|
||||
and "tier >= 3" both become "tier >= N"), so grouping by normalized
|
||||
string alone would produce ONE merged count-4 group, hiding which
|
||||
tier's candidate set went empty. The discriminator is the structured
|
||||
``task_tier`` column, rendered back into the body as "(tier N; ...)".
|
||||
"""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
ts = _now().isoformat()
|
||||
for tier, tokens in ((1, 242486), (1, 50000), (3, 229702), (3, 30000)):
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=tier,
|
||||
reason=f"tier >= {tier}; context >= {tokens} tokens; interactive",
|
||||
observed_at=ts,
|
||||
)
|
||||
|
||||
warnings = rejection_warnings(conn, CFG)
|
||||
assert len(warnings) == 2
|
||||
tier1 = [w for w in warnings if "(tier 1;" in w]
|
||||
tier3 = [w for w in warnings if "(tier 3;" in w]
|
||||
assert len(tier1) == 1
|
||||
assert len(tier3) == 1
|
||||
assert "2 rejection(s)" in tier1[0]
|
||||
assert "2 rejection(s)" in tier3[0]
|
||||
|
||||
|
||||
def test_rejection_warnings_old_rejections_not_counted(tmp_path):
|
||||
"""A rejection outside both windows is not counted at all.
|
||||
|
||||
Two same-shape rows 3 days old reach the novelty floor of 2, so the
|
||||
only thing that keeps them silent is their age — outside the 1h alert
|
||||
window and beyond the 25h total pull window they are not even read.
|
||||
"""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
for tokens in (242486, 195000):
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=1,
|
||||
reason=f"tier >= 1; context >= {tokens} tokens; interactive",
|
||||
observed_at=(_now() - timedelta(days=3)).isoformat(),
|
||||
)
|
||||
|
||||
assert rejection_warnings(conn, CFG) == []
|
||||
|
||||
|
||||
def test_rejection_warnings_custom_window(tmp_path):
|
||||
"""``rejection_warning_window_hours`` widens the alert window.
|
||||
|
||||
Two same-shape rows 2h old sit outside the default 1h window but
|
||||
inside a 3h window; with the knob set they form a novel group at the
|
||||
incident volume and must fire.
|
||||
"""
|
||||
conn = _make_db(tmp_path)
|
||||
_seed_models(conn)
|
||||
two_hours_ago = (_now() - timedelta(hours=2)).isoformat()
|
||||
for tokens in (242486, 195000):
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=1,
|
||||
reason=(
|
||||
f"tier >= 1; context >= {tokens} tokens; interactive; "
|
||||
"vision-capable model (request carries image(s))"
|
||||
),
|
||||
observed_at=two_hours_ago,
|
||||
)
|
||||
|
||||
wide_cfg = SimpleNamespace(
|
||||
objective=SimpleNamespace(
|
||||
rejection_warning_window_hours=3,
|
||||
rejection_warning_baseline_hours=24,
|
||||
rejection_warning_min_count=6,
|
||||
)
|
||||
)
|
||||
|
||||
warnings = rejection_warnings(conn, wide_cfg)
|
||||
assert len(warnings) == 1
|
||||
assert warnings[0].startswith("new rejection pattern:")
|
||||
assert "2 rejection(s)" in warnings[0]
|
||||
|
||||
|
||||
def test_scoring_coverage_includes_new_warnings(tmp_path):
|
||||
"""``scoring_coverage`` surfaces both new warning classes on the incident state.
|
||||
|
||||
The 2026-09-04 incident produced two image rejections against a catalog
|
||||
whose vision sub-ceiling had collapsed. Both the capability demand
|
||||
warning and the novel rejection pattern must reach the same warnings
|
||||
list that /metrics, the admin portal and the TUI read.
|
||||
"""
|
||||
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
||||
ts = _seed_incident_state(conn)
|
||||
for tokens in (242486, 195000):
|
||||
_insert_rejected_decision(
|
||||
conn,
|
||||
tier=1,
|
||||
reason=(
|
||||
f"context >= {tokens} tokens; "
|
||||
"vision-capable model (request carries image(s))"
|
||||
),
|
||||
observed_at=ts,
|
||||
images=1,
|
||||
)
|
||||
|
||||
result = scoring_coverage(conn, CFG)
|
||||
warnings = result["warnings"]
|
||||
assert any("vision" in w for w in warnings)
|
||||
assert any(w.startswith("new rejection pattern:") for w in warnings)
|
||||
|
||||
|
||||
# --- recent_decisions tests ---------------------------------------------------
|
||||
|
||||
|
||||
|
||||
@@ -31,10 +31,10 @@ def _now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _make_db(tmp_path: Path) -> sqlite3.Connection:
|
||||
def _make_db(tmp_path: Path, extra_sql: str = "") -> sqlite3.Connection:
|
||||
conn = sqlite3.connect(str(tmp_path / "test.db"))
|
||||
conn.row_factory = sqlite3.Row
|
||||
conn.executescript(SCHEMA_SQL)
|
||||
conn.executescript(SCHEMA_SQL + extra_sql)
|
||||
return conn
|
||||
|
||||
|
||||
@@ -358,6 +358,102 @@ def test_metrics_empty_db_returns_200(monkeypatch, tmp_path):
|
||||
assert "generated_at" in data
|
||||
|
||||
|
||||
# admin_model_overrides is NOT in schema.sql — dispatcher creates it at
|
||||
# startup via ensure_route_decisions-style migration, so endpoint tests that
|
||||
# need it build it with extra_sql (pattern from tests/test_metrics.py).
|
||||
_ADMIN_TABLE_SQL = """
|
||||
CREATE TABLE IF NOT EXISTS admin_model_overrides (
|
||||
model_id TEXT NOT NULL,
|
||||
provider TEXT NOT NULL,
|
||||
availability TEXT NOT NULL,
|
||||
reason TEXT,
|
||||
updated_at TEXT NOT NULL,
|
||||
PRIMARY KEY (model_id, provider)
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS idx_admin_model_overrides_availability
|
||||
ON admin_model_overrides (availability);
|
||||
"""
|
||||
|
||||
|
||||
def test_metrics_coverage_warnings_include_rejection_and_capability(
|
||||
tmp_path, monkeypatch
|
||||
):
|
||||
"""GET /metrics surfaces the new rejection and capability warnings.
|
||||
|
||||
Seeds the 2026-09-04 incident state — kimi-k3 and kimi-k3-fast (the only
|
||||
vision-capable tier-1 models) deprecated via admin overrides, plus two
|
||||
image rejections — into the endpoint's temp DB, and asserts the warnings
|
||||
reach the same coverage.warnings list the admin bell and TUI render.
|
||||
The rejection warning is the deterministic assertion; on this exact seed
|
||||
the capability warning fires too.
|
||||
"""
|
||||
now = _now().isoformat()
|
||||
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
||||
for model_id, vision, eff_ctx in (
|
||||
("kimi-k3", 1, 782324),
|
||||
("kimi-k3-fast", 1, 782324),
|
||||
("deepseek-v4-flash", 0, 262128),
|
||||
):
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO models (
|
||||
model_id, provider, base_model_id, tier, context_window,
|
||||
effective_context_window, max_output_tokens,
|
||||
cost_per_1m_prompt, cost_per_1m_completion,
|
||||
supports_vision, supports_json_mode,
|
||||
latency_class, reasoning_mode, context_variant,
|
||||
access_level, availability, last_updated
|
||||
) VALUES (?, 'neuralwatt', ?, 1, ?, ?, 16384, 0.30, 0.10,
|
||||
?, 1, 'standard', 'default', 'full', 'public', 'active',
|
||||
?)
|
||||
""",
|
||||
(model_id, model_id, eff_ctx, eff_ctx, vision, now),
|
||||
)
|
||||
for model_id in ("kimi-k3", "kimi-k3-fast"):
|
||||
conn.execute(
|
||||
"INSERT INTO admin_model_overrides "
|
||||
"(model_id, provider, availability, reason, updated_at) "
|
||||
"VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)",
|
||||
(model_id, now),
|
||||
)
|
||||
for tokens in (242486, 195000):
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO route_decisions (
|
||||
observed_at, kind, task_category, task_tier,
|
||||
required_context_tokens, latency_tolerance,
|
||||
selected_model, selected_provider, rejected_reason,
|
||||
tools, images, json_mode, streamed
|
||||
) VALUES (?, 'chat', 'coding_general', 1, 200000,
|
||||
'interactive', NULL, NULL, ?, 0, 1, 0, 0)
|
||||
""",
|
||||
(
|
||||
now,
|
||||
f"context >= {tokens} tokens; "
|
||||
f"vision-capable model (request carries image(s))",
|
||||
),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
|
||||
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
|
||||
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
|
||||
monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", False)
|
||||
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
|
||||
|
||||
with TestClient(dispatcher.app) as client:
|
||||
resp = client.get("/metrics")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
warnings = data["coverage"]["warnings"]
|
||||
assert len(warnings) > 0
|
||||
assert any("rejection" in w for w in warnings)
|
||||
# The capability warning is secondary per the plan (it depends on the
|
||||
# seeded models) but this seed reproduces the incident, so it fires.
|
||||
assert any("vision" in w for w in warnings)
|
||||
|
||||
|
||||
def _sse_frame(line: str | bytes) -> dict:
|
||||
"""Parse one ``data: <json>`` SSE line and return the JSON payload."""
|
||||
if isinstance(line, bytes):
|
||||
|
||||
Reference in New Issue
Block a user