feat: capability-aware ceiling warnings + reactive rejection detector #31
@@ -84,6 +84,26 @@ objective:
|
|||||||
# 29-31 to avoid shorter-month edge cases).
|
# 29-31 to avoid shorter-month edge cases).
|
||||||
billing_reset_day: 6
|
billing_reset_day: 6
|
||||||
|
|
||||||
|
# Rejection-rate detection over route_decisions rows that selected no model
|
||||||
|
# (422 "no model satisfies the hard filters"). The signal is novelty or rate —
|
||||||
|
# NEVER mere presence: this deployment routinely has ~3 rejections/hour of
|
||||||
|
# ordinary over-large tier-3 requests that are behaving as designed (6 in the
|
||||||
|
# last 24h, 17 all-time at 2026-09-05), so a count-only tripwire would be
|
||||||
|
# permanently on.
|
||||||
|
#
|
||||||
|
# alert window (hours): rejections this recent are counted per
|
||||||
|
# (task_tier, digit-normalized reason) group.
|
||||||
|
rejection_warning_window_hours: 1
|
||||||
|
# baseline window (hours) BEFORE the alert window: groups absent here are
|
||||||
|
# "new". A new group warns from 2 occurrences — the 2026-09-04 vision
|
||||||
|
# incident produced exactly 2 and no rate threshold can sit below routine
|
||||||
|
# noise yet above that.
|
||||||
|
rejection_warning_baseline_hours: 24
|
||||||
|
# a group already present in the baseline warns only at this many
|
||||||
|
# occurrences inside the alert window: 6 = 2x the observed routine hourly
|
||||||
|
# peak, far below the dozens/hour a deprecation flare produces.
|
||||||
|
rejection_warning_min_count: 6
|
||||||
|
|
||||||
context:
|
context:
|
||||||
safety_factor: 0.75 # fraction of advertised context treated as usable
|
safety_factor: 0.75 # fraction of advertised context treated as usable
|
||||||
default_output_reserve_tokens: 4096
|
default_output_reserve_tokens: 4096
|
||||||
|
|||||||
@@ -54,6 +54,9 @@ class Objective(StrictModel):
|
|||||||
quota_runway_warning_hours: Optional[int] = None
|
quota_runway_warning_hours: Optional[int] = None
|
||||||
quota_burn_min_segment_samples: Optional[int] = None
|
quota_burn_min_segment_samples: Optional[int] = None
|
||||||
quota_burn_min_segment_hours: Optional[float] = None
|
quota_burn_min_segment_hours: Optional[float] = None
|
||||||
|
rejection_warning_window_hours: Optional[int] = None
|
||||||
|
rejection_warning_baseline_hours: Optional[int] = None
|
||||||
|
rejection_warning_min_count: Optional[int] = None
|
||||||
|
|
||||||
@field_validator("quality_tolerance")
|
@field_validator("quality_tolerance")
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -115,6 +118,33 @@ class Objective(StrictModel):
|
|||||||
)
|
)
|
||||||
return v
|
return v
|
||||||
|
|
||||||
|
@field_validator("rejection_warning_window_hours")
|
||||||
|
@classmethod
|
||||||
|
def rejection_window_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||||
|
if v is not None and v <= 0:
|
||||||
|
raise ValueError(
|
||||||
|
"objective.rejection_warning_window_hours must be > 0"
|
||||||
|
)
|
||||||
|
return v
|
||||||
|
|
||||||
|
@field_validator("rejection_warning_baseline_hours")
|
||||||
|
@classmethod
|
||||||
|
def rejection_baseline_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||||
|
if v is not None and v <= 0:
|
||||||
|
raise ValueError(
|
||||||
|
"objective.rejection_warning_baseline_hours must be > 0"
|
||||||
|
)
|
||||||
|
return v
|
||||||
|
|
||||||
|
@field_validator("rejection_warning_min_count")
|
||||||
|
@classmethod
|
||||||
|
def rejection_min_count_positive(cls, v: Optional[int]) -> Optional[int]:
|
||||||
|
if v is not None and v <= 0:
|
||||||
|
raise ValueError(
|
||||||
|
"objective.rejection_warning_min_count must be > 0"
|
||||||
|
)
|
||||||
|
return v
|
||||||
|
|
||||||
|
|
||||||
class ContextOverride(StrictModel):
|
class ContextOverride(StrictModel):
|
||||||
"""Per-model context handling, for a row whose real limits are known.
|
"""Per-model context handling, for a row whose real limits are known.
|
||||||
|
|||||||
268
src/metrics.py
268
src/metrics.py
@@ -14,6 +14,9 @@ quota_burn — kWh metered in the last 30 d and in the current billing
|
|||||||
credit balance, burn rate and runway from
|
credit balance, burn rate and runway from
|
||||||
allowance_remaining_usd
|
allowance_remaining_usd
|
||||||
scoring_coverage — which scoring axes actually have data
|
scoring_coverage — which scoring axes actually have data
|
||||||
|
capability_ceilings — vision / json_mode context-window sub-ceilings
|
||||||
|
capability_demand_warnings — demand-relative warnings for those sub-ceilings
|
||||||
|
rejection_warnings — grouped rejection-rate / new-pattern warnings
|
||||||
recent_decisions — last N rows from the route_decisions observability table
|
recent_decisions — last N rows from the route_decisions observability table
|
||||||
per_model — per-model aggregates over energy_observations (last 30 d)
|
per_model — per-model aggregates over energy_observations (last 30 d)
|
||||||
verdict_mix — counts by verdict from verifications (last N days)
|
verdict_mix — counts by verdict from verifications (last N days)
|
||||||
@@ -23,9 +26,10 @@ pinch_summary — context-pruning savings over the last 30 d
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
import sqlite3
|
import sqlite3
|
||||||
from datetime import date, datetime, timedelta, timezone
|
from datetime import date, datetime, timedelta, timezone
|
||||||
from typing import Any, List, Optional
|
from typing import Any, Final, List, Optional
|
||||||
|
|
||||||
import routing
|
import routing
|
||||||
|
|
||||||
@@ -406,6 +410,10 @@ def scoring_coverage(
|
|||||||
f"tier {tier} has 0 eligible models for latency_tolerance={lat_tol}"
|
f"tier {tier} has 0 eligible models for latency_tolerance={lat_tol}"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
cap_ctx = capability_ceilings(conn, cfg)
|
||||||
|
warnings.extend(capability_demand_warnings(conn, cfg, cap_ctx))
|
||||||
|
warnings.extend(rejection_warnings(conn, cfg))
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"routable_models": total,
|
"routable_models": total,
|
||||||
"with_energy_data": total - len(missing_energy),
|
"with_energy_data": total - len(missing_energy),
|
||||||
@@ -568,6 +576,8 @@ def context_ceilings_with_rows(
|
|||||||
cfg: Any,
|
cfg: Any,
|
||||||
rows: List[dict],
|
rows: List[dict],
|
||||||
exclude_models: Optional[set[str]] = None,
|
exclude_models: Optional[set[str]] = None,
|
||||||
|
require_vision: bool = False,
|
||||||
|
require_json_mode: bool = False,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Like ``context_ceilings`` but over caller-supplied model rows.
|
"""Like ``context_ceilings`` but over caller-supplied model rows.
|
||||||
|
|
||||||
@@ -575,6 +585,9 @@ def context_ceilings_with_rows(
|
|||||||
proficiency and because the default ``exclude_models`` set is read from
|
proficiency and because the default ``exclude_models`` set is read from
|
||||||
``admin_model_overrides``. Callers that already know the exclusion set
|
``admin_model_overrides``. Callers that already know the exclusion set
|
||||||
(e.g. admin.py applying a just-committed override) may pass it explicitly.
|
(e.g. admin.py applying a just-committed override) may pass it explicitly.
|
||||||
|
``require_vision`` / ``require_json_mode`` gate the candidate set the
|
||||||
|
same way live routing gates it, so a capability-gated sub-ceiling goes
|
||||||
|
through this one path rather than a second copy of the filters.
|
||||||
"""
|
"""
|
||||||
if exclude_models is None:
|
if exclude_models is None:
|
||||||
exclude_models = _admin_deprecated_models(conn)
|
exclude_models = _admin_deprecated_models(conn)
|
||||||
@@ -600,8 +613,8 @@ def context_ceilings_with_rows(
|
|||||||
exclude_deprecated=True,
|
exclude_deprecated=True,
|
||||||
exclude_models=exclude_models,
|
exclude_models=exclude_models,
|
||||||
min_tool_proficiency=None,
|
min_tool_proficiency=None,
|
||||||
require_vision=False,
|
require_vision=require_vision,
|
||||||
require_json_mode=False,
|
require_json_mode=require_json_mode,
|
||||||
)
|
)
|
||||||
ceiling = max((r["effective_context_window"] for r in candidates), default=0)
|
ceiling = max((r["effective_context_window"] for r in candidates), default=0)
|
||||||
result[(tier, latency_tolerance)] = {
|
result[(tier, latency_tolerance)] = {
|
||||||
@@ -611,6 +624,255 @@ def context_ceilings_with_rows(
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def capability_ceilings(
|
||||||
|
conn: sqlite3.Connection,
|
||||||
|
cfg: Any,
|
||||||
|
) -> dict:
|
||||||
|
"""Context-window ceiling per capability-gated subset, by tier.
|
||||||
|
|
||||||
|
The plain ``(tier, latency_tolerance)`` buckets from
|
||||||
|
``context_ceilings`` are blind to capability gates: during the
|
||||||
|
2026-09-04 incident the tier-1 interactive ceiling stayed at 782,324
|
||||||
|
while the vision-capable subset had collapsed to 192,500 (the only
|
||||||
|
vision rows large enough had been deprecated through admin overrides),
|
||||||
|
so every existing demand check was silent on requests that 422ed.
|
||||||
|
Vision and JSON mode are the two flags that can independently empty
|
||||||
|
the candidate set — both fail CLOSED on an unknown catalog flag, which
|
||||||
|
is what lets them shrink the set sharply — so each gets its own series
|
||||||
|
here. Deliberately NOT one bucket per capability combination; that is
|
||||||
|
a combinatorial explosion over dimensions that do not interact.
|
||||||
|
|
||||||
|
Delegates to ``context_ceilings_with_rows`` once per dimension, so every
|
||||||
|
sub-ceiling comes from the same ``routing.select_candidates`` path
|
||||||
|
live routing uses, admin deprecations included.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
``{"vision": {(tier, latency_tolerance): {"ceiling", "count"}}, "json_mode":
|
||||||
|
{...}}``, one sub-dict per capability dimension, each keyed by the
|
||||||
|
``(tier, latency_tolerance)`` pair present in the DB.
|
||||||
|
"""
|
||||||
|
rows = [dict(r) for r in conn.execute("SELECT * FROM models")]
|
||||||
|
exclude_models = _admin_deprecated_models(conn)
|
||||||
|
result: dict[str, dict[tuple, dict[str, int]]] = {}
|
||||||
|
for dimension in ("vision", "json_mode"):
|
||||||
|
sub_ceilings = context_ceilings_with_rows(
|
||||||
|
conn,
|
||||||
|
cfg,
|
||||||
|
rows,
|
||||||
|
exclude_models=exclude_models,
|
||||||
|
require_vision=dimension == "vision",
|
||||||
|
require_json_mode=dimension == "json_mode",
|
||||||
|
)
|
||||||
|
result[dimension] = sub_ceilings
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def capability_demand_warnings(
|
||||||
|
conn: sqlite3.Connection,
|
||||||
|
cfg: Any,
|
||||||
|
cap_ctx: dict,
|
||||||
|
) -> List[str]:
|
||||||
|
"""Demand-relative warnings for the capability-gated sub-ceilings.
|
||||||
|
|
||||||
|
Mirrors ``demand_ceiling_warnings`` for each dimension — 7d MAX probe
|
||||||
|
first with a 30d fallback, silence for a bucket no such request ever
|
||||||
|
hit, else window = 7 when the 7d row count is >= 10 (30 otherwise),
|
||||||
|
warning only when that window's observed max exceeds the sub-ceiling.
|
||||||
|
The one difference is what "demand" means: request rows must carry the
|
||||||
|
capability (``images`` / ``json_mode`` columns on route_decisions), so
|
||||||
|
a vision sub-ceiling is compared only against requests that actually
|
||||||
|
carry images, for the ``(tier, "interactive")`` bucket. ``cfg`` is
|
||||||
|
accepted for signature symmetry with ``demand_ceiling_warnings``; this
|
||||||
|
detector has no escalation counterpart to read from it.
|
||||||
|
|
||||||
|
The demand read is trailing-window, so a warning stays alive until a
|
||||||
|
big historical request ages out of it; the ``window: last {n}d``
|
||||||
|
annotation in the message is what keeps that self-explaining.
|
||||||
|
"""
|
||||||
|
warnings: List[str] = []
|
||||||
|
|
||||||
|
dimension_columns = {
|
||||||
|
"vision": "images",
|
||||||
|
"json_mode": "json_mode",
|
||||||
|
}
|
||||||
|
capability_nouns = {
|
||||||
|
"vision": "for requests carrying images",
|
||||||
|
"json_mode": "for requests requesting JSON mode",
|
||||||
|
}
|
||||||
|
|
||||||
|
for dimension in ("vision", "json_mode"):
|
||||||
|
buckets = cap_ctx.get(dimension) or {}
|
||||||
|
for tier in sorted({t for (t, _) in buckets}):
|
||||||
|
bucket = buckets.get((tier, "interactive"))
|
||||||
|
if bucket is None:
|
||||||
|
continue
|
||||||
|
column = dimension_columns.get(dimension)
|
||||||
|
if column is None:
|
||||||
|
continue
|
||||||
|
row = conn.execute(
|
||||||
|
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
|
||||||
|
" WHERE julianday(observed_at) > julianday('now', '-7 days')"
|
||||||
|
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
|
||||||
|
(tier,),
|
||||||
|
).fetchone()
|
||||||
|
mx_7d = row["mx"] if row else None
|
||||||
|
if mx_7d is None:
|
||||||
|
row = conn.execute(
|
||||||
|
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
|
||||||
|
" WHERE julianday(observed_at) > julianday('now', '-30 days')"
|
||||||
|
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
|
||||||
|
(tier,),
|
||||||
|
).fetchone()
|
||||||
|
mx_7d = row["mx"] if row else None
|
||||||
|
if mx_7d is None:
|
||||||
|
# No request carrying this capability was ever routed for this
|
||||||
|
# tier — an unexercised sub-ceiling warns about nothing.
|
||||||
|
continue
|
||||||
|
row_cnt = conn.execute(
|
||||||
|
"SELECT COUNT(*) AS n FROM route_decisions"
|
||||||
|
" WHERE julianday(observed_at) > julianday('now', '-7 days')"
|
||||||
|
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
|
||||||
|
(tier,),
|
||||||
|
).fetchone()["n"]
|
||||||
|
window = 7 if row_cnt >= 10 else 30
|
||||||
|
row = conn.execute(
|
||||||
|
"SELECT MAX(required_context_tokens) AS mx FROM route_decisions"
|
||||||
|
f" WHERE julianday(observed_at) > julianday('now', '-{window} days')"
|
||||||
|
f" AND kind = 'chat' AND task_tier = ? AND {column} = 1",
|
||||||
|
(tier,),
|
||||||
|
).fetchone()
|
||||||
|
observed_max = row["mx"] if row else 0
|
||||||
|
ceiling = bucket["ceiling"]
|
||||||
|
if observed_max > ceiling:
|
||||||
|
warnings.append(
|
||||||
|
f"{dimension}-capable tier {tier} context ceiling "
|
||||||
|
f"({ceiling}) is below observed max demand "
|
||||||
|
f"({observed_max}, window: last {window}d) "
|
||||||
|
f"{capability_nouns[dimension]} — requests above this may "
|
||||||
|
"return 422"
|
||||||
|
)
|
||||||
|
|
||||||
|
return warnings
|
||||||
|
|
||||||
|
|
||||||
|
# A novel (tier, normalized_reason) group must reach this count inside the
|
||||||
|
# alert window before warning. The 2026-09-04 incident produced exactly 2
|
||||||
|
# image rejections, so the floor must be <= 2; 1 would fire on any one-off
|
||||||
|
# genuinely-impossible request that correctly 422s.
|
||||||
|
NOVEL_GROUP_MIN_COUNT: Final = 2
|
||||||
|
|
||||||
|
# How many rejections a group needs inside the alert window before
|
||||||
|
# "new rejection pattern:" is reported. A single rejection is
|
||||||
|
# indistinguishable from a genuinely impossible request, which should 422 —
|
||||||
|
# the signal is a rate, not the existence of a rejection.
|
||||||
|
#
|
||||||
|
# Caveat: "novel" means ZERO occurrences in the prior baseline window, so
|
||||||
|
# a familiar-but-intermittent pattern can read as novel after a quiet gap
|
||||||
|
# and fire one spurious "new rejection pattern:" warning. It is
|
||||||
|
# self-clearing as the pattern's own rejections age into the baseline
|
||||||
|
# window; widen objective.rejection_warning_baseline_hours to suppress it.
|
||||||
|
|
||||||
|
|
||||||
|
def rejection_warnings(
|
||||||
|
conn: sqlite3.Connection,
|
||||||
|
cfg: Any,
|
||||||
|
) -> List[str]:
|
||||||
|
"""Reactive safety net over the rejections the router already logged.
|
||||||
|
|
||||||
|
The ceiling checks above only catch dimensions someone modelled; this
|
||||||
|
one catches everything, at the cost of firing after the first failures
|
||||||
|
rather than before. Rejections (``selected_model IS NULL AND
|
||||||
|
rejected_reason IS NOT NULL``) are pulled once across the alert plus
|
||||||
|
baseline window and split by exact age — alert bucket at
|
||||||
|
``rejection_warning_window_hours`` or younger, baseline bucket older
|
||||||
|
than the alert window but within ``rejection_warning_baseline_hours``
|
||||||
|
of it — then grouped by *normalized* reason plus ``task_tier``.
|
||||||
|
|
||||||
|
Two signals, one line per group:
|
||||||
|
|
||||||
|
* ``rejection rate:`` — the group reached
|
||||||
|
``objective.rejection_warning_min_count`` rejections in the alert
|
||||||
|
window.
|
||||||
|
* ``new rejection pattern:`` — the group is ABSENT from the baseline
|
||||||
|
window and reached ``NOVEL_GROUP_MIN_COUNT`` in the alert window.
|
||||||
|
This prefix wins when both would fire.
|
||||||
|
|
||||||
|
Zero alert-window rejections produce nothing at all: a rejection is
|
||||||
|
not inherently an error, and a genuinely impossible request should
|
||||||
|
422 — the signal is a rate.
|
||||||
|
"""
|
||||||
|
window = getattr(cfg.objective, "rejection_warning_window_hours", None)
|
||||||
|
if window is None:
|
||||||
|
window = 1
|
||||||
|
baseline = getattr(cfg.objective, "rejection_warning_baseline_hours", None)
|
||||||
|
if baseline is None:
|
||||||
|
baseline = 24
|
||||||
|
min_count = getattr(cfg.objective, "rejection_warning_min_count", None)
|
||||||
|
if min_count is None:
|
||||||
|
min_count = 6
|
||||||
|
|
||||||
|
now = datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
rows = conn.execute(
|
||||||
|
"""
|
||||||
|
SELECT observed_at, task_tier, rejected_reason
|
||||||
|
FROM route_decisions
|
||||||
|
WHERE selected_model IS NULL
|
||||||
|
AND rejected_reason IS NOT NULL
|
||||||
|
AND julianday(observed_at) > julianday('now', '-' || ? || ' hours')
|
||||||
|
ORDER BY observed_at DESC
|
||||||
|
""",
|
||||||
|
(str(window + baseline),),
|
||||||
|
).fetchall()
|
||||||
|
|
||||||
|
alert_groups: dict[tuple[Optional[int], str], dict] = {}
|
||||||
|
baseline_keys: set[tuple[Optional[int], str]] = set()
|
||||||
|
for row in rows:
|
||||||
|
try:
|
||||||
|
observed = datetime.fromisoformat(row["observed_at"])
|
||||||
|
except ValueError:
|
||||||
|
# Defensive, same shape as quota_balance_and_burn: one malformed
|
||||||
|
# timestamp must not take the whole report down.
|
||||||
|
continue
|
||||||
|
# Group on digit-normalized reasons: the raw strings embed the
|
||||||
|
# request's own numbers ("context >= 242486 tokens"), so literal
|
||||||
|
# grouping yields N groups of one and hides the pattern entirely.
|
||||||
|
key = (row["task_tier"], re.sub(r"\d+", "N", row["rejected_reason"]))
|
||||||
|
age_hours = (now - observed).total_seconds() / 3600.0
|
||||||
|
if age_hours <= window:
|
||||||
|
group = alert_groups.get(key)
|
||||||
|
if group is None:
|
||||||
|
# Rows arrive newest-first, so the first touch per group is
|
||||||
|
# its most recent occurrence.
|
||||||
|
alert_groups[key] = {"count": 1, "most_recent": observed}
|
||||||
|
else:
|
||||||
|
group["count"] += 1
|
||||||
|
elif age_hours <= window + baseline:
|
||||||
|
baseline_keys.add(key)
|
||||||
|
|
||||||
|
if not alert_groups:
|
||||||
|
return []
|
||||||
|
|
||||||
|
warnings: List[str] = []
|
||||||
|
for (tier, normalized_reason), group in alert_groups.items():
|
||||||
|
count = group["count"]
|
||||||
|
is_novel = (tier, normalized_reason) not in baseline_keys
|
||||||
|
prefix: Optional[str] = None
|
||||||
|
if is_novel and count >= NOVEL_GROUP_MIN_COUNT:
|
||||||
|
prefix = "new rejection pattern:"
|
||||||
|
elif count >= min_count:
|
||||||
|
prefix = "rejection rate:"
|
||||||
|
if prefix is None:
|
||||||
|
continue
|
||||||
|
tier_label = "unknown" if tier is None else str(tier)
|
||||||
|
warnings.append(
|
||||||
|
f"{prefix} {count} rejection(s) in the last {window}h: "
|
||||||
|
f"{normalized_reason} (tier {tier_label}; "
|
||||||
|
f"most recent: {group['most_recent'].isoformat()})"
|
||||||
|
)
|
||||||
|
return warnings
|
||||||
|
|
||||||
|
|
||||||
def recent_decisions(
|
def recent_decisions(
|
||||||
conn: sqlite3.Connection,
|
conn: sqlite3.Connection,
|
||||||
limit: int = 50,
|
limit: int = 50,
|
||||||
|
|||||||
@@ -30,12 +30,16 @@ import dispatcher
|
|||||||
from config import load_config
|
from config import load_config
|
||||||
from metrics import (
|
from metrics import (
|
||||||
_next_reset_date,
|
_next_reset_date,
|
||||||
|
capability_ceilings,
|
||||||
|
capability_demand_warnings,
|
||||||
context_ceilings,
|
context_ceilings,
|
||||||
|
demand_ceiling_warnings,
|
||||||
local_energy_summary,
|
local_energy_summary,
|
||||||
per_model,
|
per_model,
|
||||||
pinch_summary,
|
pinch_summary,
|
||||||
quota_burn,
|
quota_burn,
|
||||||
recent_decisions,
|
recent_decisions,
|
||||||
|
rejection_warnings,
|
||||||
scoring_coverage,
|
scoring_coverage,
|
||||||
top_proficiency,
|
top_proficiency,
|
||||||
verdict_mix,
|
verdict_mix,
|
||||||
@@ -794,6 +798,482 @@ def test_context_ceilings_excludes_admin_deprecated_models(tmp_path):
|
|||||||
assert ctx[key]["count"] == 1
|
assert ctx[key]["count"] == 1
|
||||||
|
|
||||||
|
|
||||||
|
# --- capability sub-ceiling demand warnings (2026-09-04 incident) ---------------
|
||||||
|
|
||||||
|
|
||||||
|
def _insert_model_row(
|
||||||
|
conn: sqlite3.Connection,
|
||||||
|
*,
|
||||||
|
model_id: str,
|
||||||
|
tier: int,
|
||||||
|
effective_context_window: int,
|
||||||
|
supports_vision: int,
|
||||||
|
supports_json_mode: int = 1,
|
||||||
|
) -> None:
|
||||||
|
"""Insert one active, routable catalog row with explicit capability flags.
|
||||||
|
|
||||||
|
``_seed_models`` covers the common case but hardcodes its capability
|
||||||
|
flags and effective window; the incident-reproduction tests need both
|
||||||
|
per-row.
|
||||||
|
"""
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
INSERT INTO models (
|
||||||
|
model_id, provider, base_model_id, tier, context_window,
|
||||||
|
effective_context_window, max_output_tokens,
|
||||||
|
cost_per_1m_prompt, cost_per_1m_completion,
|
||||||
|
supports_vision, supports_json_mode,
|
||||||
|
latency_class, reasoning_mode, context_variant,
|
||||||
|
access_level, availability, last_updated
|
||||||
|
) VALUES (?, 'neuralwatt', ?, ?, ?, ?, 16384, 0.30, 0.10,
|
||||||
|
?, ?, 'standard', 'default', 'full', 'public', 'active',
|
||||||
|
?)
|
||||||
|
""",
|
||||||
|
(
|
||||||
|
model_id,
|
||||||
|
model_id,
|
||||||
|
tier,
|
||||||
|
effective_context_window,
|
||||||
|
effective_context_window,
|
||||||
|
supports_vision,
|
||||||
|
supports_json_mode,
|
||||||
|
_now().isoformat(),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def _deprecate_model(conn: sqlite3.Connection, model_id: str) -> None:
|
||||||
|
"""Insert an admin override marking *model_id* deprecated (reason: cost)."""
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO admin_model_overrides "
|
||||||
|
"(model_id, provider, availability, reason, updated_at) "
|
||||||
|
"VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)",
|
||||||
|
(model_id, _now().isoformat()),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def _insert_rejected_decision(
|
||||||
|
conn: sqlite3.Connection,
|
||||||
|
*,
|
||||||
|
tier: int,
|
||||||
|
reason: str,
|
||||||
|
observed_at: str,
|
||||||
|
tokens: int = 200000,
|
||||||
|
images: int = 0,
|
||||||
|
json_mode: int = 0,
|
||||||
|
) -> None:
|
||||||
|
"""Insert a route_decisions row that selected no model (a 422 rejection).
|
||||||
|
|
||||||
|
``kind='chat'`` so the demand queries (which filter on kind) see it;
|
||||||
|
``latency_tolerance='interactive'`` to match the tier-1 bucket key the
|
||||||
|
ceiling checks use.
|
||||||
|
"""
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
INSERT INTO route_decisions (
|
||||||
|
observed_at, kind, task_category, task_tier,
|
||||||
|
required_context_tokens, confidence, classifier_ms,
|
||||||
|
classification_source, latency_tolerance, candidates_considered,
|
||||||
|
selected_model, selected_provider, rejected_reason,
|
||||||
|
session_key, tools, images, json_mode, streamed
|
||||||
|
) VALUES (?, 'chat', 'coding_general', ?, ?, NULL, NULL,
|
||||||
|
'override', 'interactive', 0, NULL, NULL, ?,
|
||||||
|
'sess', 0, ?, ?, 0)
|
||||||
|
""",
|
||||||
|
(observed_at, tier, tokens, reason, images, json_mode),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def _seed_incident_state(conn: sqlite3.Connection) -> str:
|
||||||
|
"""Seed the 2026-09-04 incident model state; returns the shared 'now' stamp.
|
||||||
|
|
||||||
|
Three tier-1 rows: kimi-k3 and kimi-k3-fast — the only vision-capable
|
||||||
|
large-context models — deprecated through admin overrides, and
|
||||||
|
deepseek-v4-flash (no vision, 262,128 effective tokens) left active.
|
||||||
|
The overall tier-1 interactive ceiling stays high while the vision
|
||||||
|
sub-ceiling collapses to zero, which is exactly the shape every
|
||||||
|
pre-2026-09-05 ceiling check was blind to.
|
||||||
|
"""
|
||||||
|
for model_id, vision, eff_ctx in (
|
||||||
|
("kimi-k3", 1, 782324),
|
||||||
|
("kimi-k3-fast", 1, 782324),
|
||||||
|
("deepseek-v4-flash", 0, 262128),
|
||||||
|
):
|
||||||
|
_insert_model_row(
|
||||||
|
conn,
|
||||||
|
model_id=model_id,
|
||||||
|
tier=1,
|
||||||
|
effective_context_window=eff_ctx,
|
||||||
|
supports_vision=vision,
|
||||||
|
)
|
||||||
|
for model_id in ("kimi-k3", "kimi-k3-fast"):
|
||||||
|
_deprecate_model(conn, model_id)
|
||||||
|
return _now().isoformat()
|
||||||
|
|
||||||
|
|
||||||
|
def test_capability_demand_warning_incident_reproduction(tmp_path):
|
||||||
|
"""The 2026-09-04 incident state warns on the vision sub-ceiling while
|
||||||
|
every existing ``(tier, latency_tolerance)`` check stays silent.
|
||||||
|
|
||||||
|
In the incident, kimi-k3 and kimi-k3-fast — the only vision-capable
|
||||||
|
tier-1 models with a large context — were deprecated through admin
|
||||||
|
overrides, so a 200,000-token image request was unservable even though
|
||||||
|
the overall tier-1 ceiling (262,128, from non-vision deepseek-v4-flash)
|
||||||
|
comfortably covered that demand. The new capability check must fire;
|
||||||
|
the existing demand_ceiling_warnings must NOT — that silence is what
|
||||||
|
hid the incident for 19 hours.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
||||||
|
ts = _seed_incident_state(conn)
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=1,
|
||||||
|
reason=(
|
||||||
|
"context >= 242486 tokens; "
|
||||||
|
"vision-capable model (request carries image(s))"
|
||||||
|
),
|
||||||
|
observed_at=ts,
|
||||||
|
tokens=200000,
|
||||||
|
images=1,
|
||||||
|
)
|
||||||
|
|
||||||
|
# The overall tier-1 interactive ceiling is untouched by the deprecations.
|
||||||
|
ctx = context_ceilings(conn, CFG)
|
||||||
|
assert ctx[(1, "interactive")]["ceiling"] == 262128
|
||||||
|
|
||||||
|
# The vision-gated subset collapsed: no candidate survives the gate.
|
||||||
|
cap_ctx = capability_ceilings(conn, CFG)
|
||||||
|
assert cap_ctx["vision"][(1, "interactive")]["ceiling"] == 0
|
||||||
|
assert cap_ctx["vision"][(1, "interactive")]["count"] == 0
|
||||||
|
# deepseek-v4-flash still serves the json_mode subset.
|
||||||
|
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 262128
|
||||||
|
|
||||||
|
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
|
||||||
|
assert len(warnings) == 1
|
||||||
|
w = warnings[0]
|
||||||
|
assert "vision" in w
|
||||||
|
assert "tier 1" in w
|
||||||
|
assert "ceiling (0)" in w
|
||||||
|
assert "200000" in w
|
||||||
|
|
||||||
|
# The pre-existing check must stay silent on the same state: the overall
|
||||||
|
# tier-1 ceiling (262128) still exceeds the observed max demand (200000).
|
||||||
|
assert demand_ceiling_warnings(conn, CFG, ctx, [1]) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_capability_demand_no_demand_no_warning(tmp_path):
|
||||||
|
"""A state with demand but no capability-flagged demand warns about nothing.
|
||||||
|
|
||||||
|
Requests exist at the tier — far above every ceiling — but none carry
|
||||||
|
images or ask for JSON mode, so both capability subsets are unexercised:
|
||||||
|
an unexercised sub-ceiling is not a warning.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=2,
|
||||||
|
reason="context >= 999999 tokens",
|
||||||
|
observed_at=_now().isoformat(),
|
||||||
|
tokens=999999,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert capability_demand_warnings(conn, CFG, capability_ceilings(conn, CFG)) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_capability_json_mode_demand_warning(tmp_path):
|
||||||
|
"""Deprecating the only json_mode-capable model trips the json_mode warning.
|
||||||
|
|
||||||
|
Same incident shape as the vision case, different capability dimension:
|
||||||
|
the json_mode sub-ceiling collapses to zero while the overall ceiling
|
||||||
|
stays fine, and a json_mode request that cannot be served produces a
|
||||||
|
warning naming json_mode.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
||||||
|
_insert_model_row(
|
||||||
|
conn,
|
||||||
|
model_id="jm-big",
|
||||||
|
tier=1,
|
||||||
|
effective_context_window=782324,
|
||||||
|
supports_vision=0,
|
||||||
|
supports_json_mode=1,
|
||||||
|
)
|
||||||
|
_insert_model_row(
|
||||||
|
conn,
|
||||||
|
model_id="plain-fallback",
|
||||||
|
tier=1,
|
||||||
|
effective_context_window=262128,
|
||||||
|
supports_vision=0,
|
||||||
|
supports_json_mode=0,
|
||||||
|
)
|
||||||
|
_deprecate_model(conn, "jm-big")
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=1,
|
||||||
|
reason=(
|
||||||
|
"context >= 242486 tokens; json-mode-capable model "
|
||||||
|
"(json_mode requested)"
|
||||||
|
),
|
||||||
|
observed_at=_now().isoformat(),
|
||||||
|
tokens=200000,
|
||||||
|
json_mode=1,
|
||||||
|
)
|
||||||
|
|
||||||
|
cap_ctx = capability_ceilings(conn, CFG)
|
||||||
|
assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 0
|
||||||
|
|
||||||
|
warnings = capability_demand_warnings(conn, CFG, cap_ctx)
|
||||||
|
assert len(warnings) == 1
|
||||||
|
w = warnings[0]
|
||||||
|
assert "json_mode" in w
|
||||||
|
assert "tier 1" in w
|
||||||
|
assert "ceiling (0)" in w
|
||||||
|
assert "200000" in w
|
||||||
|
|
||||||
|
|
||||||
|
# --- rejection warnings (reactive rejection detector) --------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_rejection_warnings_zero_rejections(tmp_path):
|
||||||
|
"""Zero NULL-selection rows produce no rejection warnings.
|
||||||
|
|
||||||
|
The absence of rejections is not a signal worth reporting — the
|
||||||
|
detector watches rates and new patterns, not mere existence.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
assert rejection_warnings(conn, CFG) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_rejection_warnings_novel_group_at_incident_volume_fires(tmp_path):
|
||||||
|
"""The 2026-09-04 incident shape — a novel group of only 2 — must fire.
|
||||||
|
|
||||||
|
The incident produced exactly 2 image rejections, below the routine
|
||||||
|
noise floor (about 3/hour on this deployment), so no rate threshold can
|
||||||
|
catch it; the novelty prong — absent from the 24h baseline, count
|
||||||
|
>= 2 — is what must fire here. Both rows share one timestamp so the
|
||||||
|
rendered most-recent value is deterministic.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
ts = _now().isoformat()
|
||||||
|
for reason in (
|
||||||
|
"tier >= 1; context >= 242486 tokens; interactive; "
|
||||||
|
"vision-capable model (request carries image(s))",
|
||||||
|
"tier >= 1; context >= 195000 tokens; interactive; "
|
||||||
|
"vision-capable model (request carries image(s))",
|
||||||
|
):
|
||||||
|
_insert_rejected_decision(conn, tier=1, reason=reason, observed_at=ts)
|
||||||
|
|
||||||
|
warnings = rejection_warnings(conn, CFG)
|
||||||
|
assert len(warnings) == 1
|
||||||
|
w = warnings[0]
|
||||||
|
assert w.startswith("new rejection pattern:")
|
||||||
|
# Token counts collapse: both raw reasons normalize to "context >= N tokens".
|
||||||
|
assert "2 rejection(s)" in w
|
||||||
|
assert "context >= N tokens" in w
|
||||||
|
# The tier is rendered from the structured task_tier column, not the string.
|
||||||
|
assert "(tier 1;" in w
|
||||||
|
assert ts in w
|
||||||
|
|
||||||
|
|
||||||
|
def test_rejection_warnings_familiar_group_below_threshold_silent(tmp_path):
|
||||||
|
"""Routine over-large tier-3 traffic — 3/hour and familiar — stays silent.
|
||||||
|
|
||||||
|
Regression for the measured routine floor: this deployment routinely
|
||||||
|
sees about 3 rejections per hour of ordinary over-large tier-3 requests
|
||||||
|
that behave as designed. The group is familiar (present in the 24h
|
||||||
|
baseline window) and below ``rejection_warning_min_count`` (6), so
|
||||||
|
neither the novelty prong nor the rate prong fires.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
now = _now()
|
||||||
|
# Baseline occurrence: 2h ago — inside the 24h baseline, outside the 1h
|
||||||
|
# alert window.
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=3,
|
||||||
|
reason="tier >= 3; context >= 40000 tokens; interactive",
|
||||||
|
observed_at=(now - timedelta(hours=2)).isoformat(),
|
||||||
|
)
|
||||||
|
# The live-DB-measured routine hourly volume, with distinct token counts
|
||||||
|
# so digit normalization does real grouping work.
|
||||||
|
for tokens in (30000, 31000, 32000):
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=3,
|
||||||
|
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
|
||||||
|
observed_at=now.isoformat(),
|
||||||
|
)
|
||||||
|
|
||||||
|
assert rejection_warnings(conn, CFG) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_rejection_warnings_familiar_group_at_threshold_fires(tmp_path):
|
||||||
|
"""A familiar group at ``rejection_warning_min_count`` (6) fires as a rate."""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
now = _now()
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=3,
|
||||||
|
reason="tier >= 3; context >= 40000 tokens; interactive",
|
||||||
|
observed_at=(now - timedelta(hours=2)).isoformat(),
|
||||||
|
)
|
||||||
|
for tokens in (30000, 31000, 32000, 33000, 34000, 35000):
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=3,
|
||||||
|
reason=f"tier >= 3; context >= {tokens} tokens; interactive",
|
||||||
|
observed_at=now.isoformat(),
|
||||||
|
)
|
||||||
|
|
||||||
|
warnings = rejection_warnings(conn, CFG)
|
||||||
|
assert len(warnings) == 1
|
||||||
|
w = warnings[0]
|
||||||
|
assert w.startswith("rejection rate:")
|
||||||
|
assert "6 rejection(s)" in w
|
||||||
|
|
||||||
|
|
||||||
|
def test_rejection_warnings_novel_single_rejection_silent(tmp_path):
|
||||||
|
"""A single novel rejection stays silent.
|
||||||
|
|
||||||
|
A genuinely impossible one-off request SHOULD 422; presence of one
|
||||||
|
rejection is not a signal. n=1 is below NOVEL_GROUP_MIN_COUNT (2).
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=1,
|
||||||
|
reason=(
|
||||||
|
"tier >= 1; context >= 242486 tokens; interactive; "
|
||||||
|
"vision-capable model (request carries image(s))"
|
||||||
|
),
|
||||||
|
observed_at=_now().isoformat(),
|
||||||
|
)
|
||||||
|
|
||||||
|
assert rejection_warnings(conn, CFG) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_rejection_warnings_groups_by_structured_tier(tmp_path):
|
||||||
|
"""Tier-1 and tier-3 rejections of the same normalized shape never merge.
|
||||||
|
|
||||||
|
Regression for the sibling finding of the incident review: digit
|
||||||
|
normalization erases the tier embedded in the reason string ("tier >= 1"
|
||||||
|
and "tier >= 3" both become "tier >= N"), so grouping by normalized
|
||||||
|
string alone would produce ONE merged count-4 group, hiding which
|
||||||
|
tier's candidate set went empty. The discriminator is the structured
|
||||||
|
``task_tier`` column, rendered back into the body as "(tier N; ...)".
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
ts = _now().isoformat()
|
||||||
|
for tier, tokens in ((1, 242486), (1, 50000), (3, 229702), (3, 30000)):
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=tier,
|
||||||
|
reason=f"tier >= {tier}; context >= {tokens} tokens; interactive",
|
||||||
|
observed_at=ts,
|
||||||
|
)
|
||||||
|
|
||||||
|
warnings = rejection_warnings(conn, CFG)
|
||||||
|
assert len(warnings) == 2
|
||||||
|
tier1 = [w for w in warnings if "(tier 1;" in w]
|
||||||
|
tier3 = [w for w in warnings if "(tier 3;" in w]
|
||||||
|
assert len(tier1) == 1
|
||||||
|
assert len(tier3) == 1
|
||||||
|
assert "2 rejection(s)" in tier1[0]
|
||||||
|
assert "2 rejection(s)" in tier3[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_rejection_warnings_old_rejections_not_counted(tmp_path):
|
||||||
|
"""A rejection outside both windows is not counted at all.
|
||||||
|
|
||||||
|
Two same-shape rows 3 days old reach the novelty floor of 2, so the
|
||||||
|
only thing that keeps them silent is their age — outside the 1h alert
|
||||||
|
window and beyond the 25h total pull window they are not even read.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
for tokens in (242486, 195000):
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=1,
|
||||||
|
reason=f"tier >= 1; context >= {tokens} tokens; interactive",
|
||||||
|
observed_at=(_now() - timedelta(days=3)).isoformat(),
|
||||||
|
)
|
||||||
|
|
||||||
|
assert rejection_warnings(conn, CFG) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_rejection_warnings_custom_window(tmp_path):
|
||||||
|
"""``rejection_warning_window_hours`` widens the alert window.
|
||||||
|
|
||||||
|
Two same-shape rows 2h old sit outside the default 1h window but
|
||||||
|
inside a 3h window; with the knob set they form a novel group at the
|
||||||
|
incident volume and must fire.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path)
|
||||||
|
_seed_models(conn)
|
||||||
|
two_hours_ago = (_now() - timedelta(hours=2)).isoformat()
|
||||||
|
for tokens in (242486, 195000):
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=1,
|
||||||
|
reason=(
|
||||||
|
f"tier >= 1; context >= {tokens} tokens; interactive; "
|
||||||
|
"vision-capable model (request carries image(s))"
|
||||||
|
),
|
||||||
|
observed_at=two_hours_ago,
|
||||||
|
)
|
||||||
|
|
||||||
|
wide_cfg = SimpleNamespace(
|
||||||
|
objective=SimpleNamespace(
|
||||||
|
rejection_warning_window_hours=3,
|
||||||
|
rejection_warning_baseline_hours=24,
|
||||||
|
rejection_warning_min_count=6,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
warnings = rejection_warnings(conn, wide_cfg)
|
||||||
|
assert len(warnings) == 1
|
||||||
|
assert warnings[0].startswith("new rejection pattern:")
|
||||||
|
assert "2 rejection(s)" in warnings[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_scoring_coverage_includes_new_warnings(tmp_path):
|
||||||
|
"""``scoring_coverage`` surfaces both new warning classes on the incident state.
|
||||||
|
|
||||||
|
The 2026-09-04 incident produced two image rejections against a catalog
|
||||||
|
whose vision sub-ceiling had collapsed. Both the capability demand
|
||||||
|
warning and the novel rejection pattern must reach the same warnings
|
||||||
|
list that /metrics, the admin portal and the TUI read.
|
||||||
|
"""
|
||||||
|
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
||||||
|
ts = _seed_incident_state(conn)
|
||||||
|
for tokens in (242486, 195000):
|
||||||
|
_insert_rejected_decision(
|
||||||
|
conn,
|
||||||
|
tier=1,
|
||||||
|
reason=(
|
||||||
|
f"context >= {tokens} tokens; "
|
||||||
|
"vision-capable model (request carries image(s))"
|
||||||
|
),
|
||||||
|
observed_at=ts,
|
||||||
|
images=1,
|
||||||
|
)
|
||||||
|
|
||||||
|
result = scoring_coverage(conn, CFG)
|
||||||
|
warnings = result["warnings"]
|
||||||
|
assert any("vision" in w for w in warnings)
|
||||||
|
assert any(w.startswith("new rejection pattern:") for w in warnings)
|
||||||
|
|
||||||
|
|
||||||
# --- recent_decisions tests ---------------------------------------------------
|
# --- recent_decisions tests ---------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -31,10 +31,10 @@ def _now() -> datetime:
|
|||||||
return datetime.now(timezone.utc)
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
def _make_db(tmp_path: Path) -> sqlite3.Connection:
|
def _make_db(tmp_path: Path, extra_sql: str = "") -> sqlite3.Connection:
|
||||||
conn = sqlite3.connect(str(tmp_path / "test.db"))
|
conn = sqlite3.connect(str(tmp_path / "test.db"))
|
||||||
conn.row_factory = sqlite3.Row
|
conn.row_factory = sqlite3.Row
|
||||||
conn.executescript(SCHEMA_SQL)
|
conn.executescript(SCHEMA_SQL + extra_sql)
|
||||||
return conn
|
return conn
|
||||||
|
|
||||||
|
|
||||||
@@ -358,6 +358,102 @@ def test_metrics_empty_db_returns_200(monkeypatch, tmp_path):
|
|||||||
assert "generated_at" in data
|
assert "generated_at" in data
|
||||||
|
|
||||||
|
|
||||||
|
# admin_model_overrides is NOT in schema.sql — dispatcher creates it at
|
||||||
|
# startup via ensure_route_decisions-style migration, so endpoint tests that
|
||||||
|
# need it build it with extra_sql (pattern from tests/test_metrics.py).
|
||||||
|
_ADMIN_TABLE_SQL = """
|
||||||
|
CREATE TABLE IF NOT EXISTS admin_model_overrides (
|
||||||
|
model_id TEXT NOT NULL,
|
||||||
|
provider TEXT NOT NULL,
|
||||||
|
availability TEXT NOT NULL,
|
||||||
|
reason TEXT,
|
||||||
|
updated_at TEXT NOT NULL,
|
||||||
|
PRIMARY KEY (model_id, provider)
|
||||||
|
);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_admin_model_overrides_availability
|
||||||
|
ON admin_model_overrides (availability);
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def test_metrics_coverage_warnings_include_rejection_and_capability(
|
||||||
|
tmp_path, monkeypatch
|
||||||
|
):
|
||||||
|
"""GET /metrics surfaces the new rejection and capability warnings.
|
||||||
|
|
||||||
|
Seeds the 2026-09-04 incident state — kimi-k3 and kimi-k3-fast (the only
|
||||||
|
vision-capable tier-1 models) deprecated via admin overrides, plus two
|
||||||
|
image rejections — into the endpoint's temp DB, and asserts the warnings
|
||||||
|
reach the same coverage.warnings list the admin bell and TUI render.
|
||||||
|
The rejection warning is the deterministic assertion; on this exact seed
|
||||||
|
the capability warning fires too.
|
||||||
|
"""
|
||||||
|
now = _now().isoformat()
|
||||||
|
conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL)
|
||||||
|
for model_id, vision, eff_ctx in (
|
||||||
|
("kimi-k3", 1, 782324),
|
||||||
|
("kimi-k3-fast", 1, 782324),
|
||||||
|
("deepseek-v4-flash", 0, 262128),
|
||||||
|
):
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
INSERT INTO models (
|
||||||
|
model_id, provider, base_model_id, tier, context_window,
|
||||||
|
effective_context_window, max_output_tokens,
|
||||||
|
cost_per_1m_prompt, cost_per_1m_completion,
|
||||||
|
supports_vision, supports_json_mode,
|
||||||
|
latency_class, reasoning_mode, context_variant,
|
||||||
|
access_level, availability, last_updated
|
||||||
|
) VALUES (?, 'neuralwatt', ?, 1, ?, ?, 16384, 0.30, 0.10,
|
||||||
|
?, 1, 'standard', 'default', 'full', 'public', 'active',
|
||||||
|
?)
|
||||||
|
""",
|
||||||
|
(model_id, model_id, eff_ctx, eff_ctx, vision, now),
|
||||||
|
)
|
||||||
|
for model_id in ("kimi-k3", "kimi-k3-fast"):
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO admin_model_overrides "
|
||||||
|
"(model_id, provider, availability, reason, updated_at) "
|
||||||
|
"VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)",
|
||||||
|
(model_id, now),
|
||||||
|
)
|
||||||
|
for tokens in (242486, 195000):
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
INSERT INTO route_decisions (
|
||||||
|
observed_at, kind, task_category, task_tier,
|
||||||
|
required_context_tokens, latency_tolerance,
|
||||||
|
selected_model, selected_provider, rejected_reason,
|
||||||
|
tools, images, json_mode, streamed
|
||||||
|
) VALUES (?, 'chat', 'coding_general', 1, 200000,
|
||||||
|
'interactive', NULL, NULL, ?, 0, 1, 0, 0)
|
||||||
|
""",
|
||||||
|
(
|
||||||
|
now,
|
||||||
|
f"context >= {tokens} tokens; "
|
||||||
|
f"vision-capable model (request carries image(s))",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
|
||||||
|
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
|
||||||
|
monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False)
|
||||||
|
monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", False)
|
||||||
|
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
|
||||||
|
|
||||||
|
with TestClient(dispatcher.app) as client:
|
||||||
|
resp = client.get("/metrics")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
data = resp.json()
|
||||||
|
warnings = data["coverage"]["warnings"]
|
||||||
|
assert len(warnings) > 0
|
||||||
|
assert any("rejection" in w for w in warnings)
|
||||||
|
# The capability warning is secondary per the plan (it depends on the
|
||||||
|
# seeded models) but this seed reproduces the incident, so it fires.
|
||||||
|
assert any("vision" in w for w in warnings)
|
||||||
|
|
||||||
|
|
||||||
def _sse_frame(line: str | bytes) -> dict:
|
def _sse_frame(line: str | bytes) -> dict:
|
||||||
"""Parse one ``data: <json>`` SSE line and return the JSON payload."""
|
"""Parse one ``data: <json>`` SSE line and return the JSON payload."""
|
||||||
if isinstance(line, bytes):
|
if isinstance(line, bytes):
|
||||||
|
|||||||
Reference in New Issue
Block a user