From e46d4ed0435ba8f1dc18f4f27578334b8ac25980 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 5 Sep 2026 01:10:41 -0400 Subject: [PATCH 1/2] feat(metrics): capability-aware sub-ceiling and rejection-rate warnings --- config/config.yaml | 20 ++++ src/config.py | 30 +++++ src/metrics.py | 268 ++++++++++++++++++++++++++++++++++++++++++++- 3 files changed, 315 insertions(+), 3 deletions(-) diff --git a/config/config.yaml b/config/config.yaml index cf8f81d..adca780 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -84,6 +84,26 @@ objective: # 29-31 to avoid shorter-month edge cases). billing_reset_day: 6 + # Rejection-rate detection over route_decisions rows that selected no model + # (422 "no model satisfies the hard filters"). The signal is novelty or rate — + # NEVER mere presence: this deployment routinely has ~3 rejections/hour of + # ordinary over-large tier-3 requests that are behaving as designed (6 in the + # last 24h, 17 all-time at 2026-09-05), so a count-only tripwire would be + # permanently on. + # + # alert window (hours): rejections this recent are counted per + # (task_tier, digit-normalized reason) group. + rejection_warning_window_hours: 1 + # baseline window (hours) BEFORE the alert window: groups absent here are + # "new". A new group warns from 2 occurrences — the 2026-09-04 vision + # incident produced exactly 2 and no rate threshold can sit below routine + # noise yet above that. + rejection_warning_baseline_hours: 24 + # a group already present in the baseline warns only at this many + # occurrences inside the alert window: 6 = 2x the observed routine hourly + # peak, far below the dozens/hour a deprecation flare produces. + rejection_warning_min_count: 6 + context: safety_factor: 0.75 # fraction of advertised context treated as usable default_output_reserve_tokens: 4096 diff --git a/src/config.py b/src/config.py index b19c0b5..5c2b886 100644 --- a/src/config.py +++ b/src/config.py @@ -54,6 +54,9 @@ class Objective(StrictModel): quota_runway_warning_hours: Optional[int] = None quota_burn_min_segment_samples: Optional[int] = None quota_burn_min_segment_hours: Optional[float] = None + rejection_warning_window_hours: Optional[int] = None + rejection_warning_baseline_hours: Optional[int] = None + rejection_warning_min_count: Optional[int] = None @field_validator("quality_tolerance") @classmethod @@ -115,6 +118,33 @@ class Objective(StrictModel): ) return v + @field_validator("rejection_warning_window_hours") + @classmethod + def rejection_window_positive(cls, v: Optional[int]) -> Optional[int]: + if v is not None and v <= 0: + raise ValueError( + "objective.rejection_warning_window_hours must be > 0" + ) + return v + + @field_validator("rejection_warning_baseline_hours") + @classmethod + def rejection_baseline_positive(cls, v: Optional[int]) -> Optional[int]: + if v is not None and v <= 0: + raise ValueError( + "objective.rejection_warning_baseline_hours must be > 0" + ) + return v + + @field_validator("rejection_warning_min_count") + @classmethod + def rejection_min_count_positive(cls, v: Optional[int]) -> Optional[int]: + if v is not None and v <= 0: + raise ValueError( + "objective.rejection_warning_min_count must be > 0" + ) + return v + class ContextOverride(StrictModel): """Per-model context handling, for a row whose real limits are known. diff --git a/src/metrics.py b/src/metrics.py index efdb2df..0d948b9 100644 --- a/src/metrics.py +++ b/src/metrics.py @@ -14,6 +14,9 @@ quota_burn — kWh metered in the last 30 d and in the current billing credit balance, burn rate and runway from allowance_remaining_usd scoring_coverage — which scoring axes actually have data +capability_ceilings — vision / json_mode context-window sub-ceilings +capability_demand_warnings — demand-relative warnings for those sub-ceilings +rejection_warnings — grouped rejection-rate / new-pattern warnings recent_decisions — last N rows from the route_decisions observability table per_model — per-model aggregates over energy_observations (last 30 d) verdict_mix — counts by verdict from verifications (last N days) @@ -23,9 +26,10 @@ pinch_summary — context-pruning savings over the last 30 d from __future__ import annotations +import re import sqlite3 from datetime import date, datetime, timedelta, timezone -from typing import Any, List, Optional +from typing import Any, Final, List, Optional import routing @@ -406,6 +410,10 @@ def scoring_coverage( f"tier {tier} has 0 eligible models for latency_tolerance={lat_tol}" ) + cap_ctx = capability_ceilings(conn, cfg) + warnings.extend(capability_demand_warnings(conn, cfg, cap_ctx)) + warnings.extend(rejection_warnings(conn, cfg)) + return { "routable_models": total, "with_energy_data": total - len(missing_energy), @@ -568,6 +576,8 @@ def context_ceilings_with_rows( cfg: Any, rows: List[dict], exclude_models: Optional[set[str]] = None, + require_vision: bool = False, + require_json_mode: bool = False, ) -> dict: """Like ``context_ceilings`` but over caller-supplied model rows. @@ -575,6 +585,9 @@ def context_ceilings_with_rows( proficiency and because the default ``exclude_models`` set is read from ``admin_model_overrides``. Callers that already know the exclusion set (e.g. admin.py applying a just-committed override) may pass it explicitly. + ``require_vision`` / ``require_json_mode`` gate the candidate set the + same way live routing gates it, so a capability-gated sub-ceiling goes + through this one path rather than a second copy of the filters. """ if exclude_models is None: exclude_models = _admin_deprecated_models(conn) @@ -600,8 +613,8 @@ def context_ceilings_with_rows( exclude_deprecated=True, exclude_models=exclude_models, min_tool_proficiency=None, - require_vision=False, - require_json_mode=False, + require_vision=require_vision, + require_json_mode=require_json_mode, ) ceiling = max((r["effective_context_window"] for r in candidates), default=0) result[(tier, latency_tolerance)] = { @@ -611,6 +624,255 @@ def context_ceilings_with_rows( return result +def capability_ceilings( + conn: sqlite3.Connection, + cfg: Any, +) -> dict: + """Context-window ceiling per capability-gated subset, by tier. + + The plain ``(tier, latency_tolerance)`` buckets from + ``context_ceilings`` are blind to capability gates: during the + 2026-09-04 incident the tier-1 interactive ceiling stayed at 782,324 + while the vision-capable subset had collapsed to 192,500 (the only + vision rows large enough had been deprecated through admin overrides), + so every existing demand check was silent on requests that 422ed. + Vision and JSON mode are the two flags that can independently empty + the candidate set — both fail CLOSED on an unknown catalog flag, which + is what lets them shrink the set sharply — so each gets its own series + here. Deliberately NOT one bucket per capability combination; that is + a combinatorial explosion over dimensions that do not interact. + + Delegates to ``context_ceilings_with_rows`` once per dimension, so every + sub-ceiling comes from the same ``routing.select_candidates`` path + live routing uses, admin deprecations included. + + Returns + ``{"vision": {(tier, latency_tolerance): {"ceiling", "count"}}, "json_mode": + {...}}``, one sub-dict per capability dimension, each keyed by the + ``(tier, latency_tolerance)`` pair present in the DB. + """ + rows = [dict(r) for r in conn.execute("SELECT * FROM models")] + exclude_models = _admin_deprecated_models(conn) + result: dict[str, dict[tuple, dict[str, int]]] = {} + for dimension in ("vision", "json_mode"): + sub_ceilings = context_ceilings_with_rows( + conn, + cfg, + rows, + exclude_models=exclude_models, + require_vision=dimension == "vision", + require_json_mode=dimension == "json_mode", + ) + result[dimension] = sub_ceilings + return result + + +def capability_demand_warnings( + conn: sqlite3.Connection, + cfg: Any, + cap_ctx: dict, +) -> List[str]: + """Demand-relative warnings for the capability-gated sub-ceilings. + + Mirrors ``demand_ceiling_warnings`` for each dimension — 7d MAX probe + first with a 30d fallback, silence for a bucket no such request ever + hit, else window = 7 when the 7d row count is >= 10 (30 otherwise), + warning only when that window's observed max exceeds the sub-ceiling. + The one difference is what "demand" means: request rows must carry the + capability (``images`` / ``json_mode`` columns on route_decisions), so + a vision sub-ceiling is compared only against requests that actually + carry images, for the ``(tier, "interactive")`` bucket. ``cfg`` is + accepted for signature symmetry with ``demand_ceiling_warnings``; this + detector has no escalation counterpart to read from it. + + The demand read is trailing-window, so a warning stays alive until a + big historical request ages out of it; the ``window: last {n}d`` + annotation in the message is what keeps that self-explaining. + """ + warnings: List[str] = [] + + dimension_columns = { + "vision": "images", + "json_mode": "json_mode", + } + capability_nouns = { + "vision": "for requests carrying images", + "json_mode": "for requests requesting JSON mode", + } + + for dimension in ("vision", "json_mode"): + buckets = cap_ctx.get(dimension) or {} + for tier in sorted({t for (t, _) in buckets}): + bucket = buckets.get((tier, "interactive")) + if bucket is None: + continue + column = dimension_columns.get(dimension) + if column is None: + continue + row = conn.execute( + "SELECT MAX(required_context_tokens) AS mx FROM route_decisions" + " WHERE julianday(observed_at) > julianday('now', '-7 days')" + f" AND kind = 'chat' AND task_tier = ? AND {column} = 1", + (tier,), + ).fetchone() + mx_7d = row["mx"] if row else None + if mx_7d is None: + row = conn.execute( + "SELECT MAX(required_context_tokens) AS mx FROM route_decisions" + " WHERE julianday(observed_at) > julianday('now', '-30 days')" + f" AND kind = 'chat' AND task_tier = ? AND {column} = 1", + (tier,), + ).fetchone() + mx_7d = row["mx"] if row else None + if mx_7d is None: + # No request carrying this capability was ever routed for this + # tier — an unexercised sub-ceiling warns about nothing. + continue + row_cnt = conn.execute( + "SELECT COUNT(*) AS n FROM route_decisions" + " WHERE julianday(observed_at) > julianday('now', '-7 days')" + f" AND kind = 'chat' AND task_tier = ? AND {column} = 1", + (tier,), + ).fetchone()["n"] + window = 7 if row_cnt >= 10 else 30 + row = conn.execute( + "SELECT MAX(required_context_tokens) AS mx FROM route_decisions" + f" WHERE julianday(observed_at) > julianday('now', '-{window} days')" + f" AND kind = 'chat' AND task_tier = ? AND {column} = 1", + (tier,), + ).fetchone() + observed_max = row["mx"] if row else 0 + ceiling = bucket["ceiling"] + if observed_max > ceiling: + warnings.append( + f"{dimension}-capable tier {tier} context ceiling " + f"({ceiling}) is below observed max demand " + f"({observed_max}, window: last {window}d) " + f"{capability_nouns[dimension]} — requests above this may " + "return 422" + ) + + return warnings + + +# A novel (tier, normalized_reason) group must reach this count inside the +# alert window before warning. The 2026-09-04 incident produced exactly 2 +# image rejections, so the floor must be <= 2; 1 would fire on any one-off +# genuinely-impossible request that correctly 422s. +NOVEL_GROUP_MIN_COUNT: Final = 2 + +# How many rejections a group needs inside the alert window before +# "new rejection pattern:" is reported. A single rejection is +# indistinguishable from a genuinely impossible request, which should 422 — +# the signal is a rate, not the existence of a rejection. +# +# Caveat: "novel" means ZERO occurrences in the prior baseline window, so +# a familiar-but-intermittent pattern can read as novel after a quiet gap +# and fire one spurious "new rejection pattern:" warning. It is +# self-clearing as the pattern's own rejections age into the baseline +# window; widen objective.rejection_warning_baseline_hours to suppress it. + + +def rejection_warnings( + conn: sqlite3.Connection, + cfg: Any, +) -> List[str]: + """Reactive safety net over the rejections the router already logged. + + The ceiling checks above only catch dimensions someone modelled; this + one catches everything, at the cost of firing after the first failures + rather than before. Rejections (``selected_model IS NULL AND + rejected_reason IS NOT NULL``) are pulled once across the alert plus + baseline window and split by exact age — alert bucket at + ``rejection_warning_window_hours`` or younger, baseline bucket older + than the alert window but within ``rejection_warning_baseline_hours`` + of it — then grouped by *normalized* reason plus ``task_tier``. + + Two signals, one line per group: + + * ``rejection rate:`` — the group reached + ``objective.rejection_warning_min_count`` rejections in the alert + window. + * ``new rejection pattern:`` — the group is ABSENT from the baseline + window and reached ``NOVEL_GROUP_MIN_COUNT`` in the alert window. + This prefix wins when both would fire. + + Zero alert-window rejections produce nothing at all: a rejection is + not inherently an error, and a genuinely impossible request should + 422 — the signal is a rate. + """ + window = getattr(cfg.objective, "rejection_warning_window_hours", None) + if window is None: + window = 1 + baseline = getattr(cfg.objective, "rejection_warning_baseline_hours", None) + if baseline is None: + baseline = 24 + min_count = getattr(cfg.objective, "rejection_warning_min_count", None) + if min_count is None: + min_count = 6 + + now = datetime.now(timezone.utc) + + rows = conn.execute( + """ + SELECT observed_at, task_tier, rejected_reason + FROM route_decisions + WHERE selected_model IS NULL + AND rejected_reason IS NOT NULL + AND julianday(observed_at) > julianday('now', '-' || ? || ' hours') + ORDER BY observed_at DESC + """, + (str(window + baseline),), + ).fetchall() + + alert_groups: dict[tuple[Optional[int], str], dict] = {} + baseline_keys: set[tuple[Optional[int], str]] = set() + for row in rows: + try: + observed = datetime.fromisoformat(row["observed_at"]) + except ValueError: + # Defensive, same shape as quota_balance_and_burn: one malformed + # timestamp must not take the whole report down. + continue + # Group on digit-normalized reasons: the raw strings embed the + # request's own numbers ("context >= 242486 tokens"), so literal + # grouping yields N groups of one and hides the pattern entirely. + key = (row["task_tier"], re.sub(r"\d+", "N", row["rejected_reason"])) + age_hours = (now - observed).total_seconds() / 3600.0 + if age_hours <= window: + group = alert_groups.get(key) + if group is None: + # Rows arrive newest-first, so the first touch per group is + # its most recent occurrence. + alert_groups[key] = {"count": 1, "most_recent": observed} + else: + group["count"] += 1 + elif age_hours <= window + baseline: + baseline_keys.add(key) + + if not alert_groups: + return [] + + warnings: List[str] = [] + for (tier, normalized_reason), group in alert_groups.items(): + count = group["count"] + is_novel = (tier, normalized_reason) not in baseline_keys + prefix: Optional[str] = None + if is_novel and count >= NOVEL_GROUP_MIN_COUNT: + prefix = "new rejection pattern:" + elif count >= min_count: + prefix = "rejection rate:" + if prefix is None: + continue + tier_label = "unknown" if tier is None else str(tier) + warnings.append( + f"{prefix} {count} rejection(s) in the last {window}h: " + f"{normalized_reason} (tier {tier_label}; " + f"most recent: {group['most_recent'].isoformat()})" + ) + return warnings + + def recent_decisions( conn: sqlite3.Connection, limit: int = 50, -- 2.49.1 From 5d31dd5fc3400c31965dbf5f97708e121b511323 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 5 Sep 2026 01:26:49 -0400 Subject: [PATCH 2/2] test(metrics): capability sub-ceiling and rejection-rate regression tests --- tests/test_metrics.py | 480 +++++++++++++++++++++++++++++++++ tests/test_metrics_endpoint.py | 100 ++++++- 2 files changed, 578 insertions(+), 2 deletions(-) diff --git a/tests/test_metrics.py b/tests/test_metrics.py index c2ed780..c01b986 100644 --- a/tests/test_metrics.py +++ b/tests/test_metrics.py @@ -30,12 +30,16 @@ import dispatcher from config import load_config from metrics import ( _next_reset_date, + capability_ceilings, + capability_demand_warnings, context_ceilings, + demand_ceiling_warnings, local_energy_summary, per_model, pinch_summary, quota_burn, recent_decisions, + rejection_warnings, scoring_coverage, top_proficiency, verdict_mix, @@ -794,6 +798,482 @@ def test_context_ceilings_excludes_admin_deprecated_models(tmp_path): assert ctx[key]["count"] == 1 +# --- capability sub-ceiling demand warnings (2026-09-04 incident) --------------- + + +def _insert_model_row( + conn: sqlite3.Connection, + *, + model_id: str, + tier: int, + effective_context_window: int, + supports_vision: int, + supports_json_mode: int = 1, +) -> None: + """Insert one active, routable catalog row with explicit capability flags. + + ``_seed_models`` covers the common case but hardcodes its capability + flags and effective window; the incident-reproduction tests need both + per-row. + """ + conn.execute( + """ + INSERT INTO models ( + model_id, provider, base_model_id, tier, context_window, + effective_context_window, max_output_tokens, + cost_per_1m_prompt, cost_per_1m_completion, + supports_vision, supports_json_mode, + latency_class, reasoning_mode, context_variant, + access_level, availability, last_updated + ) VALUES (?, 'neuralwatt', ?, ?, ?, ?, 16384, 0.30, 0.10, + ?, ?, 'standard', 'default', 'full', 'public', 'active', + ?) + """, + ( + model_id, + model_id, + tier, + effective_context_window, + effective_context_window, + supports_vision, + supports_json_mode, + _now().isoformat(), + ), + ) + conn.commit() + + +def _deprecate_model(conn: sqlite3.Connection, model_id: str) -> None: + """Insert an admin override marking *model_id* deprecated (reason: cost).""" + conn.execute( + "INSERT INTO admin_model_overrides " + "(model_id, provider, availability, reason, updated_at) " + "VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)", + (model_id, _now().isoformat()), + ) + conn.commit() + + +def _insert_rejected_decision( + conn: sqlite3.Connection, + *, + tier: int, + reason: str, + observed_at: str, + tokens: int = 200000, + images: int = 0, + json_mode: int = 0, +) -> None: + """Insert a route_decisions row that selected no model (a 422 rejection). + + ``kind='chat'`` so the demand queries (which filter on kind) see it; + ``latency_tolerance='interactive'`` to match the tier-1 bucket key the + ceiling checks use. + """ + conn.execute( + """ + INSERT INTO route_decisions ( + observed_at, kind, task_category, task_tier, + required_context_tokens, confidence, classifier_ms, + classification_source, latency_tolerance, candidates_considered, + selected_model, selected_provider, rejected_reason, + session_key, tools, images, json_mode, streamed + ) VALUES (?, 'chat', 'coding_general', ?, ?, NULL, NULL, + 'override', 'interactive', 0, NULL, NULL, ?, + 'sess', 0, ?, ?, 0) + """, + (observed_at, tier, tokens, reason, images, json_mode), + ) + conn.commit() + + +def _seed_incident_state(conn: sqlite3.Connection) -> str: + """Seed the 2026-09-04 incident model state; returns the shared 'now' stamp. + + Three tier-1 rows: kimi-k3 and kimi-k3-fast — the only vision-capable + large-context models — deprecated through admin overrides, and + deepseek-v4-flash (no vision, 262,128 effective tokens) left active. + The overall tier-1 interactive ceiling stays high while the vision + sub-ceiling collapses to zero, which is exactly the shape every + pre-2026-09-05 ceiling check was blind to. + """ + for model_id, vision, eff_ctx in ( + ("kimi-k3", 1, 782324), + ("kimi-k3-fast", 1, 782324), + ("deepseek-v4-flash", 0, 262128), + ): + _insert_model_row( + conn, + model_id=model_id, + tier=1, + effective_context_window=eff_ctx, + supports_vision=vision, + ) + for model_id in ("kimi-k3", "kimi-k3-fast"): + _deprecate_model(conn, model_id) + return _now().isoformat() + + +def test_capability_demand_warning_incident_reproduction(tmp_path): + """The 2026-09-04 incident state warns on the vision sub-ceiling while + every existing ``(tier, latency_tolerance)`` check stays silent. + + In the incident, kimi-k3 and kimi-k3-fast — the only vision-capable + tier-1 models with a large context — were deprecated through admin + overrides, so a 200,000-token image request was unservable even though + the overall tier-1 ceiling (262,128, from non-vision deepseek-v4-flash) + comfortably covered that demand. The new capability check must fire; + the existing demand_ceiling_warnings must NOT — that silence is what + hid the incident for 19 hours. + """ + conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL) + ts = _seed_incident_state(conn) + _insert_rejected_decision( + conn, + tier=1, + reason=( + "context >= 242486 tokens; " + "vision-capable model (request carries image(s))" + ), + observed_at=ts, + tokens=200000, + images=1, + ) + + # The overall tier-1 interactive ceiling is untouched by the deprecations. + ctx = context_ceilings(conn, CFG) + assert ctx[(1, "interactive")]["ceiling"] == 262128 + + # The vision-gated subset collapsed: no candidate survives the gate. + cap_ctx = capability_ceilings(conn, CFG) + assert cap_ctx["vision"][(1, "interactive")]["ceiling"] == 0 + assert cap_ctx["vision"][(1, "interactive")]["count"] == 0 + # deepseek-v4-flash still serves the json_mode subset. + assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 262128 + + warnings = capability_demand_warnings(conn, CFG, cap_ctx) + assert len(warnings) == 1 + w = warnings[0] + assert "vision" in w + assert "tier 1" in w + assert "ceiling (0)" in w + assert "200000" in w + + # The pre-existing check must stay silent on the same state: the overall + # tier-1 ceiling (262128) still exceeds the observed max demand (200000). + assert demand_ceiling_warnings(conn, CFG, ctx, [1]) == [] + + +def test_capability_demand_no_demand_no_warning(tmp_path): + """A state with demand but no capability-flagged demand warns about nothing. + + Requests exist at the tier — far above every ceiling — but none carry + images or ask for JSON mode, so both capability subsets are unexercised: + an unexercised sub-ceiling is not a warning. + """ + conn = _make_db(tmp_path) + _seed_models(conn) + _insert_rejected_decision( + conn, + tier=2, + reason="context >= 999999 tokens", + observed_at=_now().isoformat(), + tokens=999999, + ) + + assert capability_demand_warnings(conn, CFG, capability_ceilings(conn, CFG)) == [] + + +def test_capability_json_mode_demand_warning(tmp_path): + """Deprecating the only json_mode-capable model trips the json_mode warning. + + Same incident shape as the vision case, different capability dimension: + the json_mode sub-ceiling collapses to zero while the overall ceiling + stays fine, and a json_mode request that cannot be served produces a + warning naming json_mode. + """ + conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL) + _insert_model_row( + conn, + model_id="jm-big", + tier=1, + effective_context_window=782324, + supports_vision=0, + supports_json_mode=1, + ) + _insert_model_row( + conn, + model_id="plain-fallback", + tier=1, + effective_context_window=262128, + supports_vision=0, + supports_json_mode=0, + ) + _deprecate_model(conn, "jm-big") + _insert_rejected_decision( + conn, + tier=1, + reason=( + "context >= 242486 tokens; json-mode-capable model " + "(json_mode requested)" + ), + observed_at=_now().isoformat(), + tokens=200000, + json_mode=1, + ) + + cap_ctx = capability_ceilings(conn, CFG) + assert cap_ctx["json_mode"][(1, "interactive")]["ceiling"] == 0 + + warnings = capability_demand_warnings(conn, CFG, cap_ctx) + assert len(warnings) == 1 + w = warnings[0] + assert "json_mode" in w + assert "tier 1" in w + assert "ceiling (0)" in w + assert "200000" in w + + +# --- rejection warnings (reactive rejection detector) -------------------------- + + +def test_rejection_warnings_zero_rejections(tmp_path): + """Zero NULL-selection rows produce no rejection warnings. + + The absence of rejections is not a signal worth reporting — the + detector watches rates and new patterns, not mere existence. + """ + conn = _make_db(tmp_path) + _seed_models(conn) + assert rejection_warnings(conn, CFG) == [] + + +def test_rejection_warnings_novel_group_at_incident_volume_fires(tmp_path): + """The 2026-09-04 incident shape — a novel group of only 2 — must fire. + + The incident produced exactly 2 image rejections, below the routine + noise floor (about 3/hour on this deployment), so no rate threshold can + catch it; the novelty prong — absent from the 24h baseline, count + >= 2 — is what must fire here. Both rows share one timestamp so the + rendered most-recent value is deterministic. + """ + conn = _make_db(tmp_path) + _seed_models(conn) + ts = _now().isoformat() + for reason in ( + "tier >= 1; context >= 242486 tokens; interactive; " + "vision-capable model (request carries image(s))", + "tier >= 1; context >= 195000 tokens; interactive; " + "vision-capable model (request carries image(s))", + ): + _insert_rejected_decision(conn, tier=1, reason=reason, observed_at=ts) + + warnings = rejection_warnings(conn, CFG) + assert len(warnings) == 1 + w = warnings[0] + assert w.startswith("new rejection pattern:") + # Token counts collapse: both raw reasons normalize to "context >= N tokens". + assert "2 rejection(s)" in w + assert "context >= N tokens" in w + # The tier is rendered from the structured task_tier column, not the string. + assert "(tier 1;" in w + assert ts in w + + +def test_rejection_warnings_familiar_group_below_threshold_silent(tmp_path): + """Routine over-large tier-3 traffic — 3/hour and familiar — stays silent. + + Regression for the measured routine floor: this deployment routinely + sees about 3 rejections per hour of ordinary over-large tier-3 requests + that behave as designed. The group is familiar (present in the 24h + baseline window) and below ``rejection_warning_min_count`` (6), so + neither the novelty prong nor the rate prong fires. + """ + conn = _make_db(tmp_path) + _seed_models(conn) + now = _now() + # Baseline occurrence: 2h ago — inside the 24h baseline, outside the 1h + # alert window. + _insert_rejected_decision( + conn, + tier=3, + reason="tier >= 3; context >= 40000 tokens; interactive", + observed_at=(now - timedelta(hours=2)).isoformat(), + ) + # The live-DB-measured routine hourly volume, with distinct token counts + # so digit normalization does real grouping work. + for tokens in (30000, 31000, 32000): + _insert_rejected_decision( + conn, + tier=3, + reason=f"tier >= 3; context >= {tokens} tokens; interactive", + observed_at=now.isoformat(), + ) + + assert rejection_warnings(conn, CFG) == [] + + +def test_rejection_warnings_familiar_group_at_threshold_fires(tmp_path): + """A familiar group at ``rejection_warning_min_count`` (6) fires as a rate.""" + conn = _make_db(tmp_path) + _seed_models(conn) + now = _now() + _insert_rejected_decision( + conn, + tier=3, + reason="tier >= 3; context >= 40000 tokens; interactive", + observed_at=(now - timedelta(hours=2)).isoformat(), + ) + for tokens in (30000, 31000, 32000, 33000, 34000, 35000): + _insert_rejected_decision( + conn, + tier=3, + reason=f"tier >= 3; context >= {tokens} tokens; interactive", + observed_at=now.isoformat(), + ) + + warnings = rejection_warnings(conn, CFG) + assert len(warnings) == 1 + w = warnings[0] + assert w.startswith("rejection rate:") + assert "6 rejection(s)" in w + + +def test_rejection_warnings_novel_single_rejection_silent(tmp_path): + """A single novel rejection stays silent. + + A genuinely impossible one-off request SHOULD 422; presence of one + rejection is not a signal. n=1 is below NOVEL_GROUP_MIN_COUNT (2). + """ + conn = _make_db(tmp_path) + _seed_models(conn) + _insert_rejected_decision( + conn, + tier=1, + reason=( + "tier >= 1; context >= 242486 tokens; interactive; " + "vision-capable model (request carries image(s))" + ), + observed_at=_now().isoformat(), + ) + + assert rejection_warnings(conn, CFG) == [] + + +def test_rejection_warnings_groups_by_structured_tier(tmp_path): + """Tier-1 and tier-3 rejections of the same normalized shape never merge. + + Regression for the sibling finding of the incident review: digit + normalization erases the tier embedded in the reason string ("tier >= 1" + and "tier >= 3" both become "tier >= N"), so grouping by normalized + string alone would produce ONE merged count-4 group, hiding which + tier's candidate set went empty. The discriminator is the structured + ``task_tier`` column, rendered back into the body as "(tier N; ...)". + """ + conn = _make_db(tmp_path) + _seed_models(conn) + ts = _now().isoformat() + for tier, tokens in ((1, 242486), (1, 50000), (3, 229702), (3, 30000)): + _insert_rejected_decision( + conn, + tier=tier, + reason=f"tier >= {tier}; context >= {tokens} tokens; interactive", + observed_at=ts, + ) + + warnings = rejection_warnings(conn, CFG) + assert len(warnings) == 2 + tier1 = [w for w in warnings if "(tier 1;" in w] + tier3 = [w for w in warnings if "(tier 3;" in w] + assert len(tier1) == 1 + assert len(tier3) == 1 + assert "2 rejection(s)" in tier1[0] + assert "2 rejection(s)" in tier3[0] + + +def test_rejection_warnings_old_rejections_not_counted(tmp_path): + """A rejection outside both windows is not counted at all. + + Two same-shape rows 3 days old reach the novelty floor of 2, so the + only thing that keeps them silent is their age — outside the 1h alert + window and beyond the 25h total pull window they are not even read. + """ + conn = _make_db(tmp_path) + _seed_models(conn) + for tokens in (242486, 195000): + _insert_rejected_decision( + conn, + tier=1, + reason=f"tier >= 1; context >= {tokens} tokens; interactive", + observed_at=(_now() - timedelta(days=3)).isoformat(), + ) + + assert rejection_warnings(conn, CFG) == [] + + +def test_rejection_warnings_custom_window(tmp_path): + """``rejection_warning_window_hours`` widens the alert window. + + Two same-shape rows 2h old sit outside the default 1h window but + inside a 3h window; with the knob set they form a novel group at the + incident volume and must fire. + """ + conn = _make_db(tmp_path) + _seed_models(conn) + two_hours_ago = (_now() - timedelta(hours=2)).isoformat() + for tokens in (242486, 195000): + _insert_rejected_decision( + conn, + tier=1, + reason=( + f"tier >= 1; context >= {tokens} tokens; interactive; " + "vision-capable model (request carries image(s))" + ), + observed_at=two_hours_ago, + ) + + wide_cfg = SimpleNamespace( + objective=SimpleNamespace( + rejection_warning_window_hours=3, + rejection_warning_baseline_hours=24, + rejection_warning_min_count=6, + ) + ) + + warnings = rejection_warnings(conn, wide_cfg) + assert len(warnings) == 1 + assert warnings[0].startswith("new rejection pattern:") + assert "2 rejection(s)" in warnings[0] + + +def test_scoring_coverage_includes_new_warnings(tmp_path): + """``scoring_coverage`` surfaces both new warning classes on the incident state. + + The 2026-09-04 incident produced two image rejections against a catalog + whose vision sub-ceiling had collapsed. Both the capability demand + warning and the novel rejection pattern must reach the same warnings + list that /metrics, the admin portal and the TUI read. + """ + conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL) + ts = _seed_incident_state(conn) + for tokens in (242486, 195000): + _insert_rejected_decision( + conn, + tier=1, + reason=( + f"context >= {tokens} tokens; " + "vision-capable model (request carries image(s))" + ), + observed_at=ts, + images=1, + ) + + result = scoring_coverage(conn, CFG) + warnings = result["warnings"] + assert any("vision" in w for w in warnings) + assert any(w.startswith("new rejection pattern:") for w in warnings) + + # --- recent_decisions tests --------------------------------------------------- diff --git a/tests/test_metrics_endpoint.py b/tests/test_metrics_endpoint.py index d5c3adf..4c8f408 100644 --- a/tests/test_metrics_endpoint.py +++ b/tests/test_metrics_endpoint.py @@ -31,10 +31,10 @@ def _now() -> datetime: return datetime.now(timezone.utc) -def _make_db(tmp_path: Path) -> sqlite3.Connection: +def _make_db(tmp_path: Path, extra_sql: str = "") -> sqlite3.Connection: conn = sqlite3.connect(str(tmp_path / "test.db")) conn.row_factory = sqlite3.Row - conn.executescript(SCHEMA_SQL) + conn.executescript(SCHEMA_SQL + extra_sql) return conn @@ -358,6 +358,102 @@ def test_metrics_empty_db_returns_200(monkeypatch, tmp_path): assert "generated_at" in data +# admin_model_overrides is NOT in schema.sql — dispatcher creates it at +# startup via ensure_route_decisions-style migration, so endpoint tests that +# need it build it with extra_sql (pattern from tests/test_metrics.py). +_ADMIN_TABLE_SQL = """ +CREATE TABLE IF NOT EXISTS admin_model_overrides ( + model_id TEXT NOT NULL, + provider TEXT NOT NULL, + availability TEXT NOT NULL, + reason TEXT, + updated_at TEXT NOT NULL, + PRIMARY KEY (model_id, provider) +); +CREATE INDEX IF NOT EXISTS idx_admin_model_overrides_availability + ON admin_model_overrides (availability); +""" + + +def test_metrics_coverage_warnings_include_rejection_and_capability( + tmp_path, monkeypatch +): + """GET /metrics surfaces the new rejection and capability warnings. + + Seeds the 2026-09-04 incident state — kimi-k3 and kimi-k3-fast (the only + vision-capable tier-1 models) deprecated via admin overrides, plus two + image rejections — into the endpoint's temp DB, and asserts the warnings + reach the same coverage.warnings list the admin bell and TUI render. + The rejection warning is the deterministic assertion; on this exact seed + the capability warning fires too. + """ + now = _now().isoformat() + conn = _make_db(tmp_path, extra_sql=_ADMIN_TABLE_SQL) + for model_id, vision, eff_ctx in ( + ("kimi-k3", 1, 782324), + ("kimi-k3-fast", 1, 782324), + ("deepseek-v4-flash", 0, 262128), + ): + conn.execute( + """ + INSERT INTO models ( + model_id, provider, base_model_id, tier, context_window, + effective_context_window, max_output_tokens, + cost_per_1m_prompt, cost_per_1m_completion, + supports_vision, supports_json_mode, + latency_class, reasoning_mode, context_variant, + access_level, availability, last_updated + ) VALUES (?, 'neuralwatt', ?, 1, ?, ?, 16384, 0.30, 0.10, + ?, 1, 'standard', 'default', 'full', 'public', 'active', + ?) + """, + (model_id, model_id, eff_ctx, eff_ctx, vision, now), + ) + for model_id in ("kimi-k3", "kimi-k3-fast"): + conn.execute( + "INSERT INTO admin_model_overrides " + "(model_id, provider, availability, reason, updated_at) " + "VALUES (?, 'neuralwatt', 'deprecated', 'cost', ?)", + (model_id, now), + ) + for tokens in (242486, 195000): + conn.execute( + """ + INSERT INTO route_decisions ( + observed_at, kind, task_category, task_tier, + required_context_tokens, latency_tolerance, + selected_model, selected_provider, rejected_reason, + tools, images, json_mode, streamed + ) VALUES (?, 'chat', 'coding_general', 1, 200000, + 'interactive', NULL, NULL, ?, 0, 1, 0, 0) + """, + ( + now, + f"context >= {tokens} tokens; " + f"vision-capable model (request carries image(s))", + ), + ) + conn.commit() + conn.close() + + monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db")) + monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False) + monkeypatch.setattr(dispatcher.cfg.routing, "require_vision", False) + monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", False) + monkeypatch.setenv("NEURALWATT_API_KEY", "test-key") + + with TestClient(dispatcher.app) as client: + resp = client.get("/metrics") + assert resp.status_code == 200 + data = resp.json() + warnings = data["coverage"]["warnings"] + assert len(warnings) > 0 + assert any("rejection" in w for w in warnings) + # The capability warning is secondary per the plan (it depends on the + # seeded models) but this seed reproduces the incident, so it fires. + assert any("vision" in w for w in warnings) + + def _sse_frame(line: str | bytes) -> dict: """Parse one ``data: `` SSE line and return the JSON payload.""" if isinstance(line, bytes): -- 2.49.1