feat(classifier): record why the classifier declined, and bound the degraded-warn knobs #112

Merged
alee merged 2 commits from feat/classifier-observability into main 2026-10-05 03:08:46 +00:00
16 changed files with 1068 additions and 55 deletions

View File

@@ -870,6 +870,24 @@ over-applied exclusion already starved feedback once (`client_capped`).
decisions. A survivable failure is exactly the kind that goes unnoticed for
weeks.
**Why a decline is now recorded, and what the numbers said (2026-10-04).**
Under `local_decision` on the 4b the classifier declines about a third of agent
turns on purpose: 769 of 2,557 (30%) that day, 8% the day before, against 0% for
`local_llm` and `local_encoder`. 22.5% were `session_history` replays and 7.6%
the static `general_chat` guess, and `session_history` has no age bound (p99 67
minutes, max 100). None of that could be tuned from the database, because
`route_decisions.confidence` is the override's hard-coded 1.0 on 99.9% of chat
rows. `classifier_confidence`, `classifier_coverage` and `classifier_reject` now
hold the classifier's own numbers and the reason (codes in
[data-model](docs/data-model.md#what-the-classifier-did-and-why-its-answer-was-not-used)),
and the degraded-share warning lists the reasons it saw. The warning's default
threshold (0.5) never fires at that rate, so the live overlay sets 0.2.
`degraded_warn_min` and `degraded_warn_threshold` have no admin control, and that
is a deferred gap, not a decision: the `classifier` section is outside the
knob-coverage gate because it has its own card, and adding any `classifier.*`
knob to the generic registries would pull every other classifier scalar into the
gate at once. Do it in one pass when classification is reworked.
### Which implementation is PRIMARY is now a config choice
`classifier.mode` in the live deployment is currently `local_encoder`, set in

View File

@@ -324,11 +324,29 @@ const SOURCE_STYLES = {
classifier: 'background:rgba(59,130,246,.18);color:#93c5fd',
cached: 'background:rgba(34,197,94,.18);color:#4ade80',
fallback: 'background:rgba(239,68,68,.18);color:#f87171',
// Borrowed labels: a real classification of this session, replayed because
// the classifier declined this turn. Amber, between a fresh answer and the
// static guess, so a run of them is visible instead of reading as neutral.
session_history: 'background:rgba(245,158,11,.18);color:#fbbf24',
session_stale: 'background:rgba(245,158,11,.18);color:#fbbf24',
};
function sourceStyle(source) {
return SOURCE_STYLES[source] || 'background:rgba(148,163,184,.18);color:#cbd5e1';
}
function attemptTitle(d) {
// What the classifier itself did, for the source badge's tooltip. `confidence`
// cannot say this on a chat row (the re-route to measured context replaces it
// with 1.0), so these come from the classifier_* columns. A row with none of
// them (older rows, overrides, cache hits) gets no tooltip rather than a
// blank one.
const parts = [];
if (d.classifier_reject) parts.push(`declined: ${d.classifier_reject}`);
if (d.classifier_confidence != null) parts.push(`confidence ${Number(d.classifier_confidence).toFixed(3)}`);
if (d.classifier_coverage != null) parts.push(`coverage ${Number(d.classifier_coverage).toFixed(3)}`);
return parts.join(', ');
}
function flagChips(d) {
// Only the flags that are actually ON — showing all four dimmed on every
// row for hundreds of rows was noise, not a cue. Nothing on -> empty cell.
@@ -678,7 +696,7 @@ function renderTable() {
<td title="${escapeHtml(d.task_category || '')}"><span class="dec-badge" style="${categoryStyle(category)}">${escapeHtml(category)}</span></td>
<td>${escapeHtml(d.profile || '')}</td>
<td class="tier-${tier}">${escapeHtml(tier)}</td>
<td><span class="dec-badge" style="${sourceStyle(source)}">${escapeHtml(source)}</span></td>
<td><span class="dec-badge" style="${sourceStyle(source)}" title="${escapeHtml(attemptTitle(d))}">${escapeHtml(source)}</span></td>
<td class="model-cell" title="${escapeHtml(model)}">${escapeHtml(model)}</td>
<td class="text-muted">${escapeHtml(provider)}</td>
<td class="num" title="${d.required_context_tokens != null ? Number(d.required_context_tokens).toLocaleString() + ' tokens' : 'not recorded'}">${ctxCell(d.required_context_tokens)}</td>

View File

@@ -344,11 +344,28 @@ CREATE TABLE IF NOT EXISTS route_decisions (
-- X-Router-Agent (slugged by the
-- plugin, e.g. atlas-plan-executor);
-- NULL when the client sends none
parent_key TEXT -- 'c:' plus the parent conversation
parent_key TEXT, -- 'c:' plus the parent conversation
-- id from X-Router-Parent, set on a
-- sub-agent's rows; NULL for
-- top-level conversations or when
-- unknown
-- What the PRIMARY classifier did with this request. `confidence` above is
-- the confidence of the classification that ROUTED it, and on a chat row
-- that is almost always 1.0 because the chat path re-routes through the
-- override branch, which hard-codes it. These three are the classifier's
-- own numbers, read before that re-route. NULL on rows from before they
-- existed, on override rows and on session-cache hits (no attempt made).
classifier_confidence REAL, -- the classifier's confidence in
-- its answer, accepted or not
classifier_coverage REAL, -- local_decision only: total
-- option mass behind that answer
classifier_reject TEXT -- NULL when the answer was used,
-- else why not: below_confidence_min,
-- below_coverage_min, no_logprobs,
-- timeout, transport_error,
-- parse_error, primary_failed,
-- skipped_<why>. classification_source
-- then says what routed instead.
);
CREATE INDEX IF NOT EXISTS idx_verifications_model ON verifications (model_id, provider);

View File

@@ -258,9 +258,9 @@ failure — it means the checker had nothing to say, not that the model failed.
| `task_category` | TEXT | |
| `task_tier` | INTEGER | 1–3 |
| `required_context_tokens` | INTEGER | |
| `confidence` | REAL | Classifier confidence |
| `confidence` | REAL | Confidence of the classification that ROUTED the request. On a `chat` row this is almost always `1.0`, because the chat path re-routes through the override branch, which hard-codes it. Use `classifier_confidence` for what the classifier said |
| `classifier_ms` | INTEGER | Classification latency |
| `classification_source` | TEXT | `classifier` \| `override` \| `fallback` \| `cached` |
| `classification_source` | TEXT | `classifier` \| `override` \| `fallback` \| `cached` \| `session_stale` \| `session_history` \| `classifier_cloud` |
| `latency_tolerance` | TEXT | `interactive` \| `batch` |
| `candidates_considered` | INTEGER | How many survived hard filters |
| `selected_model` | TEXT | Null when no model was selected |
@@ -283,6 +283,43 @@ failure — it means the checker had nothing to say, not that the model failed.
| `prefix_divergence_index` | INTEGER | First message position whose bytes differ from the previous turn in this session |
| `prefix_tokens_after_divergence` | INTEGER | Estimated tokens at or after that position in THIS turn |
| `prefix_prev_message_count` | INTEGER | The previous turn's message count |
| `classifier_confidence` | REAL | The primary classifier's confidence in its answer, accepted or not. NULL when it produced none (a timeout) or made no attempt (`override`, `cached`) |
| `classifier_coverage` | REAL | `local_decision` only: the total option mass behind that answer |
| `classifier_reject` | TEXT | NULL when the answer was used; otherwise why it was not (codes below) |
### What the classifier did, and why its answer was not used
The three `classifier_*` columns record the **primary classifier's own
attempt**, read before the chat path re-routes on measured context. They exist
because a declined answer used to leave one number behind and it was in a log
line: `classification_source = session_history` says a borrowed label routed the
turn, not whether the classifier was a hair under its floor or nowhere near it,
and `confidence` could not say so (see its row above).
`classifier_reject` is NULL when the answer was used, and otherwise one of:
| code | meaning | numbers recorded |
|---|---|---|
| `below_confidence_min` | answered, but under `classifier.*.confidence_min` | confidence (and coverage for `local_decision`) |
| `below_coverage_min` | `local_decision`: total option mass under `coverage_min` | coverage, and the would-be confidence |
| `no_logprobs` | `local_decision`: Ollama returned none | none |
| `timeout` | the endpoint did not answer in time | none |
| `transport_error` | the endpoint was unreachable or errored | none |
| `parse_error` | the reply could not be parsed | none |
| `primary_failed` | any other failure of the primary | none |
| `skipped_gaming_mode`, `skipped_backoff` | the local classifier was never asked | none |
`classification_source` then says what routed the turn instead: `session_history`
(this session's last real label), `fallback` (the static guess) or
`classifier_cloud`. Tune `confidence_min` from the distribution: for
`local_decision` rows, compare `classifier_confidence` where `classifier_reject`
is NULL against where it is `below_confidence_min`.
Rows from before these columns existed hold NULL in all three, which is the
honest value; the numbers were never stored, so there is nothing to backfill.
The admin decisions page shows them as the tooltip on the source badge, the TUI
detail popup shows all three, and the `/metrics` degraded-share warning lists the
reasons it saw.
The three `prefix_*` columns are the prefix-stability probe
(`pinch.prefix_probe`, on by default). The provider bills the longest

View File

@@ -358,6 +358,11 @@ Mechanism, a Jev-style first-token-logprob classifier:
- As with `local_encoder`, it produces only `task_category` unless
`classifier.decision.tier_enabled` is set, in which case it may also choose
`task_tier`; otherwise tier falls back to `classifier.fallback_tier`.
- A declined answer is recorded, not just logged: `route_decisions.classifier_confidence`,
`classifier_coverage` and `classifier_reject` hold the numbers and the reason
(see [data-model](data-model.md#what-the-classifier-did-and-why-its-answer-was-not-used)).
On the 4b the decline is by design, not an outage: it ran at 30% of turns on
2026-10-04 (8% the day before), concentrated in long agent sessions.
Pull and enable it:

View File

@@ -0,0 +1,59 @@
"""Why the primary classifier's answer was not used, as data.
A classifier that declines to answer is the router's most consequential quiet
failure: the request is then routed on a borrowed or guessed label. Before this
module the reason existed only as a log line, and the number that decided it
(how far below ``confidence_min`` the answer was) existed nowhere durable, so a
threshold could not be tuned from the database. ``ClassifierRejected`` carries
both out of the raise site, and the ``REASON_*`` strings are the stable codes
stored in ``route_decisions.classifier_reject``.
Stdlib only, and it imports nothing from this package, so ``local_decision`` can
raise it without importing ``dispatcher`` (which imports ``local_decision``).
"""
from __future__ import annotations
from typing import Final
# The model answered, and the dispatcher declined the answer.
REASON_BELOW_CONFIDENCE: Final = "below_confidence_min"
REASON_BELOW_COVERAGE: Final = "below_coverage_min"
REASON_NO_LOGPROBS: Final = "no_logprobs"
# The model did not answer usefully.
REASON_TIMEOUT: Final = "timeout"
REASON_TRANSPORT: Final = "transport_error"
REASON_PARSE: Final = "parse_error"
REASON_PRIMARY_FAILED: Final = "primary_failed"
# The model was never asked. Followed by the skip reason, e.g.
# ``skipped_gaming_mode`` or ``skipped_backoff``.
REASON_SKIPPED_PREFIX: Final = "skipped_"
class ClassifierRejected(RuntimeError):
"""The classifier answered, but the answer was refused.
A ``RuntimeError`` subclass on purpose: ``classify()`` already treats a
``RuntimeError`` from the primary as "the primary failed" and walks the
cascade, so existing handlers keep working and only the data is new. The
message text is unchanged from the plain ``RuntimeError`` it replaces,
because it is what the ``fallback`` log line prints.
``confidence`` and ``coverage`` are whatever the raise site knew, and None
where it did not: a coverage rejection happens before a confidence exists.
"""
def __init__(
self,
reason: str,
message: str,
*,
confidence: float | None = None,
coverage: float | None = None,
) -> None:
super().__init__(message)
self.reason = reason
self.confidence = confidence
self.coverage = coverage

View File

@@ -1155,6 +1155,16 @@ class LocalDecisionConfig(StrictModel):
return v
# Bounds for the degraded-share warning. Named, not inlined in the validators,
# because a runtime write goes straight past Pydantic and the admin registry's
# bounds are then the only check: it imports these so the runtime path and the
# load validator cannot disagree about what a legal value is. The threshold is
# exclusive of 0 (a share of 0 is always reached) and inclusive of 1.
DEGRADED_WARN_MIN_FLOOR = 1
DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE = 0.0
DEGRADED_WARN_THRESHOLD_MAX = 1.0
class ClassifierConfig(StrictModel):
provider: str
base_url: str
@@ -1236,6 +1246,28 @@ class ClassifierConfig(StrictModel):
# Only read when mode == "local_decision".
decision: Optional["LocalDecisionConfig"] = None
# Refused at load rather than accepted and inert. A threshold above 1 can
# never be reached by a share, and one at or below 0 is always reached, so
# either reads as a working detector that has quietly stopped warning (or
# never stops). The minimum has to be a real sample size for the same
# reason: 0 would compute a share over an empty window.
@field_validator("degraded_warn_threshold")
@classmethod
def degraded_warn_threshold_in_range(cls, v: float) -> float:
if not (DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE < v <= DEGRADED_WARN_THRESHOLD_MAX):
raise ValueError(
"classifier.degraded_warn_threshold is a share of decisions "
"and must be in (0, 1]; a value above 1 can never fire"
)
return v
@field_validator("degraded_warn_min")
@classmethod
def degraded_warn_min_positive(cls, v: int) -> int:
if v < DEGRADED_WARN_MIN_FLOOR:
raise ValueError("classifier.degraded_warn_min must be at least 1")
return v
class LocalComputeConfig(StrictModel):
"""The outer gate over every call this router makes to local hardware.

View File

@@ -54,7 +54,7 @@ import requests
from dotenv import load_dotenv
from fastapi import BackgroundTasks, FastAPI, HTTPException, Request
from fastapi.responses import JSONResponse, Response, StreamingResponse
from openai import APIStatusError, OpenAI, OpenAIError
from openai import APIStatusError, APITimeoutError, OpenAI, OpenAIError
from pydantic import BaseModel, Field, field_validator
import admin
@@ -63,6 +63,15 @@ import events
import exploration
import local_decision
import local_encoder
from classifier_rejection import (
REASON_BELOW_CONFIDENCE,
REASON_PARSE,
REASON_PRIMARY_FAILED,
REASON_SKIPPED_PREFIX,
REASON_TIMEOUT,
REASON_TRANSPORT,
ClassifierRejected,
)
import local_energy
import logs
import prefix_probe
@@ -249,6 +258,22 @@ class TaskRequest(BaseModel):
required_context_tokens: Optional[int] = Field(None, ge=0)
class ClassifierAttempt(BaseModel):
"""What the primary classifier did with this request, kept as data.
Attached to every classification that went through ``classify()``, accepted
or not, so ``route_decisions`` can answer "how close to the floor was that
abstention" instead of only "it abstained". ``reject_reason`` is None when
the answer was accepted and one of ``classifier_rejection``'s codes when it
was not; ``confidence`` and ``coverage`` are whatever the classifier
produced, None where it produced no such number (a timeout has neither).
"""
confidence: Optional[float] = None
coverage: Optional[float] = None
reject_reason: Optional[str] = None
class Classification(BaseModel):
task_category: str
task_tier: int
@@ -267,6 +292,10 @@ class Classification(BaseModel):
"session_history",
"classifier_cloud",
] = "classifier"
# The primary classifier's own attempt, which a degraded source does not
# otherwise remember: a session_history row has confidence 0.0 by
# construction, so without this the rejected 0.43 would be lost.
attempt: Optional[ClassifierAttempt] = None
class Candidate(BaseModel):
@@ -456,7 +485,10 @@ def ensure_route_decisions(conn: sqlite3.Connection) -> None:
prefix_tokens_after_divergence INTEGER,
prefix_prev_message_count INTEGER,
agent TEXT,
parent_key TEXT
parent_key TEXT,
classifier_confidence REAL,
classifier_coverage REAL,
classifier_reject TEXT
)
"""
)
@@ -499,6 +531,14 @@ def ensure_route_decisions(conn: sqlite3.Connection) -> None:
# that decided it or link to a parent decision.
("agent", "TEXT"),
("parent_key", "TEXT"),
# What the primary classifier did with the request: its confidence,
# its coverage (local_decision only) and, when its answer was not
# used, why. Additive and NULL on every existing row, which is the
# honest value: the number that decided an abstention was only ever
# in a log line, so no earlier row can be backfilled.
("classifier_confidence", "REAL"),
("classifier_coverage", "REAL"),
("classifier_reject", "TEXT"),
):
if name not in existing:
conn.execute(f"ALTER TABLE route_decisions ADD COLUMN {name} {decl}")
@@ -884,6 +924,49 @@ def _provider_cost_multipliers() -> Optional[dict[str, float]]:
return multipliers
def _below_confidence(
mode: str,
section: str,
confidence: float,
floor: float,
coverage: Optional[float] = None,
) -> ClassifierRejected:
"""The one refusal for a floor miss, so message and reason cannot drift.
The text is what the ``fallback`` log line has always printed; the reason
and the number travel with it as data.
"""
return ClassifierRejected(
REASON_BELOW_CONFIDENCE,
f"{mode} confidence {confidence:.3f} is below "
f"classifier.{section}.confidence_min ({floor})",
confidence=confidence,
coverage=coverage,
)
def _attempt_from_exception(exc: BaseException) -> ClassifierAttempt:
"""Turn a primary-classifier failure into the record stored on the row.
Order matters: ``APITimeoutError`` is an ``OpenAIError``, and
``requests.Timeout`` is a ``RequestException``, so each specific case sits
ahead of its parent.
"""
if isinstance(exc, ClassifierRejected):
return ClassifierAttempt(
confidence=exc.confidence,
coverage=exc.coverage,
reject_reason=exc.reason,
)
if isinstance(exc, (requests.Timeout, APITimeoutError)):
return ClassifierAttempt(reject_reason=REASON_TIMEOUT)
if isinstance(exc, (requests.RequestException, OpenAIError)):
return ClassifierAttempt(reject_reason=REASON_TRANSPORT)
if isinstance(exc, (ValueError, KeyError, TypeError, json.JSONDecodeError)):
return ClassifierAttempt(reject_reason=REASON_PARSE)
return ClassifierAttempt(reject_reason=REASON_PRIMARY_FAILED)
class _ClassifierSkipped(Exception):
"""The local classifier was not attempted at all -- gaming mode, or an
already-open backoff circuit (see _local_classifier_skip_reason).
@@ -1040,9 +1123,8 @@ def _classify_via_local_encoder(task: str) -> Classification:
device=enc.device,
)
if confidence < enc.confidence_min:
raise RuntimeError(
f"local_encoder confidence {confidence:.3f} is below "
f"classifier.encoder.confidence_min ({enc.confidence_min})"
raise _below_confidence(
"local_encoder", "encoder", confidence, enc.confidence_min
)
result = Classification(
task_category=category,
@@ -1071,9 +1153,8 @@ def _classify_via_local_encoder(task: str) -> Classification:
device=enc.device,
)
if confidence < enc.confidence_min:
raise RuntimeError(
f"local_encoder confidence {confidence:.3f} is below "
f"classifier.encoder.confidence_min ({enc.confidence_min})"
raise _below_confidence(
"local_encoder", "encoder", confidence, enc.confidence_min
)
return Classification(
task_category=category,
@@ -1135,19 +1216,21 @@ def _classify_via_local_decision(
tier_future = pool.submit(
_classify_tier_via_local_decision, user_content
)
category, confidence, _coverage = cat_future.result()
category, confidence, coverage = cat_future.result()
if confidence < dec.confidence_min:
raise RuntimeError(
f"local_decision confidence {confidence:.3f} "
f"is below classifier.decision.confidence_min "
f"({dec.confidence_min})"
raise _below_confidence(
"local_decision",
"decision",
confidence,
dec.confidence_min,
coverage,
)
try:
tier = tier_future.result()
except Exception:
tier = cfg.classifier.fallback_tier
else:
category, confidence, _coverage = (
category, confidence, coverage = (
local_decision.classify_category(
user_content,
base_url=dec.base_url,
@@ -1158,10 +1241,12 @@ def _classify_via_local_decision(
)
)
if confidence < dec.confidence_min:
raise RuntimeError(
f"local_decision confidence {confidence:.3f} "
f"is below classifier.decision.confidence_min "
f"({dec.confidence_min})"
raise _below_confidence(
"local_decision",
"decision",
confidence,
dec.confidence_min,
coverage,
)
_log_local_energy(
@@ -1175,6 +1260,9 @@ def _classify_via_local_decision(
required_context_tokens=0,
confidence=confidence,
source="classifier",
attempt=ClassifierAttempt(
confidence=confidence, coverage=coverage
),
)
except Exception as exc: # noqa: BLE001 — meter the real draw on ANY failure
if isinstance(exc, requests.RequestException):
@@ -1208,18 +1296,21 @@ def _classify_via_local_decision(
tier_future = pool.submit(
_classify_tier_via_local_decision, user_content
)
category, confidence, _coverage = cat_future.result()
category, confidence, coverage = cat_future.result()
if confidence < dec.confidence_min:
raise RuntimeError(
f"local_decision confidence {confidence:.3f} is below "
f"classifier.decision.confidence_min ({dec.confidence_min})"
raise _below_confidence(
"local_decision",
"decision",
confidence,
dec.confidence_min,
coverage,
)
try:
tier = tier_future.result()
except Exception:
tier = cfg.classifier.fallback_tier
else:
category, confidence, _coverage = local_decision.classify_category(
category, confidence, coverage = local_decision.classify_category(
user_content,
base_url=dec.base_url,
model=dec.model,
@@ -1228,9 +1319,12 @@ def _classify_via_local_decision(
coverage_min=dec.coverage_min,
)
if confidence < dec.confidence_min:
raise RuntimeError(
f"local_decision confidence {confidence:.3f} is below "
f"classifier.decision.confidence_min ({dec.confidence_min})"
raise _below_confidence(
"local_decision",
"decision",
confidence,
dec.confidence_min,
coverage,
)
tier = _classify_tier_via_local_decision(user_content)
return Classification(
@@ -1239,6 +1333,9 @@ def _classify_via_local_decision(
required_context_tokens=0,
confidence=confidence,
source="classifier",
attempt=ClassifierAttempt(
confidence=confidence, coverage=coverage
),
)
except Exception as exc: # noqa: BLE001
if isinstance(exc, requests.RequestException):
@@ -1512,22 +1609,30 @@ def _degraded_classification(
session_key: Optional[str],
system_prompt: str,
user_content: str,
attempt: Optional[ClassifierAttempt] = None,
) -> Classification:
"""The cascade, then the static guess if every step of it misses.
One place, because the three callers used to carry byte-identical copies
of the fallback construction and a fourth was about to.
``attempt`` is what the primary classifier did before the cascade took
over. It is stamped on whatever the cascade returns, because every
degraded source is a borrowed or guessed label that has forgotten why the
primary was not used.
"""
degraded = _classify_cascade(session_key, system_prompt, user_content)
if degraded is not None:
return degraded
return Classification(
task_category=cfg.classifier.fallback_category,
task_tier=cfg.classifier.fallback_tier,
required_context_tokens=0,
confidence=0.0,
source="fallback",
)
result = _classify_cascade(session_key, system_prompt, user_content)
if result is None:
result = Classification(
task_category=cfg.classifier.fallback_category,
task_tier=cfg.classifier.fallback_tier,
required_context_tokens=0,
confidence=0.0,
source="fallback",
)
if attempt is not None:
result = result.model_copy(update={"attempt": attempt})
return result
def _provider_client(provider: str) -> OpenAI:
@@ -1613,14 +1718,19 @@ def classify(task: str, context: Optional[str]) -> Classification:
started = time.perf_counter()
try:
resp = _classify_via_configured_mode(task, system_prompt, user_content)
except _ClassifierSkipped:
except _ClassifierSkipped as skipped:
# Already logged specifically (classify_local_skipped, with WHY --
# gaming_mode or backoff). No generic "fallback" line here: that one
# means an attempt was made and failed, and a deliberate skip is
# neither -- conflating the two would tell an operator reading raw
# logs that the local classifier is failing when it was never asked.
return _degraded_classification(
_current_session_key.get(), system_prompt, user_content
_current_session_key.get(),
system_prompt,
user_content,
attempt=ClassifierAttempt(
reject_reason=f"{REASON_SKIPPED_PREFIX}{skipped}"
),
)
except (
OpenAIError,
@@ -1649,7 +1759,10 @@ def classify(task: str, context: Optional[str]) -> Classification:
ms=_ms(started),
)
return _degraded_classification(
_current_session_key.get(), system_prompt, user_content
_current_session_key.get(),
system_prompt,
user_content,
attempt=_attempt_from_exception(e),
)
# The single blocking LLM call on the request path, so its latency is the
@@ -1671,6 +1784,12 @@ def classify(task: str, context: Optional[str]) -> Classification:
# modes, which is correct -- there is no local-endpoint circuit for them
# to close.
_record_success_cooldown()
if resp.attempt is None:
# local_decision stamps its own, because it also knows the coverage;
# every other mode has only a confidence to record.
resp = resp.model_copy(
update={"attempt": ClassifierAttempt(confidence=resp.confidence)}
)
return resp
@@ -2277,6 +2396,7 @@ def persist_route_decision(
prefix_divergence: Optional[prefix_probe.Divergence] = None,
agent: Optional[str] = None,
parent_key: Optional[str] = None,
attempt_of: Optional[Classification] = None,
) -> Optional[int]:
"""Record one routing decision to route_decisions, best-effort and gated.
@@ -2298,6 +2418,11 @@ def persist_route_decision(
the RouteResponse would otherwise say — the chat re-route's
Classification reads ``source='override'`` even though the classifier did
decide the request, so the caller passes the first route's actual source.
``attempt_of`` is the same idea for the classifier's own attempt (its
confidence, coverage and rejection reason): the re-routed override carries
none of it, so the chat path passes the Classification as the first route
produced it. Absent, the attempt is read from ``classification``.
"""
# The gate is read before anything can fail: turning the table off must be
# a guaranteed no-op even on a broken DB.
@@ -2337,6 +2462,8 @@ def persist_route_decision(
if clf is not None and classification_source is None:
classification_source = clf.source
attempt_src = attempt_of if attempt_of is not None else clf
attempt = attempt_src.attempt if attempt_src is not None else None
# An override-created Classification never consulted the classifier, so it
# has no classifier latency worth recording — even when a caller passed a
# timing value (route_endpoint's whole-route ms would otherwise leak in).
@@ -2386,8 +2513,9 @@ def persist_route_decision(
flex_preference, flex_swapped, flex_forced, exploration,
request_id, pinch_original_tokens, pinch_final_tokens, profile,
prefix_divergence_index, prefix_tokens_after_divergence,
prefix_prev_message_count, agent, parent_key
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
prefix_prev_message_count, agent, parent_key,
classifier_confidence, classifier_coverage, classifier_reject
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
""",
(
observed_at,
@@ -2424,6 +2552,9 @@ def persist_route_decision(
prefix_prev_count,
agent,
parent_key,
attempt.confidence if attempt is not None else None,
attempt.coverage if attempt is not None else None,
attempt.reject_reason if attempt is not None else None,
),
)
decision_id: Optional[int] = int(cursor.lastrowid)
@@ -2469,6 +2600,15 @@ def persist_route_decision(
"prefix_prev_message_count": prefix_prev_count,
"agent": agent,
"parent_key": parent_key,
"classifier_confidence": (
attempt.confidence if attempt is not None else None
),
"classifier_coverage": (
attempt.coverage if attempt is not None else None
),
"classifier_reject": (
attempt.reject_reason if attempt is not None else None
),
}
)
return decision_id
@@ -4616,6 +4756,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks, request:
# the category/tier are reused. The required context is this
# request's own measurement.
classified_src = "cached"
classifier_attempt = None
classifier_ms = 0
ctx_src = "measured"
decision = route(
@@ -4660,6 +4801,9 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks, request:
# The classifier's own verdict, before the re-route below rewrites
# the Classification's source to 'override'.
classified_src = decision.classification.source
# Same reason as the source above: the re-route below replaces the
# Classification with an override that remembers none of this.
classifier_attempt = decision.classification
classifier_ms = _ms(classify_started)
ctx_src = "classifier"
if measured > decision.classification.required_context_tokens:
@@ -4790,6 +4934,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks, request:
streamed=streamed,
classification_source=classified_src,
classifier_ms=classifier_ms,
attempt_of=classifier_attempt,
pinch_original_tokens=pinch_stats.get("original_tokens")
if pinch_stats is not None else None,
pinch_final_tokens=pinch_stats.get("final_tokens")

View File

@@ -39,6 +39,12 @@ from typing import Any, Final, Optional
import requests
from classifier_rejection import (
REASON_BELOW_COVERAGE,
REASON_NO_LOGPROBS,
ClassifierRejected,
)
_DEFAULT_COVERAGE_MIN: float = 0.3
# Decision-category descriptions tuned against the measured confusion
@@ -116,14 +122,19 @@ def parse_logprobs(
Raises
------
RuntimeError
When ``coverage < coverage_min`` or when no logprobs are found.
ClassifierRejected
(a :exc:`RuntimeError`) When ``coverage < coverage_min`` or when no
logprobs are found. ``reason`` says which, and a coverage rejection
also carries the coverage and the would-be confidence, because how
far below the floor an answer fell is what tuning ``coverage_min``
needs.
"""
logprobs_list: list[dict[str, Any]] = response_json.get("logprobs", [])
if not logprobs_list:
raise RuntimeError(
"parse_logprobs: no logprobs found in classifier response"
raise ClassifierRejected(
REASON_NO_LOGPROBS,
"parse_logprobs: no logprobs found in classifier response",
)
# Accumulate exp(logprob) mass per option letter.
@@ -147,9 +158,13 @@ def parse_logprobs(
total = sum(mass.values())
if total < coverage_min:
raise RuntimeError(
best = max(mass.values(), default=0.0)
raise ClassifierRejected(
REASON_BELOW_COVERAGE,
f"parse_logprobs: coverage {total:.4f} below "
f"minimum {coverage_min}"
f"minimum {coverage_min}",
confidence=(best / total) if total > 0 else None,
coverage=total,
)
# Winner is the option with the most accumulated mass.

View File

@@ -1773,12 +1773,44 @@ def classifier_degradation_warning(conn: sqlite3.Connection, cfg: Any) -> list[s
return []
return [
f"{share:.0%} of the last {total} classifications came from a degraded "
f"source ({degraded} of {total}) — the local classifier has been "
f"failing. Routing still works on borrowed categories, but their "
f"outcomes are excluded from proficiency."
f"source ({degraded} of {total}). {_declined_for(conn)}Routing still "
f"works on borrowed categories, but their outcomes are excluded from "
f"proficiency."
]
def _declined_for(conn: sqlite3.Connection) -> str:
"""The recorded reasons the primary classifier's answer was not used.
The warning used to end "the local classifier has been failing", which is
wrong for the case that now dominates: under ``local_decision`` the
classifier answers and is declined for being unsure, and nothing is down.
Saying which reason is what tells the operator whether to look at the
endpoint (``transport_error``, ``timeout``) or at a threshold
(``below_confidence_min``). Empty when no reasons were recorded, which is
every row from before the column existed, and when the column is absent on
a database that has not been through dispatcher start-up.
"""
if not _has_column(conn, "route_decisions", "classifier_reject"):
return ""
rows = conn.execute(
"""
SELECT classifier_reject AS reason, COUNT(*) AS n
FROM route_decisions
WHERE classifier_reject IS NOT NULL
AND classification_source IN
('fallback', 'session_stale', 'session_history',
'classifier_cloud')
AND observed_at >= datetime('now', '-24 hours')
GROUP BY classifier_reject
ORDER BY n DESC, classifier_reject
"""
).fetchall()
if not rows:
return ""
return "Declined for: " + ", ".join(f"{r['n']} {r['reason']}" for r in rows) + ". "
# Gate prefixes whose failure means a DERIVED field is broken, rather than the
# operator having switched something off.
#
@@ -2935,6 +2967,16 @@ _CONVERSATION_IDENTITY_COLUMNS: Final = (
"parent_key",
)
# The classifier's own attempt (confidence, coverage, why it was rejected).
# Added by ALTER at dispatcher start-up like the groups above, so probed and
# backfilled the same way: the live router.db has none of them until its next
# restart, and a missing column must read as NULL rather than 500 the payload.
_CLASSIFIER_ATTEMPT_COLUMNS: Final = (
"classifier_confidence",
"classifier_coverage",
"classifier_reject",
)
def recent_decisions(
conn: sqlite3.Connection,
@@ -2961,8 +3003,14 @@ def recent_decisions(
for column in _CONVERSATION_IDENTITY_COLUMNS
if _has_column(conn, "route_decisions", column)
]
attempt_present = [
column
for column in _CLASSIFIER_ATTEMPT_COLUMNS
if _has_column(conn, "route_decisions", column)
]
probe_select = "".join(f", {column}" for column in present)
identity_select = "".join(f", {column}" for column in identity_present)
attempt_select = "".join(f", {column}" for column in attempt_present)
rows = [
dict(row)
for row in conn.execute(
@@ -2975,7 +3023,7 @@ def recent_decisions(
rejected_reason, session_key, tools, images, json_mode, streamed,
flex_preference, flex_swapped, flex_forced,
exploration, request_id,
pinch_original_tokens, pinch_final_tokens, profile{probe_select}{identity_select}
pinch_original_tokens, pinch_final_tokens, profile{probe_select}{identity_select}{attempt_select}
FROM route_decisions
ORDER BY id DESC
LIMIT ?
@@ -2991,6 +3039,8 @@ def recent_decisions(
row.setdefault(column, None)
for column in _CONVERSATION_IDENTITY_COLUMNS:
row.setdefault(column, None)
for column in _CLASSIFIER_ATTEMPT_COLUMNS:
row.setdefault(column, None)
return rows

View File

@@ -249,6 +249,14 @@ def decision_row(r: dict) -> dict:
"request_id": r.get("request_id"),
"session_key": r.get("session_key"),
"agent": r.get("agent"),
# The classifier's own attempt: how confident it was, how much option
# mass sat behind that, and why its answer was not used. Detail popup
# only, like the prefix probe: the three read together ("0.43, below
# the 0.5 floor, so session_history routed it") and `confidence` above
# cannot say this on a chat row, where it is the override's 1.0.
"classifier_confidence": r.get("classifier_confidence"),
"classifier_coverage": r.get("classifier_coverage"),
"classifier_reject": r.get("classifier_reject"),
}

View File

@@ -2,6 +2,9 @@
"agent": null,
"candidates_considered": 2,
"classification_source": "classifier",
"classifier_confidence": null,
"classifier_coverage": null,
"classifier_reject": null,
"confidence": 0.9,
"est_cost_usd": 0.00016,
"est_proficiency": 0.5,

View File

@@ -0,0 +1,532 @@
"""The classifier's own attempt reaches route_decisions.
Before this, an abstention left one number behind and it was in a log line:
``local_decision confidence 0.431 is below classifier.decision.confidence_min
(0.5)``. The database said only ``classification_source = session_history``, and
``route_decisions.confidence`` read 1.0 on 99.9% of chat rows because the chat
path re-routes through the override branch, which hard-codes it. So
``confidence_min`` could not be tuned from data: nobody could say whether the
abstentions sat just under the floor or nowhere near it.
These tests pin the three columns end to end: ``classifier_confidence``,
``classifier_coverage`` and ``classifier_reject``. The chat-path test runs the
REAL ``classify()`` (only the Ollama call is stubbed), because the stub that
``test_no_header_snapshot`` installs replaces ``classify`` itself and would hide
exactly the stamping this file is about.
"""
import json
import sqlite3
from pathlib import Path
from types import SimpleNamespace
import openai
import pytest
import requests
import test_no_header_snapshot as snap
from test_route_decisions import _schema_minus_profile_column
import dispatcher
import local_decision
import prefix_probe
import session_cache
from classifier_rejection import (
REASON_BELOW_CONFIDENCE,
REASON_BELOW_COVERAGE,
REASON_NO_LOGPROBS,
REASON_PARSE,
REASON_PRIMARY_FAILED,
REASON_TIMEOUT,
REASON_TRANSPORT,
ClassifierRejected,
)
from dispatcher import Classification, ClassifierAttempt
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
ATTEMPT_COLUMNS = ("classifier_confidence", "classifier_coverage", "classifier_reject")
@pytest.fixture(autouse=True)
def _clean_state(monkeypatch):
session_cache.clear()
prefix_probe._store.clear()
monkeypatch.setattr(dispatcher, "_last_classifier_failure", 0.0)
dispatcher._provider_refusal_since.clear()
token = dispatcher._current_session_key.set(None)
yield
dispatcher._current_session_key.reset(token)
dispatcher._provider_refusal_since.clear()
session_cache.clear()
# The chat-path test posts the snapshot's own payload, and the probe keeps
# per-session fingerprints in process memory: left behind, the headerless
# snapshot test (which runs after this file) sees a "previous turn" and
# records a non-NULL prefix divergence.
prefix_probe._store.clear()
def _use_local_decision(monkeypatch, *, confidence_min=0.5):
"""local_decision as the primary, unmetered, with nothing skipped."""
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
monkeypatch.setattr(
dispatcher.cfg.classifier,
"decision",
SimpleNamespace(
base_url="http://localhost:11434",
model="stub-model",
num_ctx=8192,
timeout_s=10,
confidence_min=confidence_min,
coverage_min=0.3,
tier_enabled=False,
),
)
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", False)
monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True)
def _answers(monkeypatch, result):
"""Stub the Ollama call: return ``result`` or raise it if it is an exception."""
def fake(*args, **kwargs):
if isinstance(result, BaseException):
raise result
return result
monkeypatch.setattr(local_decision, "classify_category", fake)
# --- the raise sites carry the numbers -------------------------------------
def test_a_floor_miss_records_the_confidence_that_missed_it(monkeypatch):
_use_local_decision(monkeypatch)
_answers(monkeypatch, ("debugging", 0.431, 0.874))
got = dispatcher.classify("fix the failing test", None)
assert got.source == "fallback"
assert got.attempt == ClassifierAttempt(
confidence=0.431, coverage=0.874, reject_reason=REASON_BELOW_CONFIDENCE
)
def test_a_coverage_miss_records_the_coverage_and_the_would_be_confidence(monkeypatch):
"""Both gates are recorded, because tuning one without the other misleads."""
_use_local_decision(monkeypatch)
# Total option mass 0.0498 sits under coverage_min (0.3); letter A holds all of it.
low_mass = {
"logprobs": [
{"token": "A", "logprob": -3.0, "top_logprobs": [{"token": "A", "logprob": -3.0}]}
]
}
monkeypatch.setattr(
local_decision,
"classify_category",
lambda *a, **k: local_decision.parse_logprobs(low_mass, ["A", "B"], coverage_min=0.3),
)
got = dispatcher.classify("fix the failing test", None)
assert got.attempt.reject_reason == REASON_BELOW_COVERAGE
assert got.attempt.coverage == pytest.approx(0.0498, abs=1e-4)
assert got.attempt.confidence == pytest.approx(1.0)
def test_parse_logprobs_raises_a_runtime_error_with_its_original_message():
"""Subclassing RuntimeError keeps every existing handler and log line intact."""
with pytest.raises(ClassifierRejected, match="no logprobs found") as no_lp:
local_decision.parse_logprobs({}, ["A", "B"])
assert isinstance(no_lp.value, RuntimeError)
assert no_lp.value.reason == REASON_NO_LOGPROBS
thin = {"logprobs": [{"token": "A", "top_logprobs": [{"token": "A", "logprob": -4.0}]}]}
with pytest.raises(RuntimeError, match=r"coverage 0\.0183 below minimum 0\.3") as thin_err:
local_decision.parse_logprobs(thin, ["A", "B"], coverage_min=0.3)
assert thin_err.value.reason == REASON_BELOW_COVERAGE
def test_the_rejection_message_is_unchanged(monkeypatch):
"""The `fallback` journal line prints this text; operators grep for it."""
_use_local_decision(monkeypatch)
_answers(monkeypatch, ("debugging", 0.431, 0.874))
with pytest.raises(ClassifierRejected) as err:
dispatcher._classify_via_local_decision("sys", "do a thing")
assert str(err.value) == (
"local_decision confidence 0.431 is below classifier.decision.confidence_min (0.5)"
)
# --- every failure class gets a code ---------------------------------------
@pytest.mark.parametrize(
"failure, reason",
[
(requests.ConnectionError("refused"), REASON_TRANSPORT),
(requests.Timeout("slow"), REASON_TIMEOUT),
# The SDK's own timeout type, which is an OpenAIError subclass and so
# would be filed as a transport error if the order of the checks slipped.
(
openai.APITimeoutError(request=SimpleNamespace(method="POST", url="http://x")),
REASON_TIMEOUT,
),
(ValueError("not json"), REASON_PARSE),
(KeyError("choices"), REASON_PARSE),
(json.JSONDecodeError("bad", "{", 0), REASON_PARSE),
(RuntimeError("something else"), REASON_PRIMARY_FAILED),
],
ids=lambda p: type(p).__name__ if isinstance(p, BaseException) else p,
)
def test_each_failure_class_has_a_stable_code_and_no_invented_numbers(
monkeypatch, failure, reason
):
_use_local_decision(monkeypatch)
_answers(monkeypatch, failure)
got = dispatcher.classify("fix the failing test", None)
assert got.source == "fallback"
assert got.attempt == ClassifierAttempt(reject_reason=reason)
@pytest.mark.parametrize("why", ["gaming_mode", "backoff"])
def test_a_skip_is_recorded_as_a_skip_not_a_failure(monkeypatch, why):
_use_local_decision(monkeypatch)
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: why)
got = dispatcher.classify("fix the failing test", None)
assert got.attempt == ClassifierAttempt(reject_reason=f"skipped_{why}")
def test_an_accepted_answer_carries_its_numbers_and_no_reason(monkeypatch):
_use_local_decision(monkeypatch)
_answers(monkeypatch, ("debugging", 0.83, 0.91))
got = dispatcher.classify("fix the failing test", None)
assert got.source == "classifier"
assert got.attempt == ClassifierAttempt(confidence=0.83, coverage=0.91, reject_reason=None)
def test_modes_without_a_coverage_still_record_their_confidence(monkeypatch):
"""The generic stamp: local_llm, cloud_llm and the encoder have no coverage."""
monkeypatch.setattr(
dispatcher,
"_classify_via_configured_mode",
lambda *a, **k: Classification(
task_category="coding_general",
task_tier=2,
required_context_tokens=100,
confidence=0.9,
),
)
got = dispatcher.classify("fix the failing test", None)
assert got.attempt == ClassifierAttempt(confidence=0.9)
# --- the degraded builder keeps what the cascade forgot ---------------------
def test_the_cascade_result_is_stamped_without_changing_what_it_routes_on(monkeypatch):
borrowed = Classification(
task_category="debugging",
task_tier=3,
required_context_tokens=0,
confidence=0.0,
source="session_history",
)
monkeypatch.setattr(dispatcher, "_classify_cascade", lambda *a: borrowed)
attempt = ClassifierAttempt(confidence=0.431, reject_reason=REASON_BELOW_CONFIDENCE)
got = dispatcher._degraded_classification("k", "sys", "user", attempt=attempt)
assert got.attempt == attempt
assert (got.source, got.task_category, got.task_tier, got.confidence) == (
"session_history",
"debugging",
3,
0.0,
)
assert borrowed.attempt is None, "the cascade's own object must not be mutated"
def test_the_static_guess_is_stamped_too(monkeypatch):
monkeypatch.setattr(dispatcher, "_classify_cascade", lambda *a: None)
attempt = ClassifierAttempt(reject_reason=REASON_TIMEOUT)
got = dispatcher._degraded_classification("k", "sys", "user", attempt=attempt)
assert got.source == "fallback"
assert got.attempt == attempt
assert dispatcher._degraded_classification("k", "sys", "user").attempt is None
# --- the row ----------------------------------------------------------------
def _fresh_db(tmp_path, monkeypatch):
db_path = tmp_path / "attempt.db"
conn = sqlite3.connect(db_path)
conn.executescript(SCHEMA_SQL)
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(db_path))
monkeypatch.setattr(dispatcher.cfg.logging, "log_route_decisions", True)
return db_path
def _rows(db_path):
conn = sqlite3.connect(db_path)
conn.row_factory = sqlite3.Row
try:
return [dict(r) for r in conn.execute("SELECT * FROM route_decisions ORDER BY id")]
finally:
conn.close()
def _degraded(attempt):
return Classification(
task_category="debugging",
task_tier=2,
required_context_tokens=0,
confidence=0.0,
source="session_history",
attempt=attempt,
)
def test_persist_writes_the_attempt_from_the_classification(tmp_path, monkeypatch):
db_path = _fresh_db(tmp_path, monkeypatch)
attempt = ClassifierAttempt(
confidence=0.431, coverage=0.874, reject_reason=REASON_BELOW_CONFIDENCE
)
dispatcher.persist_route_decision(
"route", classification=_degraded(attempt), selected_model="m", selected_provider="p"
)
(row,) = _rows(db_path)
assert row["classifier_confidence"] == pytest.approx(0.431)
assert row["classifier_coverage"] == pytest.approx(0.874)
assert row["classifier_reject"] == REASON_BELOW_CONFIDENCE
assert row["confidence"] == 0.0, "the routing classification's own confidence is untouched"
def test_persist_prefers_attempt_of_when_the_classification_was_rerouted(tmp_path, monkeypatch):
"""The chat path rewrites `classification` to an override that remembers nothing."""
db_path = _fresh_db(tmp_path, monkeypatch)
override = Classification(
task_category="debugging",
task_tier=2,
required_context_tokens=90000,
confidence=1.0,
source="override",
)
first = _degraded(
ClassifierAttempt(confidence=0.431, reject_reason=REASON_BELOW_CONFIDENCE)
)
dispatcher.persist_route_decision(
"chat",
classification=override,
classification_source="session_history",
attempt_of=first,
selected_model="m",
selected_provider="p",
)
(row,) = _rows(db_path)
assert row["confidence"] == 1.0, "legacy column: the override's hard-coded value"
assert row["classifier_confidence"] == pytest.approx(0.431)
assert row["classifier_reject"] == REASON_BELOW_CONFIDENCE
def test_persist_without_an_attempt_leaves_all_three_null(tmp_path, monkeypatch):
db_path = _fresh_db(tmp_path, monkeypatch)
override = Classification(
task_category="debugging",
task_tier=2,
required_context_tokens=100,
confidence=1.0,
source="override",
)
dispatcher.persist_route_decision(
"route", classification=override, selected_model="m", selected_provider="p"
)
(row,) = _rows(db_path)
assert [row[c] for c in ATTEMPT_COLUMNS] == [None, None, None]
def test_the_live_event_carries_the_same_three_keys(tmp_path, monkeypatch):
_fresh_db(tmp_path, monkeypatch)
published = []
monkeypatch.setattr(dispatcher.events, "publish_decision", published.append)
attempt = ClassifierAttempt(confidence=0.431, reject_reason=REASON_BELOW_CONFIDENCE)
dispatcher.persist_route_decision(
"route", classification=_degraded(attempt), selected_model="m", selected_provider="p"
)
(event,) = published
assert event["classifier_confidence"] == pytest.approx(0.431)
assert event["classifier_coverage"] is None
assert event["classifier_reject"] == REASON_BELOW_CONFIDENCE
# --- migration ---------------------------------------------------------------
def test_a_live_table_gains_the_columns_once_and_keeps_its_rows(tmp_path):
conn = sqlite3.connect(tmp_path / "live.db")
conn.executescript(_schema_minus_profile_column())
conn.execute(
"INSERT INTO route_decisions (observed_at, kind) VALUES ('2026-10-01T00:00:00+00:00', 'chat')"
)
conn.commit()
before = {r[1] for r in conn.execute("PRAGMA table_info(route_decisions)")}
assert not set(ATTEMPT_COLUMNS) & before
dispatcher.ensure_route_decisions(conn)
dispatcher.ensure_route_decisions(conn) # idempotent
cols = [r[1] for r in conn.execute("PRAGMA table_info(route_decisions)")]
for column in ATTEMPT_COLUMNS:
assert cols.count(column) == 1
old = conn.execute(f"SELECT {', '.join(ATTEMPT_COLUMNS)} FROM route_decisions").fetchone()
assert tuple(old) == (None, None, None), "an old row cannot be backfilled, and says so"
conn.close()
@pytest.mark.skipif(
sqlite3.sqlite_version_info < (3, 35),
reason="ALTER TABLE ... DROP COLUMN needs SQLite 3.35",
)
def test_recent_decisions_reads_a_database_that_has_not_been_migrated(tmp_path):
"""/metrics and the admin page share metrics.py; the live DB lacks the columns
until the router restarts, and that must read as NULL rather than a 500.
The shape under test is a CURRENT schema minus only these three columns,
which is what the live router.db is between the deploy and the restart.
"""
import metrics
conn = sqlite3.connect(tmp_path / "old.db")
conn.row_factory = sqlite3.Row
conn.executescript(SCHEMA_SQL)
for column in ATTEMPT_COLUMNS:
conn.execute(f"ALTER TABLE route_decisions DROP COLUMN {column}")
conn.execute(
"INSERT INTO route_decisions (observed_at, kind) VALUES ('2026-10-01T00:00:00+00:00', 'chat')"
)
conn.commit()
assert not set(ATTEMPT_COLUMNS) & {r[1] for r in conn.execute("PRAGMA table_info(route_decisions)")}
(row,) = metrics.recent_decisions(conn, limit=5)
conn.close()
for column in ATTEMPT_COLUMNS:
assert column in row and row[column] is None
# --- the chat path, with the real classify() ---------------------------------
def test_chat_rows_record_the_attempt_even_though_the_reroute_hides_it(tmp_path, monkeypatch):
"""Turn 1 is accepted; turn 2 of the same session is declined and replays turn 1's label.
Both rows must say what the classifier did. The re-route to measured
context replaces the Classification with an override in between, which is
why ``confidence`` could never have shown this.
"""
real_classify = dispatcher.classify
client, db_path = snap.make_router(tmp_path, monkeypatch)
monkeypatch.setattr(dispatcher, "classify", real_classify)
_use_local_decision(monkeypatch)
answers = iter([("coding_general", 0.91, 0.95), ("coding_general", 0.431, 0.874)])
monkeypatch.setattr(local_decision, "classify_category", lambda *a, **k: next(answers))
for _ in range(2):
resp = snap.post_no_header(client)
assert resp.status_code == 200, resp.text
first, second = _rows(db_path)
assert first["classification_source"] == "classifier"
assert first["classifier_confidence"] == pytest.approx(0.91)
assert first["classifier_coverage"] == pytest.approx(0.95)
assert first["classifier_reject"] is None
assert second["classification_source"] == "session_history"
assert second["classifier_confidence"] == pytest.approx(0.431)
assert second["classifier_coverage"] == pytest.approx(0.874)
assert second["classifier_reject"] == REASON_BELOW_CONFIDENCE
# --- the warning knobs refuse values that could never behave -------------------
def _classifier_with(**overrides):
"""The real loaded classifier block with only the knob under test changed.
ClassifierConfig has required fields, so building one bare would fail on
those and bury what the test is about.
"""
from config import ClassifierConfig
return ClassifierConfig(**{**dispatcher.cfg.classifier.model_dump(), **overrides})
@pytest.mark.parametrize("threshold", [0.0, -0.1, 1.5, 5, 50])
def test_degraded_warn_threshold_outside_a_share_is_refused_at_load(threshold):
"""`80` meant as 80% once shipped as a raw number into another knob and made
every score read as below threshold; a share above 1 here is the same
mistake and would make the warning inert instead."""
from pydantic import ValidationError
with pytest.raises(ValidationError, match="degraded_warn_threshold"):
_classifier_with(degraded_warn_threshold=threshold)
@pytest.mark.parametrize("threshold", [0.01, 0.2, 0.5, 1.0])
def test_degraded_warn_threshold_accepts_any_real_share(threshold):
got = _classifier_with(degraded_warn_threshold=threshold)
assert got.degraded_warn_threshold == threshold
@pytest.mark.parametrize("minimum", [0, -3])
def test_degraded_warn_min_must_be_a_real_sample_size(minimum):
from pydantic import ValidationError
with pytest.raises(ValidationError, match="degraded_warn_min"):
_classifier_with(degraded_warn_min=minimum)
def test_degraded_warn_bound_constants_are_the_validators_boundaries():
"""A runtime admin write skips Pydantic, so its registry imports these named
bounds. They are only trustworthy if they ARE the load validator's edges:
the exclusive low and the floor are refused, the high and floor+0 accepted."""
from pydantic import ValidationError
from config import (
DEGRADED_WARN_MIN_FLOOR,
DEGRADED_WARN_THRESHOLD_MAX,
DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE,
)
with pytest.raises(ValidationError):
_classifier_with(degraded_warn_threshold=DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE)
assert (
_classifier_with(degraded_warn_threshold=DEGRADED_WARN_THRESHOLD_MAX).degraded_warn_threshold
== DEGRADED_WARN_THRESHOLD_MAX
)
with pytest.raises(ValidationError):
_classifier_with(degraded_warn_min=DEGRADED_WARN_MIN_FLOOR - 1)
assert (
_classifier_with(degraded_warn_min=DEGRADED_WARN_MIN_FLOOR).degraded_warn_min
== DEGRADED_WARN_MIN_FLOOR
)

View File

@@ -270,3 +270,67 @@ def test_degradation_warning_silent_when_healthy(db):
)
_seed_sources(db, [("classifier", 24), ("fallback", 1)])
assert classifier_degradation_warning(db, cfg) == []
def _seed_declined(conn, source, reason, count):
for _ in range(count):
conn.execute(
"""
INSERT INTO route_decisions
(kind, task_category, task_tier, required_context_tokens,
selected_model, selected_provider, classification_source,
classifier_reject, observed_at)
VALUES ('chat','general_chat',2,0,'m','neuralwatt',?,?,?)
""",
(source, reason, _now().isoformat()),
)
conn.commit()
def test_degradation_warning_says_why_the_classifier_was_not_used(db):
""""The local classifier has been failing" was wrong for local_decision, where
nothing fails and the answer is declined for being unsure."""
from metrics import classifier_degradation_warning
cfg = SimpleNamespace(
classifier=SimpleNamespace(degraded_warn_min=20, degraded_warn_threshold=0.2)
)
_seed_sources(db, [("classifier", 60)])
_seed_declined(db, "session_history", "below_confidence_min", 18)
_seed_declined(db, "fallback", "below_coverage_min", 5)
_seed_declined(db, "session_history", "below_confidence_min", 2)
(warning,) = classifier_degradation_warning(db, cfg)
assert "degraded source" in warning # the TUI's classifier-degraded matcher
assert "Declined for: 20 below_confidence_min, 5 below_coverage_min." in warning
assert "failing" not in warning
def test_degradation_warning_without_recorded_reasons_makes_no_claim_about_failure(db):
from metrics import classifier_degradation_warning
cfg = SimpleNamespace(
classifier=SimpleNamespace(degraded_warn_min=20, degraded_warn_threshold=0.5)
)
_seed_sources(db, [("fallback", 15), ("classifier", 10)])
(warning,) = classifier_degradation_warning(db, cfg)
assert "Declined for" not in warning
assert "failing" not in warning
assert "degraded source (15 of 25)" in warning
def test_degradation_warning_survives_a_database_without_the_column(db, monkeypatch):
import metrics
cfg = SimpleNamespace(
classifier=SimpleNamespace(degraded_warn_min=20, degraded_warn_threshold=0.5)
)
_seed_sources(db, [("fallback", 15), ("classifier", 10)])
monkeypatch.setattr(metrics, "_has_column", lambda *a, **k: False)
(warning,) = metrics.classifier_degradation_warning(db, cfg)
assert "60%" in warning and "Declined for" not in warning

View File

@@ -72,6 +72,9 @@ ROUTE_DECISIONS_COLUMNS = [
"prefix_prev_message_count",
"agent",
"parent_key",
"classifier_confidence",
"classifier_coverage",
"classifier_reject",
]

View File

@@ -75,6 +75,10 @@ SCHEMA_TO_MODEL_KEYS = {
"prefix_tokens_after_divergence": "prefix_tokens_after_divergence",
"prefix_prev_message_count": "prefix_prev_message_count",
"agent": "agent",
# The classifier's own attempt. Detail popup, like the prefix probe above.
"classifier_confidence": "classifier_confidence",
"classifier_coverage": "classifier_coverage",
"classifier_reject": "classifier_reject",
}
# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a
# one-line reason (callers must keep the comment). A new schema column that lands
@@ -126,6 +130,9 @@ FULL_ROW = {
"prefix_prev_message_count": 81,
"agent": "classifier",
"parent_key": None,
"classifier_confidence": 0.431,
"classifier_coverage": 0.874,
"classifier_reject": "below_confidence_min",
}