feat(classifier): record why the classifier declined, and bound the degraded-warn knobs #112
18
CLAUDE.md
18
CLAUDE.md
@@ -870,6 +870,24 @@ over-applied exclusion already starved feedback once (`client_capped`).
|
||||
decisions. A survivable failure is exactly the kind that goes unnoticed for
|
||||
weeks.
|
||||
|
||||
**Why a decline is now recorded, and what the numbers said (2026-10-04).**
|
||||
Under `local_decision` on the 4b the classifier declines about a third of agent
|
||||
turns on purpose: 769 of 2,557 (30%) that day, 8% the day before, against 0% for
|
||||
`local_llm` and `local_encoder`. 22.5% were `session_history` replays and 7.6%
|
||||
the static `general_chat` guess, and `session_history` has no age bound (p99 67
|
||||
minutes, max 100). None of that could be tuned from the database, because
|
||||
`route_decisions.confidence` is the override's hard-coded 1.0 on 99.9% of chat
|
||||
rows. `classifier_confidence`, `classifier_coverage` and `classifier_reject` now
|
||||
hold the classifier's own numbers and the reason (codes in
|
||||
[data-model](docs/data-model.md#what-the-classifier-did-and-why-its-answer-was-not-used)),
|
||||
and the degraded-share warning lists the reasons it saw. The warning's default
|
||||
threshold (0.5) never fires at that rate, so the live overlay sets 0.2.
|
||||
`degraded_warn_min` and `degraded_warn_threshold` have no admin control, and that
|
||||
is a deferred gap, not a decision: the `classifier` section is outside the
|
||||
knob-coverage gate because it has its own card, and adding any `classifier.*`
|
||||
knob to the generic registries would pull every other classifier scalar into the
|
||||
gate at once. Do it in one pass when classification is reworked.
|
||||
|
||||
### Which implementation is PRIMARY is now a config choice
|
||||
|
||||
`classifier.mode` in the live deployment is currently `local_encoder`, set in
|
||||
|
||||
@@ -324,11 +324,29 @@ const SOURCE_STYLES = {
|
||||
classifier: 'background:rgba(59,130,246,.18);color:#93c5fd',
|
||||
cached: 'background:rgba(34,197,94,.18);color:#4ade80',
|
||||
fallback: 'background:rgba(239,68,68,.18);color:#f87171',
|
||||
// Borrowed labels: a real classification of this session, replayed because
|
||||
// the classifier declined this turn. Amber, between a fresh answer and the
|
||||
// static guess, so a run of them is visible instead of reading as neutral.
|
||||
session_history: 'background:rgba(245,158,11,.18);color:#fbbf24',
|
||||
session_stale: 'background:rgba(245,158,11,.18);color:#fbbf24',
|
||||
};
|
||||
function sourceStyle(source) {
|
||||
return SOURCE_STYLES[source] || 'background:rgba(148,163,184,.18);color:#cbd5e1';
|
||||
}
|
||||
|
||||
function attemptTitle(d) {
|
||||
// What the classifier itself did, for the source badge's tooltip. `confidence`
|
||||
// cannot say this on a chat row (the re-route to measured context replaces it
|
||||
// with 1.0), so these come from the classifier_* columns. A row with none of
|
||||
// them (older rows, overrides, cache hits) gets no tooltip rather than a
|
||||
// blank one.
|
||||
const parts = [];
|
||||
if (d.classifier_reject) parts.push(`declined: ${d.classifier_reject}`);
|
||||
if (d.classifier_confidence != null) parts.push(`confidence ${Number(d.classifier_confidence).toFixed(3)}`);
|
||||
if (d.classifier_coverage != null) parts.push(`coverage ${Number(d.classifier_coverage).toFixed(3)}`);
|
||||
return parts.join(', ');
|
||||
}
|
||||
|
||||
function flagChips(d) {
|
||||
// Only the flags that are actually ON — showing all four dimmed on every
|
||||
// row for hundreds of rows was noise, not a cue. Nothing on -> empty cell.
|
||||
@@ -678,7 +696,7 @@ function renderTable() {
|
||||
<td title="${escapeHtml(d.task_category || '')}"><span class="dec-badge" style="${categoryStyle(category)}">${escapeHtml(category)}</span></td>
|
||||
<td>${escapeHtml(d.profile || '')}</td>
|
||||
<td class="tier-${tier}">${escapeHtml(tier)}</td>
|
||||
<td><span class="dec-badge" style="${sourceStyle(source)}">${escapeHtml(source)}</span></td>
|
||||
<td><span class="dec-badge" style="${sourceStyle(source)}" title="${escapeHtml(attemptTitle(d))}">${escapeHtml(source)}</span></td>
|
||||
<td class="model-cell" title="${escapeHtml(model)}">${escapeHtml(model)}</td>
|
||||
<td class="text-muted">${escapeHtml(provider)}</td>
|
||||
<td class="num" title="${d.required_context_tokens != null ? Number(d.required_context_tokens).toLocaleString() + ' tokens' : 'not recorded'}">${ctxCell(d.required_context_tokens)}</td>
|
||||
|
||||
@@ -344,11 +344,28 @@ CREATE TABLE IF NOT EXISTS route_decisions (
|
||||
-- X-Router-Agent (slugged by the
|
||||
-- plugin, e.g. atlas-plan-executor);
|
||||
-- NULL when the client sends none
|
||||
parent_key TEXT -- 'c:' plus the parent conversation
|
||||
parent_key TEXT, -- 'c:' plus the parent conversation
|
||||
-- id from X-Router-Parent, set on a
|
||||
-- sub-agent's rows; NULL for
|
||||
-- top-level conversations or when
|
||||
-- unknown
|
||||
-- What the PRIMARY classifier did with this request. `confidence` above is
|
||||
-- the confidence of the classification that ROUTED it, and on a chat row
|
||||
-- that is almost always 1.0 because the chat path re-routes through the
|
||||
-- override branch, which hard-codes it. These three are the classifier's
|
||||
-- own numbers, read before that re-route. NULL on rows from before they
|
||||
-- existed, on override rows and on session-cache hits (no attempt made).
|
||||
classifier_confidence REAL, -- the classifier's confidence in
|
||||
-- its answer, accepted or not
|
||||
classifier_coverage REAL, -- local_decision only: total
|
||||
-- option mass behind that answer
|
||||
classifier_reject TEXT -- NULL when the answer was used,
|
||||
-- else why not: below_confidence_min,
|
||||
-- below_coverage_min, no_logprobs,
|
||||
-- timeout, transport_error,
|
||||
-- parse_error, primary_failed,
|
||||
-- skipped_<why>. classification_source
|
||||
-- then says what routed instead.
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_verifications_model ON verifications (model_id, provider);
|
||||
|
||||
@@ -258,9 +258,9 @@ failure — it means the checker had nothing to say, not that the model failed.
|
||||
| `task_category` | TEXT | |
|
||||
| `task_tier` | INTEGER | 1–3 |
|
||||
| `required_context_tokens` | INTEGER | |
|
||||
| `confidence` | REAL | Classifier confidence |
|
||||
| `confidence` | REAL | Confidence of the classification that ROUTED the request. On a `chat` row this is almost always `1.0`, because the chat path re-routes through the override branch, which hard-codes it. Use `classifier_confidence` for what the classifier said |
|
||||
| `classifier_ms` | INTEGER | Classification latency |
|
||||
| `classification_source` | TEXT | `classifier` \| `override` \| `fallback` \| `cached` |
|
||||
| `classification_source` | TEXT | `classifier` \| `override` \| `fallback` \| `cached` \| `session_stale` \| `session_history` \| `classifier_cloud` |
|
||||
| `latency_tolerance` | TEXT | `interactive` \| `batch` |
|
||||
| `candidates_considered` | INTEGER | How many survived hard filters |
|
||||
| `selected_model` | TEXT | Null when no model was selected |
|
||||
@@ -283,6 +283,43 @@ failure — it means the checker had nothing to say, not that the model failed.
|
||||
| `prefix_divergence_index` | INTEGER | First message position whose bytes differ from the previous turn in this session |
|
||||
| `prefix_tokens_after_divergence` | INTEGER | Estimated tokens at or after that position in THIS turn |
|
||||
| `prefix_prev_message_count` | INTEGER | The previous turn's message count |
|
||||
| `classifier_confidence` | REAL | The primary classifier's confidence in its answer, accepted or not. NULL when it produced none (a timeout) or made no attempt (`override`, `cached`) |
|
||||
| `classifier_coverage` | REAL | `local_decision` only: the total option mass behind that answer |
|
||||
| `classifier_reject` | TEXT | NULL when the answer was used; otherwise why it was not (codes below) |
|
||||
|
||||
### What the classifier did, and why its answer was not used
|
||||
|
||||
The three `classifier_*` columns record the **primary classifier's own
|
||||
attempt**, read before the chat path re-routes on measured context. They exist
|
||||
because a declined answer used to leave one number behind and it was in a log
|
||||
line: `classification_source = session_history` says a borrowed label routed the
|
||||
turn, not whether the classifier was a hair under its floor or nowhere near it,
|
||||
and `confidence` could not say so (see its row above).
|
||||
|
||||
`classifier_reject` is NULL when the answer was used, and otherwise one of:
|
||||
|
||||
| code | meaning | numbers recorded |
|
||||
|---|---|---|
|
||||
| `below_confidence_min` | answered, but under `classifier.*.confidence_min` | confidence (and coverage for `local_decision`) |
|
||||
| `below_coverage_min` | `local_decision`: total option mass under `coverage_min` | coverage, and the would-be confidence |
|
||||
| `no_logprobs` | `local_decision`: Ollama returned none | none |
|
||||
| `timeout` | the endpoint did not answer in time | none |
|
||||
| `transport_error` | the endpoint was unreachable or errored | none |
|
||||
| `parse_error` | the reply could not be parsed | none |
|
||||
| `primary_failed` | any other failure of the primary | none |
|
||||
| `skipped_gaming_mode`, `skipped_backoff` | the local classifier was never asked | none |
|
||||
|
||||
`classification_source` then says what routed the turn instead: `session_history`
|
||||
(this session's last real label), `fallback` (the static guess) or
|
||||
`classifier_cloud`. Tune `confidence_min` from the distribution: for
|
||||
`local_decision` rows, compare `classifier_confidence` where `classifier_reject`
|
||||
is NULL against where it is `below_confidence_min`.
|
||||
|
||||
Rows from before these columns existed hold NULL in all three, which is the
|
||||
honest value; the numbers were never stored, so there is nothing to backfill.
|
||||
The admin decisions page shows them as the tooltip on the source badge, the TUI
|
||||
detail popup shows all three, and the `/metrics` degraded-share warning lists the
|
||||
reasons it saw.
|
||||
|
||||
The three `prefix_*` columns are the prefix-stability probe
|
||||
(`pinch.prefix_probe`, on by default). The provider bills the longest
|
||||
|
||||
@@ -358,6 +358,11 @@ Mechanism, a Jev-style first-token-logprob classifier:
|
||||
- As with `local_encoder`, it produces only `task_category` unless
|
||||
`classifier.decision.tier_enabled` is set, in which case it may also choose
|
||||
`task_tier`; otherwise tier falls back to `classifier.fallback_tier`.
|
||||
- A declined answer is recorded, not just logged: `route_decisions.classifier_confidence`,
|
||||
`classifier_coverage` and `classifier_reject` hold the numbers and the reason
|
||||
(see [data-model](data-model.md#what-the-classifier-did-and-why-its-answer-was-not-used)).
|
||||
On the 4b the decline is by design, not an outage: it ran at 30% of turns on
|
||||
2026-10-04 (8% the day before), concentrated in long agent sessions.
|
||||
|
||||
Pull and enable it:
|
||||
|
||||
|
||||
59
src/classifier_rejection.py
Normal file
59
src/classifier_rejection.py
Normal file
@@ -0,0 +1,59 @@
|
||||
"""Why the primary classifier's answer was not used, as data.
|
||||
|
||||
A classifier that declines to answer is the router's most consequential quiet
|
||||
failure: the request is then routed on a borrowed or guessed label. Before this
|
||||
module the reason existed only as a log line, and the number that decided it
|
||||
(how far below ``confidence_min`` the answer was) existed nowhere durable, so a
|
||||
threshold could not be tuned from the database. ``ClassifierRejected`` carries
|
||||
both out of the raise site, and the ``REASON_*`` strings are the stable codes
|
||||
stored in ``route_decisions.classifier_reject``.
|
||||
|
||||
Stdlib only, and it imports nothing from this package, so ``local_decision`` can
|
||||
raise it without importing ``dispatcher`` (which imports ``local_decision``).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
# The model answered, and the dispatcher declined the answer.
|
||||
REASON_BELOW_CONFIDENCE: Final = "below_confidence_min"
|
||||
REASON_BELOW_COVERAGE: Final = "below_coverage_min"
|
||||
REASON_NO_LOGPROBS: Final = "no_logprobs"
|
||||
|
||||
# The model did not answer usefully.
|
||||
REASON_TIMEOUT: Final = "timeout"
|
||||
REASON_TRANSPORT: Final = "transport_error"
|
||||
REASON_PARSE: Final = "parse_error"
|
||||
REASON_PRIMARY_FAILED: Final = "primary_failed"
|
||||
|
||||
# The model was never asked. Followed by the skip reason, e.g.
|
||||
# ``skipped_gaming_mode`` or ``skipped_backoff``.
|
||||
REASON_SKIPPED_PREFIX: Final = "skipped_"
|
||||
|
||||
|
||||
class ClassifierRejected(RuntimeError):
|
||||
"""The classifier answered, but the answer was refused.
|
||||
|
||||
A ``RuntimeError`` subclass on purpose: ``classify()`` already treats a
|
||||
``RuntimeError`` from the primary as "the primary failed" and walks the
|
||||
cascade, so existing handlers keep working and only the data is new. The
|
||||
message text is unchanged from the plain ``RuntimeError`` it replaces,
|
||||
because it is what the ``fallback`` log line prints.
|
||||
|
||||
``confidence`` and ``coverage`` are whatever the raise site knew, and None
|
||||
where it did not: a coverage rejection happens before a confidence exists.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
reason: str,
|
||||
message: str,
|
||||
*,
|
||||
confidence: float | None = None,
|
||||
coverage: float | None = None,
|
||||
) -> None:
|
||||
super().__init__(message)
|
||||
self.reason = reason
|
||||
self.confidence = confidence
|
||||
self.coverage = coverage
|
||||
@@ -1155,6 +1155,16 @@ class LocalDecisionConfig(StrictModel):
|
||||
return v
|
||||
|
||||
|
||||
# Bounds for the degraded-share warning. Named, not inlined in the validators,
|
||||
# because a runtime write goes straight past Pydantic and the admin registry's
|
||||
# bounds are then the only check: it imports these so the runtime path and the
|
||||
# load validator cannot disagree about what a legal value is. The threshold is
|
||||
# exclusive of 0 (a share of 0 is always reached) and inclusive of 1.
|
||||
DEGRADED_WARN_MIN_FLOOR = 1
|
||||
DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE = 0.0
|
||||
DEGRADED_WARN_THRESHOLD_MAX = 1.0
|
||||
|
||||
|
||||
class ClassifierConfig(StrictModel):
|
||||
provider: str
|
||||
base_url: str
|
||||
@@ -1236,6 +1246,28 @@ class ClassifierConfig(StrictModel):
|
||||
# Only read when mode == "local_decision".
|
||||
decision: Optional["LocalDecisionConfig"] = None
|
||||
|
||||
# Refused at load rather than accepted and inert. A threshold above 1 can
|
||||
# never be reached by a share, and one at or below 0 is always reached, so
|
||||
# either reads as a working detector that has quietly stopped warning (or
|
||||
# never stops). The minimum has to be a real sample size for the same
|
||||
# reason: 0 would compute a share over an empty window.
|
||||
@field_validator("degraded_warn_threshold")
|
||||
@classmethod
|
||||
def degraded_warn_threshold_in_range(cls, v: float) -> float:
|
||||
if not (DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE < v <= DEGRADED_WARN_THRESHOLD_MAX):
|
||||
raise ValueError(
|
||||
"classifier.degraded_warn_threshold is a share of decisions "
|
||||
"and must be in (0, 1]; a value above 1 can never fire"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("degraded_warn_min")
|
||||
@classmethod
|
||||
def degraded_warn_min_positive(cls, v: int) -> int:
|
||||
if v < DEGRADED_WARN_MIN_FLOOR:
|
||||
raise ValueError("classifier.degraded_warn_min must be at least 1")
|
||||
return v
|
||||
|
||||
|
||||
class LocalComputeConfig(StrictModel):
|
||||
"""The outer gate over every call this router makes to local hardware.
|
||||
|
||||
@@ -54,7 +54,7 @@ import requests
|
||||
from dotenv import load_dotenv
|
||||
from fastapi import BackgroundTasks, FastAPI, HTTPException, Request
|
||||
from fastapi.responses import JSONResponse, Response, StreamingResponse
|
||||
from openai import APIStatusError, OpenAI, OpenAIError
|
||||
from openai import APIStatusError, APITimeoutError, OpenAI, OpenAIError
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
|
||||
import admin
|
||||
@@ -63,6 +63,15 @@ import events
|
||||
import exploration
|
||||
import local_decision
|
||||
import local_encoder
|
||||
from classifier_rejection import (
|
||||
REASON_BELOW_CONFIDENCE,
|
||||
REASON_PARSE,
|
||||
REASON_PRIMARY_FAILED,
|
||||
REASON_SKIPPED_PREFIX,
|
||||
REASON_TIMEOUT,
|
||||
REASON_TRANSPORT,
|
||||
ClassifierRejected,
|
||||
)
|
||||
import local_energy
|
||||
import logs
|
||||
import prefix_probe
|
||||
@@ -249,6 +258,22 @@ class TaskRequest(BaseModel):
|
||||
required_context_tokens: Optional[int] = Field(None, ge=0)
|
||||
|
||||
|
||||
class ClassifierAttempt(BaseModel):
|
||||
"""What the primary classifier did with this request, kept as data.
|
||||
|
||||
Attached to every classification that went through ``classify()``, accepted
|
||||
or not, so ``route_decisions`` can answer "how close to the floor was that
|
||||
abstention" instead of only "it abstained". ``reject_reason`` is None when
|
||||
the answer was accepted and one of ``classifier_rejection``'s codes when it
|
||||
was not; ``confidence`` and ``coverage`` are whatever the classifier
|
||||
produced, None where it produced no such number (a timeout has neither).
|
||||
"""
|
||||
|
||||
confidence: Optional[float] = None
|
||||
coverage: Optional[float] = None
|
||||
reject_reason: Optional[str] = None
|
||||
|
||||
|
||||
class Classification(BaseModel):
|
||||
task_category: str
|
||||
task_tier: int
|
||||
@@ -267,6 +292,10 @@ class Classification(BaseModel):
|
||||
"session_history",
|
||||
"classifier_cloud",
|
||||
] = "classifier"
|
||||
# The primary classifier's own attempt, which a degraded source does not
|
||||
# otherwise remember: a session_history row has confidence 0.0 by
|
||||
# construction, so without this the rejected 0.43 would be lost.
|
||||
attempt: Optional[ClassifierAttempt] = None
|
||||
|
||||
|
||||
class Candidate(BaseModel):
|
||||
@@ -456,7 +485,10 @@ def ensure_route_decisions(conn: sqlite3.Connection) -> None:
|
||||
prefix_tokens_after_divergence INTEGER,
|
||||
prefix_prev_message_count INTEGER,
|
||||
agent TEXT,
|
||||
parent_key TEXT
|
||||
parent_key TEXT,
|
||||
classifier_confidence REAL,
|
||||
classifier_coverage REAL,
|
||||
classifier_reject TEXT
|
||||
)
|
||||
"""
|
||||
)
|
||||
@@ -499,6 +531,14 @@ def ensure_route_decisions(conn: sqlite3.Connection) -> None:
|
||||
# that decided it or link to a parent decision.
|
||||
("agent", "TEXT"),
|
||||
("parent_key", "TEXT"),
|
||||
# What the primary classifier did with the request: its confidence,
|
||||
# its coverage (local_decision only) and, when its answer was not
|
||||
# used, why. Additive and NULL on every existing row, which is the
|
||||
# honest value: the number that decided an abstention was only ever
|
||||
# in a log line, so no earlier row can be backfilled.
|
||||
("classifier_confidence", "REAL"),
|
||||
("classifier_coverage", "REAL"),
|
||||
("classifier_reject", "TEXT"),
|
||||
):
|
||||
if name not in existing:
|
||||
conn.execute(f"ALTER TABLE route_decisions ADD COLUMN {name} {decl}")
|
||||
@@ -884,6 +924,49 @@ def _provider_cost_multipliers() -> Optional[dict[str, float]]:
|
||||
return multipliers
|
||||
|
||||
|
||||
def _below_confidence(
|
||||
mode: str,
|
||||
section: str,
|
||||
confidence: float,
|
||||
floor: float,
|
||||
coverage: Optional[float] = None,
|
||||
) -> ClassifierRejected:
|
||||
"""The one refusal for a floor miss, so message and reason cannot drift.
|
||||
|
||||
The text is what the ``fallback`` log line has always printed; the reason
|
||||
and the number travel with it as data.
|
||||
"""
|
||||
return ClassifierRejected(
|
||||
REASON_BELOW_CONFIDENCE,
|
||||
f"{mode} confidence {confidence:.3f} is below "
|
||||
f"classifier.{section}.confidence_min ({floor})",
|
||||
confidence=confidence,
|
||||
coverage=coverage,
|
||||
)
|
||||
|
||||
|
||||
def _attempt_from_exception(exc: BaseException) -> ClassifierAttempt:
|
||||
"""Turn a primary-classifier failure into the record stored on the row.
|
||||
|
||||
Order matters: ``APITimeoutError`` is an ``OpenAIError``, and
|
||||
``requests.Timeout`` is a ``RequestException``, so each specific case sits
|
||||
ahead of its parent.
|
||||
"""
|
||||
if isinstance(exc, ClassifierRejected):
|
||||
return ClassifierAttempt(
|
||||
confidence=exc.confidence,
|
||||
coverage=exc.coverage,
|
||||
reject_reason=exc.reason,
|
||||
)
|
||||
if isinstance(exc, (requests.Timeout, APITimeoutError)):
|
||||
return ClassifierAttempt(reject_reason=REASON_TIMEOUT)
|
||||
if isinstance(exc, (requests.RequestException, OpenAIError)):
|
||||
return ClassifierAttempt(reject_reason=REASON_TRANSPORT)
|
||||
if isinstance(exc, (ValueError, KeyError, TypeError, json.JSONDecodeError)):
|
||||
return ClassifierAttempt(reject_reason=REASON_PARSE)
|
||||
return ClassifierAttempt(reject_reason=REASON_PRIMARY_FAILED)
|
||||
|
||||
|
||||
class _ClassifierSkipped(Exception):
|
||||
"""The local classifier was not attempted at all -- gaming mode, or an
|
||||
already-open backoff circuit (see _local_classifier_skip_reason).
|
||||
@@ -1040,9 +1123,8 @@ def _classify_via_local_encoder(task: str) -> Classification:
|
||||
device=enc.device,
|
||||
)
|
||||
if confidence < enc.confidence_min:
|
||||
raise RuntimeError(
|
||||
f"local_encoder confidence {confidence:.3f} is below "
|
||||
f"classifier.encoder.confidence_min ({enc.confidence_min})"
|
||||
raise _below_confidence(
|
||||
"local_encoder", "encoder", confidence, enc.confidence_min
|
||||
)
|
||||
result = Classification(
|
||||
task_category=category,
|
||||
@@ -1071,9 +1153,8 @@ def _classify_via_local_encoder(task: str) -> Classification:
|
||||
device=enc.device,
|
||||
)
|
||||
if confidence < enc.confidence_min:
|
||||
raise RuntimeError(
|
||||
f"local_encoder confidence {confidence:.3f} is below "
|
||||
f"classifier.encoder.confidence_min ({enc.confidence_min})"
|
||||
raise _below_confidence(
|
||||
"local_encoder", "encoder", confidence, enc.confidence_min
|
||||
)
|
||||
return Classification(
|
||||
task_category=category,
|
||||
@@ -1135,19 +1216,21 @@ def _classify_via_local_decision(
|
||||
tier_future = pool.submit(
|
||||
_classify_tier_via_local_decision, user_content
|
||||
)
|
||||
category, confidence, _coverage = cat_future.result()
|
||||
category, confidence, coverage = cat_future.result()
|
||||
if confidence < dec.confidence_min:
|
||||
raise RuntimeError(
|
||||
f"local_decision confidence {confidence:.3f} "
|
||||
f"is below classifier.decision.confidence_min "
|
||||
f"({dec.confidence_min})"
|
||||
raise _below_confidence(
|
||||
"local_decision",
|
||||
"decision",
|
||||
confidence,
|
||||
dec.confidence_min,
|
||||
coverage,
|
||||
)
|
||||
try:
|
||||
tier = tier_future.result()
|
||||
except Exception:
|
||||
tier = cfg.classifier.fallback_tier
|
||||
else:
|
||||
category, confidence, _coverage = (
|
||||
category, confidence, coverage = (
|
||||
local_decision.classify_category(
|
||||
user_content,
|
||||
base_url=dec.base_url,
|
||||
@@ -1158,10 +1241,12 @@ def _classify_via_local_decision(
|
||||
)
|
||||
)
|
||||
if confidence < dec.confidence_min:
|
||||
raise RuntimeError(
|
||||
f"local_decision confidence {confidence:.3f} "
|
||||
f"is below classifier.decision.confidence_min "
|
||||
f"({dec.confidence_min})"
|
||||
raise _below_confidence(
|
||||
"local_decision",
|
||||
"decision",
|
||||
confidence,
|
||||
dec.confidence_min,
|
||||
coverage,
|
||||
)
|
||||
|
||||
_log_local_energy(
|
||||
@@ -1175,6 +1260,9 @@ def _classify_via_local_decision(
|
||||
required_context_tokens=0,
|
||||
confidence=confidence,
|
||||
source="classifier",
|
||||
attempt=ClassifierAttempt(
|
||||
confidence=confidence, coverage=coverage
|
||||
),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — meter the real draw on ANY failure
|
||||
if isinstance(exc, requests.RequestException):
|
||||
@@ -1208,18 +1296,21 @@ def _classify_via_local_decision(
|
||||
tier_future = pool.submit(
|
||||
_classify_tier_via_local_decision, user_content
|
||||
)
|
||||
category, confidence, _coverage = cat_future.result()
|
||||
category, confidence, coverage = cat_future.result()
|
||||
if confidence < dec.confidence_min:
|
||||
raise RuntimeError(
|
||||
f"local_decision confidence {confidence:.3f} is below "
|
||||
f"classifier.decision.confidence_min ({dec.confidence_min})"
|
||||
raise _below_confidence(
|
||||
"local_decision",
|
||||
"decision",
|
||||
confidence,
|
||||
dec.confidence_min,
|
||||
coverage,
|
||||
)
|
||||
try:
|
||||
tier = tier_future.result()
|
||||
except Exception:
|
||||
tier = cfg.classifier.fallback_tier
|
||||
else:
|
||||
category, confidence, _coverage = local_decision.classify_category(
|
||||
category, confidence, coverage = local_decision.classify_category(
|
||||
user_content,
|
||||
base_url=dec.base_url,
|
||||
model=dec.model,
|
||||
@@ -1228,9 +1319,12 @@ def _classify_via_local_decision(
|
||||
coverage_min=dec.coverage_min,
|
||||
)
|
||||
if confidence < dec.confidence_min:
|
||||
raise RuntimeError(
|
||||
f"local_decision confidence {confidence:.3f} is below "
|
||||
f"classifier.decision.confidence_min ({dec.confidence_min})"
|
||||
raise _below_confidence(
|
||||
"local_decision",
|
||||
"decision",
|
||||
confidence,
|
||||
dec.confidence_min,
|
||||
coverage,
|
||||
)
|
||||
tier = _classify_tier_via_local_decision(user_content)
|
||||
return Classification(
|
||||
@@ -1239,6 +1333,9 @@ def _classify_via_local_decision(
|
||||
required_context_tokens=0,
|
||||
confidence=confidence,
|
||||
source="classifier",
|
||||
attempt=ClassifierAttempt(
|
||||
confidence=confidence, coverage=coverage
|
||||
),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
if isinstance(exc, requests.RequestException):
|
||||
@@ -1512,22 +1609,30 @@ def _degraded_classification(
|
||||
session_key: Optional[str],
|
||||
system_prompt: str,
|
||||
user_content: str,
|
||||
attempt: Optional[ClassifierAttempt] = None,
|
||||
) -> Classification:
|
||||
"""The cascade, then the static guess if every step of it misses.
|
||||
|
||||
One place, because the three callers used to carry byte-identical copies
|
||||
of the fallback construction and a fourth was about to.
|
||||
|
||||
``attempt`` is what the primary classifier did before the cascade took
|
||||
over. It is stamped on whatever the cascade returns, because every
|
||||
degraded source is a borrowed or guessed label that has forgotten why the
|
||||
primary was not used.
|
||||
"""
|
||||
degraded = _classify_cascade(session_key, system_prompt, user_content)
|
||||
if degraded is not None:
|
||||
return degraded
|
||||
return Classification(
|
||||
task_category=cfg.classifier.fallback_category,
|
||||
task_tier=cfg.classifier.fallback_tier,
|
||||
required_context_tokens=0,
|
||||
confidence=0.0,
|
||||
source="fallback",
|
||||
)
|
||||
result = _classify_cascade(session_key, system_prompt, user_content)
|
||||
if result is None:
|
||||
result = Classification(
|
||||
task_category=cfg.classifier.fallback_category,
|
||||
task_tier=cfg.classifier.fallback_tier,
|
||||
required_context_tokens=0,
|
||||
confidence=0.0,
|
||||
source="fallback",
|
||||
)
|
||||
if attempt is not None:
|
||||
result = result.model_copy(update={"attempt": attempt})
|
||||
return result
|
||||
|
||||
|
||||
def _provider_client(provider: str) -> OpenAI:
|
||||
@@ -1613,14 +1718,19 @@ def classify(task: str, context: Optional[str]) -> Classification:
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
resp = _classify_via_configured_mode(task, system_prompt, user_content)
|
||||
except _ClassifierSkipped:
|
||||
except _ClassifierSkipped as skipped:
|
||||
# Already logged specifically (classify_local_skipped, with WHY --
|
||||
# gaming_mode or backoff). No generic "fallback" line here: that one
|
||||
# means an attempt was made and failed, and a deliberate skip is
|
||||
# neither -- conflating the two would tell an operator reading raw
|
||||
# logs that the local classifier is failing when it was never asked.
|
||||
return _degraded_classification(
|
||||
_current_session_key.get(), system_prompt, user_content
|
||||
_current_session_key.get(),
|
||||
system_prompt,
|
||||
user_content,
|
||||
attempt=ClassifierAttempt(
|
||||
reject_reason=f"{REASON_SKIPPED_PREFIX}{skipped}"
|
||||
),
|
||||
)
|
||||
except (
|
||||
OpenAIError,
|
||||
@@ -1649,7 +1759,10 @@ def classify(task: str, context: Optional[str]) -> Classification:
|
||||
ms=_ms(started),
|
||||
)
|
||||
return _degraded_classification(
|
||||
_current_session_key.get(), system_prompt, user_content
|
||||
_current_session_key.get(),
|
||||
system_prompt,
|
||||
user_content,
|
||||
attempt=_attempt_from_exception(e),
|
||||
)
|
||||
|
||||
# The single blocking LLM call on the request path, so its latency is the
|
||||
@@ -1671,6 +1784,12 @@ def classify(task: str, context: Optional[str]) -> Classification:
|
||||
# modes, which is correct -- there is no local-endpoint circuit for them
|
||||
# to close.
|
||||
_record_success_cooldown()
|
||||
if resp.attempt is None:
|
||||
# local_decision stamps its own, because it also knows the coverage;
|
||||
# every other mode has only a confidence to record.
|
||||
resp = resp.model_copy(
|
||||
update={"attempt": ClassifierAttempt(confidence=resp.confidence)}
|
||||
)
|
||||
return resp
|
||||
|
||||
|
||||
@@ -2277,6 +2396,7 @@ def persist_route_decision(
|
||||
prefix_divergence: Optional[prefix_probe.Divergence] = None,
|
||||
agent: Optional[str] = None,
|
||||
parent_key: Optional[str] = None,
|
||||
attempt_of: Optional[Classification] = None,
|
||||
) -> Optional[int]:
|
||||
"""Record one routing decision to route_decisions, best-effort and gated.
|
||||
|
||||
@@ -2298,6 +2418,11 @@ def persist_route_decision(
|
||||
the RouteResponse would otherwise say — the chat re-route's
|
||||
Classification reads ``source='override'`` even though the classifier did
|
||||
decide the request, so the caller passes the first route's actual source.
|
||||
|
||||
``attempt_of`` is the same idea for the classifier's own attempt (its
|
||||
confidence, coverage and rejection reason): the re-routed override carries
|
||||
none of it, so the chat path passes the Classification as the first route
|
||||
produced it. Absent, the attempt is read from ``classification``.
|
||||
"""
|
||||
# The gate is read before anything can fail: turning the table off must be
|
||||
# a guaranteed no-op even on a broken DB.
|
||||
@@ -2337,6 +2462,8 @@ def persist_route_decision(
|
||||
|
||||
if clf is not None and classification_source is None:
|
||||
classification_source = clf.source
|
||||
attempt_src = attempt_of if attempt_of is not None else clf
|
||||
attempt = attempt_src.attempt if attempt_src is not None else None
|
||||
# An override-created Classification never consulted the classifier, so it
|
||||
# has no classifier latency worth recording — even when a caller passed a
|
||||
# timing value (route_endpoint's whole-route ms would otherwise leak in).
|
||||
@@ -2386,8 +2513,9 @@ def persist_route_decision(
|
||||
flex_preference, flex_swapped, flex_forced, exploration,
|
||||
request_id, pinch_original_tokens, pinch_final_tokens, profile,
|
||||
prefix_divergence_index, prefix_tokens_after_divergence,
|
||||
prefix_prev_message_count, agent, parent_key
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
prefix_prev_message_count, agent, parent_key,
|
||||
classifier_confidence, classifier_coverage, classifier_reject
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
observed_at,
|
||||
@@ -2424,6 +2552,9 @@ def persist_route_decision(
|
||||
prefix_prev_count,
|
||||
agent,
|
||||
parent_key,
|
||||
attempt.confidence if attempt is not None else None,
|
||||
attempt.coverage if attempt is not None else None,
|
||||
attempt.reject_reason if attempt is not None else None,
|
||||
),
|
||||
)
|
||||
decision_id: Optional[int] = int(cursor.lastrowid)
|
||||
@@ -2469,6 +2600,15 @@ def persist_route_decision(
|
||||
"prefix_prev_message_count": prefix_prev_count,
|
||||
"agent": agent,
|
||||
"parent_key": parent_key,
|
||||
"classifier_confidence": (
|
||||
attempt.confidence if attempt is not None else None
|
||||
),
|
||||
"classifier_coverage": (
|
||||
attempt.coverage if attempt is not None else None
|
||||
),
|
||||
"classifier_reject": (
|
||||
attempt.reject_reason if attempt is not None else None
|
||||
),
|
||||
}
|
||||
)
|
||||
return decision_id
|
||||
@@ -4616,6 +4756,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks, request:
|
||||
# the category/tier are reused. The required context is this
|
||||
# request's own measurement.
|
||||
classified_src = "cached"
|
||||
classifier_attempt = None
|
||||
classifier_ms = 0
|
||||
ctx_src = "measured"
|
||||
decision = route(
|
||||
@@ -4660,6 +4801,9 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks, request:
|
||||
# The classifier's own verdict, before the re-route below rewrites
|
||||
# the Classification's source to 'override'.
|
||||
classified_src = decision.classification.source
|
||||
# Same reason as the source above: the re-route below replaces the
|
||||
# Classification with an override that remembers none of this.
|
||||
classifier_attempt = decision.classification
|
||||
classifier_ms = _ms(classify_started)
|
||||
ctx_src = "classifier"
|
||||
if measured > decision.classification.required_context_tokens:
|
||||
@@ -4790,6 +4934,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks, request:
|
||||
streamed=streamed,
|
||||
classification_source=classified_src,
|
||||
classifier_ms=classifier_ms,
|
||||
attempt_of=classifier_attempt,
|
||||
pinch_original_tokens=pinch_stats.get("original_tokens")
|
||||
if pinch_stats is not None else None,
|
||||
pinch_final_tokens=pinch_stats.get("final_tokens")
|
||||
|
||||
@@ -39,6 +39,12 @@ from typing import Any, Final, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from classifier_rejection import (
|
||||
REASON_BELOW_COVERAGE,
|
||||
REASON_NO_LOGPROBS,
|
||||
ClassifierRejected,
|
||||
)
|
||||
|
||||
_DEFAULT_COVERAGE_MIN: float = 0.3
|
||||
|
||||
# Decision-category descriptions tuned against the measured confusion
|
||||
@@ -116,14 +122,19 @@ def parse_logprobs(
|
||||
|
||||
Raises
|
||||
------
|
||||
RuntimeError
|
||||
When ``coverage < coverage_min`` or when no logprobs are found.
|
||||
ClassifierRejected
|
||||
(a :exc:`RuntimeError`) When ``coverage < coverage_min`` or when no
|
||||
logprobs are found. ``reason`` says which, and a coverage rejection
|
||||
also carries the coverage and the would-be confidence, because how
|
||||
far below the floor an answer fell is what tuning ``coverage_min``
|
||||
needs.
|
||||
"""
|
||||
logprobs_list: list[dict[str, Any]] = response_json.get("logprobs", [])
|
||||
|
||||
if not logprobs_list:
|
||||
raise RuntimeError(
|
||||
"parse_logprobs: no logprobs found in classifier response"
|
||||
raise ClassifierRejected(
|
||||
REASON_NO_LOGPROBS,
|
||||
"parse_logprobs: no logprobs found in classifier response",
|
||||
)
|
||||
|
||||
# Accumulate exp(logprob) mass per option letter.
|
||||
@@ -147,9 +158,13 @@ def parse_logprobs(
|
||||
total = sum(mass.values())
|
||||
|
||||
if total < coverage_min:
|
||||
raise RuntimeError(
|
||||
best = max(mass.values(), default=0.0)
|
||||
raise ClassifierRejected(
|
||||
REASON_BELOW_COVERAGE,
|
||||
f"parse_logprobs: coverage {total:.4f} below "
|
||||
f"minimum {coverage_min}"
|
||||
f"minimum {coverage_min}",
|
||||
confidence=(best / total) if total > 0 else None,
|
||||
coverage=total,
|
||||
)
|
||||
|
||||
# Winner is the option with the most accumulated mass.
|
||||
|
||||
@@ -1773,12 +1773,44 @@ def classifier_degradation_warning(conn: sqlite3.Connection, cfg: Any) -> list[s
|
||||
return []
|
||||
return [
|
||||
f"{share:.0%} of the last {total} classifications came from a degraded "
|
||||
f"source ({degraded} of {total}) — the local classifier has been "
|
||||
f"failing. Routing still works on borrowed categories, but their "
|
||||
f"outcomes are excluded from proficiency."
|
||||
f"source ({degraded} of {total}). {_declined_for(conn)}Routing still "
|
||||
f"works on borrowed categories, but their outcomes are excluded from "
|
||||
f"proficiency."
|
||||
]
|
||||
|
||||
|
||||
def _declined_for(conn: sqlite3.Connection) -> str:
|
||||
"""The recorded reasons the primary classifier's answer was not used.
|
||||
|
||||
The warning used to end "the local classifier has been failing", which is
|
||||
wrong for the case that now dominates: under ``local_decision`` the
|
||||
classifier answers and is declined for being unsure, and nothing is down.
|
||||
Saying which reason is what tells the operator whether to look at the
|
||||
endpoint (``transport_error``, ``timeout``) or at a threshold
|
||||
(``below_confidence_min``). Empty when no reasons were recorded, which is
|
||||
every row from before the column existed, and when the column is absent on
|
||||
a database that has not been through dispatcher start-up.
|
||||
"""
|
||||
if not _has_column(conn, "route_decisions", "classifier_reject"):
|
||||
return ""
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT classifier_reject AS reason, COUNT(*) AS n
|
||||
FROM route_decisions
|
||||
WHERE classifier_reject IS NOT NULL
|
||||
AND classification_source IN
|
||||
('fallback', 'session_stale', 'session_history',
|
||||
'classifier_cloud')
|
||||
AND observed_at >= datetime('now', '-24 hours')
|
||||
GROUP BY classifier_reject
|
||||
ORDER BY n DESC, classifier_reject
|
||||
"""
|
||||
).fetchall()
|
||||
if not rows:
|
||||
return ""
|
||||
return "Declined for: " + ", ".join(f"{r['n']} {r['reason']}" for r in rows) + ". "
|
||||
|
||||
|
||||
# Gate prefixes whose failure means a DERIVED field is broken, rather than the
|
||||
# operator having switched something off.
|
||||
#
|
||||
@@ -2935,6 +2967,16 @@ _CONVERSATION_IDENTITY_COLUMNS: Final = (
|
||||
"parent_key",
|
||||
)
|
||||
|
||||
# The classifier's own attempt (confidence, coverage, why it was rejected).
|
||||
# Added by ALTER at dispatcher start-up like the groups above, so probed and
|
||||
# backfilled the same way: the live router.db has none of them until its next
|
||||
# restart, and a missing column must read as NULL rather than 500 the payload.
|
||||
_CLASSIFIER_ATTEMPT_COLUMNS: Final = (
|
||||
"classifier_confidence",
|
||||
"classifier_coverage",
|
||||
"classifier_reject",
|
||||
)
|
||||
|
||||
|
||||
def recent_decisions(
|
||||
conn: sqlite3.Connection,
|
||||
@@ -2961,8 +3003,14 @@ def recent_decisions(
|
||||
for column in _CONVERSATION_IDENTITY_COLUMNS
|
||||
if _has_column(conn, "route_decisions", column)
|
||||
]
|
||||
attempt_present = [
|
||||
column
|
||||
for column in _CLASSIFIER_ATTEMPT_COLUMNS
|
||||
if _has_column(conn, "route_decisions", column)
|
||||
]
|
||||
probe_select = "".join(f", {column}" for column in present)
|
||||
identity_select = "".join(f", {column}" for column in identity_present)
|
||||
attempt_select = "".join(f", {column}" for column in attempt_present)
|
||||
rows = [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
@@ -2975,7 +3023,7 @@ def recent_decisions(
|
||||
rejected_reason, session_key, tools, images, json_mode, streamed,
|
||||
flex_preference, flex_swapped, flex_forced,
|
||||
exploration, request_id,
|
||||
pinch_original_tokens, pinch_final_tokens, profile{probe_select}{identity_select}
|
||||
pinch_original_tokens, pinch_final_tokens, profile{probe_select}{identity_select}{attempt_select}
|
||||
FROM route_decisions
|
||||
ORDER BY id DESC
|
||||
LIMIT ?
|
||||
@@ -2991,6 +3039,8 @@ def recent_decisions(
|
||||
row.setdefault(column, None)
|
||||
for column in _CONVERSATION_IDENTITY_COLUMNS:
|
||||
row.setdefault(column, None)
|
||||
for column in _CLASSIFIER_ATTEMPT_COLUMNS:
|
||||
row.setdefault(column, None)
|
||||
return rows
|
||||
|
||||
|
||||
|
||||
@@ -249,6 +249,14 @@ def decision_row(r: dict) -> dict:
|
||||
"request_id": r.get("request_id"),
|
||||
"session_key": r.get("session_key"),
|
||||
"agent": r.get("agent"),
|
||||
# The classifier's own attempt: how confident it was, how much option
|
||||
# mass sat behind that, and why its answer was not used. Detail popup
|
||||
# only, like the prefix probe: the three read together ("0.43, below
|
||||
# the 0.5 floor, so session_history routed it") and `confidence` above
|
||||
# cannot say this on a chat row, where it is the override's 1.0.
|
||||
"classifier_confidence": r.get("classifier_confidence"),
|
||||
"classifier_coverage": r.get("classifier_coverage"),
|
||||
"classifier_reject": r.get("classifier_reject"),
|
||||
}
|
||||
|
||||
|
||||
|
||||
3
tests/fixtures/route_decision_no_header.json
vendored
3
tests/fixtures/route_decision_no_header.json
vendored
@@ -2,6 +2,9 @@
|
||||
"agent": null,
|
||||
"candidates_considered": 2,
|
||||
"classification_source": "classifier",
|
||||
"classifier_confidence": null,
|
||||
"classifier_coverage": null,
|
||||
"classifier_reject": null,
|
||||
"confidence": 0.9,
|
||||
"est_cost_usd": 0.00016,
|
||||
"est_proficiency": 0.5,
|
||||
|
||||
532
tests/test_classifier_attempt.py
Normal file
532
tests/test_classifier_attempt.py
Normal file
@@ -0,0 +1,532 @@
|
||||
"""The classifier's own attempt reaches route_decisions.
|
||||
|
||||
Before this, an abstention left one number behind and it was in a log line:
|
||||
``local_decision confidence 0.431 is below classifier.decision.confidence_min
|
||||
(0.5)``. The database said only ``classification_source = session_history``, and
|
||||
``route_decisions.confidence`` read 1.0 on 99.9% of chat rows because the chat
|
||||
path re-routes through the override branch, which hard-codes it. So
|
||||
``confidence_min`` could not be tuned from data: nobody could say whether the
|
||||
abstentions sat just under the floor or nowhere near it.
|
||||
|
||||
These tests pin the three columns end to end: ``classifier_confidence``,
|
||||
``classifier_coverage`` and ``classifier_reject``. The chat-path test runs the
|
||||
REAL ``classify()`` (only the Ollama call is stubbed), because the stub that
|
||||
``test_no_header_snapshot`` installs replaces ``classify`` itself and would hide
|
||||
exactly the stamping this file is about.
|
||||
"""
|
||||
|
||||
import json
|
||||
import sqlite3
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import openai
|
||||
import pytest
|
||||
import requests
|
||||
import test_no_header_snapshot as snap
|
||||
from test_route_decisions import _schema_minus_profile_column
|
||||
|
||||
import dispatcher
|
||||
import local_decision
|
||||
import prefix_probe
|
||||
import session_cache
|
||||
from classifier_rejection import (
|
||||
REASON_BELOW_CONFIDENCE,
|
||||
REASON_BELOW_COVERAGE,
|
||||
REASON_NO_LOGPROBS,
|
||||
REASON_PARSE,
|
||||
REASON_PRIMARY_FAILED,
|
||||
REASON_TIMEOUT,
|
||||
REASON_TRANSPORT,
|
||||
ClassifierRejected,
|
||||
)
|
||||
from dispatcher import Classification, ClassifierAttempt
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
||||
ATTEMPT_COLUMNS = ("classifier_confidence", "classifier_coverage", "classifier_reject")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clean_state(monkeypatch):
|
||||
session_cache.clear()
|
||||
prefix_probe._store.clear()
|
||||
monkeypatch.setattr(dispatcher, "_last_classifier_failure", 0.0)
|
||||
dispatcher._provider_refusal_since.clear()
|
||||
token = dispatcher._current_session_key.set(None)
|
||||
yield
|
||||
dispatcher._current_session_key.reset(token)
|
||||
dispatcher._provider_refusal_since.clear()
|
||||
session_cache.clear()
|
||||
# The chat-path test posts the snapshot's own payload, and the probe keeps
|
||||
# per-session fingerprints in process memory: left behind, the headerless
|
||||
# snapshot test (which runs after this file) sees a "previous turn" and
|
||||
# records a non-NULL prefix divergence.
|
||||
prefix_probe._store.clear()
|
||||
|
||||
|
||||
def _use_local_decision(monkeypatch, *, confidence_min=0.5):
|
||||
"""local_decision as the primary, unmetered, with nothing skipped."""
|
||||
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
||||
monkeypatch.setattr(
|
||||
dispatcher.cfg.classifier,
|
||||
"decision",
|
||||
SimpleNamespace(
|
||||
base_url="http://localhost:11434",
|
||||
model="stub-model",
|
||||
num_ctx=8192,
|
||||
timeout_s=10,
|
||||
confidence_min=confidence_min,
|
||||
coverage_min=0.3,
|
||||
tier_enabled=False,
|
||||
),
|
||||
)
|
||||
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
||||
monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", False)
|
||||
monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True)
|
||||
|
||||
|
||||
def _answers(monkeypatch, result):
|
||||
"""Stub the Ollama call: return ``result`` or raise it if it is an exception."""
|
||||
|
||||
def fake(*args, **kwargs):
|
||||
if isinstance(result, BaseException):
|
||||
raise result
|
||||
return result
|
||||
|
||||
monkeypatch.setattr(local_decision, "classify_category", fake)
|
||||
|
||||
|
||||
# --- the raise sites carry the numbers -------------------------------------
|
||||
|
||||
|
||||
def test_a_floor_miss_records_the_confidence_that_missed_it(monkeypatch):
|
||||
_use_local_decision(monkeypatch)
|
||||
_answers(monkeypatch, ("debugging", 0.431, 0.874))
|
||||
|
||||
got = dispatcher.classify("fix the failing test", None)
|
||||
|
||||
assert got.source == "fallback"
|
||||
assert got.attempt == ClassifierAttempt(
|
||||
confidence=0.431, coverage=0.874, reject_reason=REASON_BELOW_CONFIDENCE
|
||||
)
|
||||
|
||||
|
||||
def test_a_coverage_miss_records_the_coverage_and_the_would_be_confidence(monkeypatch):
|
||||
"""Both gates are recorded, because tuning one without the other misleads."""
|
||||
_use_local_decision(monkeypatch)
|
||||
# Total option mass 0.0498 sits under coverage_min (0.3); letter A holds all of it.
|
||||
low_mass = {
|
||||
"logprobs": [
|
||||
{"token": "A", "logprob": -3.0, "top_logprobs": [{"token": "A", "logprob": -3.0}]}
|
||||
]
|
||||
}
|
||||
monkeypatch.setattr(
|
||||
local_decision,
|
||||
"classify_category",
|
||||
lambda *a, **k: local_decision.parse_logprobs(low_mass, ["A", "B"], coverage_min=0.3),
|
||||
)
|
||||
|
||||
got = dispatcher.classify("fix the failing test", None)
|
||||
|
||||
assert got.attempt.reject_reason == REASON_BELOW_COVERAGE
|
||||
assert got.attempt.coverage == pytest.approx(0.0498, abs=1e-4)
|
||||
assert got.attempt.confidence == pytest.approx(1.0)
|
||||
|
||||
|
||||
def test_parse_logprobs_raises_a_runtime_error_with_its_original_message():
|
||||
"""Subclassing RuntimeError keeps every existing handler and log line intact."""
|
||||
with pytest.raises(ClassifierRejected, match="no logprobs found") as no_lp:
|
||||
local_decision.parse_logprobs({}, ["A", "B"])
|
||||
assert isinstance(no_lp.value, RuntimeError)
|
||||
assert no_lp.value.reason == REASON_NO_LOGPROBS
|
||||
|
||||
thin = {"logprobs": [{"token": "A", "top_logprobs": [{"token": "A", "logprob": -4.0}]}]}
|
||||
with pytest.raises(RuntimeError, match=r"coverage 0\.0183 below minimum 0\.3") as thin_err:
|
||||
local_decision.parse_logprobs(thin, ["A", "B"], coverage_min=0.3)
|
||||
assert thin_err.value.reason == REASON_BELOW_COVERAGE
|
||||
|
||||
|
||||
def test_the_rejection_message_is_unchanged(monkeypatch):
|
||||
"""The `fallback` journal line prints this text; operators grep for it."""
|
||||
_use_local_decision(monkeypatch)
|
||||
_answers(monkeypatch, ("debugging", 0.431, 0.874))
|
||||
with pytest.raises(ClassifierRejected) as err:
|
||||
dispatcher._classify_via_local_decision("sys", "do a thing")
|
||||
assert str(err.value) == (
|
||||
"local_decision confidence 0.431 is below classifier.decision.confidence_min (0.5)"
|
||||
)
|
||||
|
||||
|
||||
# --- every failure class gets a code ---------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"failure, reason",
|
||||
[
|
||||
(requests.ConnectionError("refused"), REASON_TRANSPORT),
|
||||
(requests.Timeout("slow"), REASON_TIMEOUT),
|
||||
# The SDK's own timeout type, which is an OpenAIError subclass and so
|
||||
# would be filed as a transport error if the order of the checks slipped.
|
||||
(
|
||||
openai.APITimeoutError(request=SimpleNamespace(method="POST", url="http://x")),
|
||||
REASON_TIMEOUT,
|
||||
),
|
||||
(ValueError("not json"), REASON_PARSE),
|
||||
(KeyError("choices"), REASON_PARSE),
|
||||
(json.JSONDecodeError("bad", "{", 0), REASON_PARSE),
|
||||
(RuntimeError("something else"), REASON_PRIMARY_FAILED),
|
||||
],
|
||||
ids=lambda p: type(p).__name__ if isinstance(p, BaseException) else p,
|
||||
)
|
||||
def test_each_failure_class_has_a_stable_code_and_no_invented_numbers(
|
||||
monkeypatch, failure, reason
|
||||
):
|
||||
_use_local_decision(monkeypatch)
|
||||
_answers(monkeypatch, failure)
|
||||
|
||||
got = dispatcher.classify("fix the failing test", None)
|
||||
|
||||
assert got.source == "fallback"
|
||||
assert got.attempt == ClassifierAttempt(reject_reason=reason)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("why", ["gaming_mode", "backoff"])
|
||||
def test_a_skip_is_recorded_as_a_skip_not_a_failure(monkeypatch, why):
|
||||
_use_local_decision(monkeypatch)
|
||||
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: why)
|
||||
|
||||
got = dispatcher.classify("fix the failing test", None)
|
||||
|
||||
assert got.attempt == ClassifierAttempt(reject_reason=f"skipped_{why}")
|
||||
|
||||
|
||||
def test_an_accepted_answer_carries_its_numbers_and_no_reason(monkeypatch):
|
||||
_use_local_decision(monkeypatch)
|
||||
_answers(monkeypatch, ("debugging", 0.83, 0.91))
|
||||
|
||||
got = dispatcher.classify("fix the failing test", None)
|
||||
|
||||
assert got.source == "classifier"
|
||||
assert got.attempt == ClassifierAttempt(confidence=0.83, coverage=0.91, reject_reason=None)
|
||||
|
||||
|
||||
def test_modes_without_a_coverage_still_record_their_confidence(monkeypatch):
|
||||
"""The generic stamp: local_llm, cloud_llm and the encoder have no coverage."""
|
||||
monkeypatch.setattr(
|
||||
dispatcher,
|
||||
"_classify_via_configured_mode",
|
||||
lambda *a, **k: Classification(
|
||||
task_category="coding_general",
|
||||
task_tier=2,
|
||||
required_context_tokens=100,
|
||||
confidence=0.9,
|
||||
),
|
||||
)
|
||||
|
||||
got = dispatcher.classify("fix the failing test", None)
|
||||
|
||||
assert got.attempt == ClassifierAttempt(confidence=0.9)
|
||||
|
||||
|
||||
# --- the degraded builder keeps what the cascade forgot ---------------------
|
||||
|
||||
|
||||
def test_the_cascade_result_is_stamped_without_changing_what_it_routes_on(monkeypatch):
|
||||
borrowed = Classification(
|
||||
task_category="debugging",
|
||||
task_tier=3,
|
||||
required_context_tokens=0,
|
||||
confidence=0.0,
|
||||
source="session_history",
|
||||
)
|
||||
monkeypatch.setattr(dispatcher, "_classify_cascade", lambda *a: borrowed)
|
||||
attempt = ClassifierAttempt(confidence=0.431, reject_reason=REASON_BELOW_CONFIDENCE)
|
||||
|
||||
got = dispatcher._degraded_classification("k", "sys", "user", attempt=attempt)
|
||||
|
||||
assert got.attempt == attempt
|
||||
assert (got.source, got.task_category, got.task_tier, got.confidence) == (
|
||||
"session_history",
|
||||
"debugging",
|
||||
3,
|
||||
0.0,
|
||||
)
|
||||
assert borrowed.attempt is None, "the cascade's own object must not be mutated"
|
||||
|
||||
|
||||
def test_the_static_guess_is_stamped_too(monkeypatch):
|
||||
monkeypatch.setattr(dispatcher, "_classify_cascade", lambda *a: None)
|
||||
attempt = ClassifierAttempt(reject_reason=REASON_TIMEOUT)
|
||||
|
||||
got = dispatcher._degraded_classification("k", "sys", "user", attempt=attempt)
|
||||
|
||||
assert got.source == "fallback"
|
||||
assert got.attempt == attempt
|
||||
assert dispatcher._degraded_classification("k", "sys", "user").attempt is None
|
||||
|
||||
|
||||
# --- the row ----------------------------------------------------------------
|
||||
|
||||
|
||||
def _fresh_db(tmp_path, monkeypatch):
|
||||
db_path = tmp_path / "attempt.db"
|
||||
conn = sqlite3.connect(db_path)
|
||||
conn.executescript(SCHEMA_SQL)
|
||||
conn.close()
|
||||
monkeypatch.setattr(dispatcher.cfg.database, "path", str(db_path))
|
||||
monkeypatch.setattr(dispatcher.cfg.logging, "log_route_decisions", True)
|
||||
return db_path
|
||||
|
||||
|
||||
def _rows(db_path):
|
||||
conn = sqlite3.connect(db_path)
|
||||
conn.row_factory = sqlite3.Row
|
||||
try:
|
||||
return [dict(r) for r in conn.execute("SELECT * FROM route_decisions ORDER BY id")]
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def _degraded(attempt):
|
||||
return Classification(
|
||||
task_category="debugging",
|
||||
task_tier=2,
|
||||
required_context_tokens=0,
|
||||
confidence=0.0,
|
||||
source="session_history",
|
||||
attempt=attempt,
|
||||
)
|
||||
|
||||
|
||||
def test_persist_writes_the_attempt_from_the_classification(tmp_path, monkeypatch):
|
||||
db_path = _fresh_db(tmp_path, monkeypatch)
|
||||
attempt = ClassifierAttempt(
|
||||
confidence=0.431, coverage=0.874, reject_reason=REASON_BELOW_CONFIDENCE
|
||||
)
|
||||
|
||||
dispatcher.persist_route_decision(
|
||||
"route", classification=_degraded(attempt), selected_model="m", selected_provider="p"
|
||||
)
|
||||
|
||||
(row,) = _rows(db_path)
|
||||
assert row["classifier_confidence"] == pytest.approx(0.431)
|
||||
assert row["classifier_coverage"] == pytest.approx(0.874)
|
||||
assert row["classifier_reject"] == REASON_BELOW_CONFIDENCE
|
||||
assert row["confidence"] == 0.0, "the routing classification's own confidence is untouched"
|
||||
|
||||
|
||||
def test_persist_prefers_attempt_of_when_the_classification_was_rerouted(tmp_path, monkeypatch):
|
||||
"""The chat path rewrites `classification` to an override that remembers nothing."""
|
||||
db_path = _fresh_db(tmp_path, monkeypatch)
|
||||
override = Classification(
|
||||
task_category="debugging",
|
||||
task_tier=2,
|
||||
required_context_tokens=90000,
|
||||
confidence=1.0,
|
||||
source="override",
|
||||
)
|
||||
first = _degraded(
|
||||
ClassifierAttempt(confidence=0.431, reject_reason=REASON_BELOW_CONFIDENCE)
|
||||
)
|
||||
|
||||
dispatcher.persist_route_decision(
|
||||
"chat",
|
||||
classification=override,
|
||||
classification_source="session_history",
|
||||
attempt_of=first,
|
||||
selected_model="m",
|
||||
selected_provider="p",
|
||||
)
|
||||
|
||||
(row,) = _rows(db_path)
|
||||
assert row["confidence"] == 1.0, "legacy column: the override's hard-coded value"
|
||||
assert row["classifier_confidence"] == pytest.approx(0.431)
|
||||
assert row["classifier_reject"] == REASON_BELOW_CONFIDENCE
|
||||
|
||||
|
||||
def test_persist_without_an_attempt_leaves_all_three_null(tmp_path, monkeypatch):
|
||||
db_path = _fresh_db(tmp_path, monkeypatch)
|
||||
override = Classification(
|
||||
task_category="debugging",
|
||||
task_tier=2,
|
||||
required_context_tokens=100,
|
||||
confidence=1.0,
|
||||
source="override",
|
||||
)
|
||||
|
||||
dispatcher.persist_route_decision(
|
||||
"route", classification=override, selected_model="m", selected_provider="p"
|
||||
)
|
||||
|
||||
(row,) = _rows(db_path)
|
||||
assert [row[c] for c in ATTEMPT_COLUMNS] == [None, None, None]
|
||||
|
||||
|
||||
def test_the_live_event_carries_the_same_three_keys(tmp_path, monkeypatch):
|
||||
_fresh_db(tmp_path, monkeypatch)
|
||||
published = []
|
||||
monkeypatch.setattr(dispatcher.events, "publish_decision", published.append)
|
||||
attempt = ClassifierAttempt(confidence=0.431, reject_reason=REASON_BELOW_CONFIDENCE)
|
||||
|
||||
dispatcher.persist_route_decision(
|
||||
"route", classification=_degraded(attempt), selected_model="m", selected_provider="p"
|
||||
)
|
||||
|
||||
(event,) = published
|
||||
assert event["classifier_confidence"] == pytest.approx(0.431)
|
||||
assert event["classifier_coverage"] is None
|
||||
assert event["classifier_reject"] == REASON_BELOW_CONFIDENCE
|
||||
|
||||
|
||||
# --- migration ---------------------------------------------------------------
|
||||
|
||||
|
||||
def test_a_live_table_gains_the_columns_once_and_keeps_its_rows(tmp_path):
|
||||
conn = sqlite3.connect(tmp_path / "live.db")
|
||||
conn.executescript(_schema_minus_profile_column())
|
||||
conn.execute(
|
||||
"INSERT INTO route_decisions (observed_at, kind) VALUES ('2026-10-01T00:00:00+00:00', 'chat')"
|
||||
)
|
||||
conn.commit()
|
||||
before = {r[1] for r in conn.execute("PRAGMA table_info(route_decisions)")}
|
||||
assert not set(ATTEMPT_COLUMNS) & before
|
||||
|
||||
dispatcher.ensure_route_decisions(conn)
|
||||
dispatcher.ensure_route_decisions(conn) # idempotent
|
||||
|
||||
cols = [r[1] for r in conn.execute("PRAGMA table_info(route_decisions)")]
|
||||
for column in ATTEMPT_COLUMNS:
|
||||
assert cols.count(column) == 1
|
||||
old = conn.execute(f"SELECT {', '.join(ATTEMPT_COLUMNS)} FROM route_decisions").fetchone()
|
||||
assert tuple(old) == (None, None, None), "an old row cannot be backfilled, and says so"
|
||||
conn.close()
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
sqlite3.sqlite_version_info < (3, 35),
|
||||
reason="ALTER TABLE ... DROP COLUMN needs SQLite 3.35",
|
||||
)
|
||||
def test_recent_decisions_reads_a_database_that_has_not_been_migrated(tmp_path):
|
||||
"""/metrics and the admin page share metrics.py; the live DB lacks the columns
|
||||
until the router restarts, and that must read as NULL rather than a 500.
|
||||
|
||||
The shape under test is a CURRENT schema minus only these three columns,
|
||||
which is what the live router.db is between the deploy and the restart.
|
||||
"""
|
||||
import metrics
|
||||
|
||||
conn = sqlite3.connect(tmp_path / "old.db")
|
||||
conn.row_factory = sqlite3.Row
|
||||
conn.executescript(SCHEMA_SQL)
|
||||
for column in ATTEMPT_COLUMNS:
|
||||
conn.execute(f"ALTER TABLE route_decisions DROP COLUMN {column}")
|
||||
conn.execute(
|
||||
"INSERT INTO route_decisions (observed_at, kind) VALUES ('2026-10-01T00:00:00+00:00', 'chat')"
|
||||
)
|
||||
conn.commit()
|
||||
assert not set(ATTEMPT_COLUMNS) & {r[1] for r in conn.execute("PRAGMA table_info(route_decisions)")}
|
||||
|
||||
(row,) = metrics.recent_decisions(conn, limit=5)
|
||||
conn.close()
|
||||
|
||||
for column in ATTEMPT_COLUMNS:
|
||||
assert column in row and row[column] is None
|
||||
|
||||
|
||||
# --- the chat path, with the real classify() ---------------------------------
|
||||
|
||||
|
||||
def test_chat_rows_record_the_attempt_even_though_the_reroute_hides_it(tmp_path, monkeypatch):
|
||||
"""Turn 1 is accepted; turn 2 of the same session is declined and replays turn 1's label.
|
||||
|
||||
Both rows must say what the classifier did. The re-route to measured
|
||||
context replaces the Classification with an override in between, which is
|
||||
why ``confidence`` could never have shown this.
|
||||
"""
|
||||
real_classify = dispatcher.classify
|
||||
client, db_path = snap.make_router(tmp_path, monkeypatch)
|
||||
monkeypatch.setattr(dispatcher, "classify", real_classify)
|
||||
_use_local_decision(monkeypatch)
|
||||
answers = iter([("coding_general", 0.91, 0.95), ("coding_general", 0.431, 0.874)])
|
||||
monkeypatch.setattr(local_decision, "classify_category", lambda *a, **k: next(answers))
|
||||
|
||||
for _ in range(2):
|
||||
resp = snap.post_no_header(client)
|
||||
assert resp.status_code == 200, resp.text
|
||||
|
||||
first, second = _rows(db_path)
|
||||
assert first["classification_source"] == "classifier"
|
||||
assert first["classifier_confidence"] == pytest.approx(0.91)
|
||||
assert first["classifier_coverage"] == pytest.approx(0.95)
|
||||
assert first["classifier_reject"] is None
|
||||
|
||||
assert second["classification_source"] == "session_history"
|
||||
assert second["classifier_confidence"] == pytest.approx(0.431)
|
||||
assert second["classifier_coverage"] == pytest.approx(0.874)
|
||||
assert second["classifier_reject"] == REASON_BELOW_CONFIDENCE
|
||||
|
||||
|
||||
# --- the warning knobs refuse values that could never behave -------------------
|
||||
|
||||
|
||||
def _classifier_with(**overrides):
|
||||
"""The real loaded classifier block with only the knob under test changed.
|
||||
|
||||
ClassifierConfig has required fields, so building one bare would fail on
|
||||
those and bury what the test is about.
|
||||
"""
|
||||
from config import ClassifierConfig
|
||||
|
||||
return ClassifierConfig(**{**dispatcher.cfg.classifier.model_dump(), **overrides})
|
||||
|
||||
|
||||
@pytest.mark.parametrize("threshold", [0.0, -0.1, 1.5, 5, 50])
|
||||
def test_degraded_warn_threshold_outside_a_share_is_refused_at_load(threshold):
|
||||
"""`80` meant as 80% once shipped as a raw number into another knob and made
|
||||
every score read as below threshold; a share above 1 here is the same
|
||||
mistake and would make the warning inert instead."""
|
||||
from pydantic import ValidationError
|
||||
|
||||
with pytest.raises(ValidationError, match="degraded_warn_threshold"):
|
||||
_classifier_with(degraded_warn_threshold=threshold)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("threshold", [0.01, 0.2, 0.5, 1.0])
|
||||
def test_degraded_warn_threshold_accepts_any_real_share(threshold):
|
||||
got = _classifier_with(degraded_warn_threshold=threshold)
|
||||
assert got.degraded_warn_threshold == threshold
|
||||
|
||||
|
||||
@pytest.mark.parametrize("minimum", [0, -3])
|
||||
def test_degraded_warn_min_must_be_a_real_sample_size(minimum):
|
||||
from pydantic import ValidationError
|
||||
|
||||
with pytest.raises(ValidationError, match="degraded_warn_min"):
|
||||
_classifier_with(degraded_warn_min=minimum)
|
||||
|
||||
|
||||
def test_degraded_warn_bound_constants_are_the_validators_boundaries():
|
||||
"""A runtime admin write skips Pydantic, so its registry imports these named
|
||||
bounds. They are only trustworthy if they ARE the load validator's edges:
|
||||
the exclusive low and the floor are refused, the high and floor+0 accepted."""
|
||||
from pydantic import ValidationError
|
||||
|
||||
from config import (
|
||||
DEGRADED_WARN_MIN_FLOOR,
|
||||
DEGRADED_WARN_THRESHOLD_MAX,
|
||||
DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE,
|
||||
)
|
||||
|
||||
with pytest.raises(ValidationError):
|
||||
_classifier_with(degraded_warn_threshold=DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE)
|
||||
assert (
|
||||
_classifier_with(degraded_warn_threshold=DEGRADED_WARN_THRESHOLD_MAX).degraded_warn_threshold
|
||||
== DEGRADED_WARN_THRESHOLD_MAX
|
||||
)
|
||||
with pytest.raises(ValidationError):
|
||||
_classifier_with(degraded_warn_min=DEGRADED_WARN_MIN_FLOOR - 1)
|
||||
assert (
|
||||
_classifier_with(degraded_warn_min=DEGRADED_WARN_MIN_FLOOR).degraded_warn_min
|
||||
== DEGRADED_WARN_MIN_FLOOR
|
||||
)
|
||||
@@ -270,3 +270,67 @@ def test_degradation_warning_silent_when_healthy(db):
|
||||
)
|
||||
_seed_sources(db, [("classifier", 24), ("fallback", 1)])
|
||||
assert classifier_degradation_warning(db, cfg) == []
|
||||
|
||||
|
||||
def _seed_declined(conn, source, reason, count):
|
||||
for _ in range(count):
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO route_decisions
|
||||
(kind, task_category, task_tier, required_context_tokens,
|
||||
selected_model, selected_provider, classification_source,
|
||||
classifier_reject, observed_at)
|
||||
VALUES ('chat','general_chat',2,0,'m','neuralwatt',?,?,?)
|
||||
""",
|
||||
(source, reason, _now().isoformat()),
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def test_degradation_warning_says_why_the_classifier_was_not_used(db):
|
||||
""""The local classifier has been failing" was wrong for local_decision, where
|
||||
nothing fails and the answer is declined for being unsure."""
|
||||
from metrics import classifier_degradation_warning
|
||||
|
||||
cfg = SimpleNamespace(
|
||||
classifier=SimpleNamespace(degraded_warn_min=20, degraded_warn_threshold=0.2)
|
||||
)
|
||||
_seed_sources(db, [("classifier", 60)])
|
||||
_seed_declined(db, "session_history", "below_confidence_min", 18)
|
||||
_seed_declined(db, "fallback", "below_coverage_min", 5)
|
||||
_seed_declined(db, "session_history", "below_confidence_min", 2)
|
||||
|
||||
(warning,) = classifier_degradation_warning(db, cfg)
|
||||
|
||||
assert "degraded source" in warning # the TUI's classifier-degraded matcher
|
||||
assert "Declined for: 20 below_confidence_min, 5 below_coverage_min." in warning
|
||||
assert "failing" not in warning
|
||||
|
||||
|
||||
def test_degradation_warning_without_recorded_reasons_makes_no_claim_about_failure(db):
|
||||
from metrics import classifier_degradation_warning
|
||||
|
||||
cfg = SimpleNamespace(
|
||||
classifier=SimpleNamespace(degraded_warn_min=20, degraded_warn_threshold=0.5)
|
||||
)
|
||||
_seed_sources(db, [("fallback", 15), ("classifier", 10)])
|
||||
|
||||
(warning,) = classifier_degradation_warning(db, cfg)
|
||||
|
||||
assert "Declined for" not in warning
|
||||
assert "failing" not in warning
|
||||
assert "degraded source (15 of 25)" in warning
|
||||
|
||||
|
||||
def test_degradation_warning_survives_a_database_without_the_column(db, monkeypatch):
|
||||
import metrics
|
||||
|
||||
cfg = SimpleNamespace(
|
||||
classifier=SimpleNamespace(degraded_warn_min=20, degraded_warn_threshold=0.5)
|
||||
)
|
||||
_seed_sources(db, [("fallback", 15), ("classifier", 10)])
|
||||
monkeypatch.setattr(metrics, "_has_column", lambda *a, **k: False)
|
||||
|
||||
(warning,) = metrics.classifier_degradation_warning(db, cfg)
|
||||
|
||||
assert "60%" in warning and "Declined for" not in warning
|
||||
|
||||
@@ -72,6 +72,9 @@ ROUTE_DECISIONS_COLUMNS = [
|
||||
"prefix_prev_message_count",
|
||||
"agent",
|
||||
"parent_key",
|
||||
"classifier_confidence",
|
||||
"classifier_coverage",
|
||||
"classifier_reject",
|
||||
]
|
||||
|
||||
|
||||
|
||||
@@ -75,6 +75,10 @@ SCHEMA_TO_MODEL_KEYS = {
|
||||
"prefix_tokens_after_divergence": "prefix_tokens_after_divergence",
|
||||
"prefix_prev_message_count": "prefix_prev_message_count",
|
||||
"agent": "agent",
|
||||
# The classifier's own attempt. Detail popup, like the prefix probe above.
|
||||
"classifier_confidence": "classifier_confidence",
|
||||
"classifier_coverage": "classifier_coverage",
|
||||
"classifier_reject": "classifier_reject",
|
||||
}
|
||||
# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a
|
||||
# one-line reason (callers must keep the comment). A new schema column that lands
|
||||
@@ -126,6 +130,9 @@ FULL_ROW = {
|
||||
"prefix_prev_message_count": 81,
|
||||
"agent": "classifier",
|
||||
"parent_key": None,
|
||||
"classifier_confidence": 0.431,
|
||||
"classifier_coverage": 0.874,
|
||||
"classifier_reject": "below_confidence_min",
|
||||
}
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user