A declined answer left one number behind and it was in a log line, so confidence_min could not be tuned from data. route_decisions.confidence cannot say it either: the chat path re-routes through the override branch, which hard-codes 1.0, and 43,804 of 43,856 live rows hold exactly that. Measured on 2026-10-04 under local_decision (qwen3.5:4b): 769 of 2,557 turns (30%) had no fresh classification, against 0% for local_llm and local_encoder. The cause was recorded in only 4 of them. - classifier_confidence, classifier_coverage and classifier_reject on route_decisions, filled from a ClassifierAttempt carried on Classification. Both accepted and declined answers carry one, so the two distributions can be compared around the floor. Reason codes are listed in docs/data-model.md. - ClassifierRejected (a RuntimeError subclass, messages unchanged) replaces the plain RuntimeErrors at the six floor-miss sites and the two local_decision refusals, so the number and reason travel out of the raise site. - The chat path captures the classifier's verdict before the re-route and passes it to persist_route_decision (attempt_of), like it already does for source. - Admin decisions page: the source badge's tooltip shows the attempt, and session_history / session_stale get an amber badge instead of neutral grey. TUI detail popup and the live event carry the same three keys. - The /metrics degraded-share warning lists the recorded reasons instead of claiming the classifier "has been failing", which was wrong for a classifier that answers and is declined. - degraded_warn_threshold must be in (0, 1] and degraded_warn_min at least 1, refused at load: a value above 1 could never fire. The three columns arrive by ALTER and are NULL on every earlier row; metrics selects them only when present, so the live DB reads NULL until its restart. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01KkCGRantZsSwmcFpet6FTa
223 lines
9.0 KiB
Python
223 lines
9.0 KiB
Python
"""Schema-drift guard: every route_decisions column reaches the TUI on purpose.
|
|
|
|
Five-plus columns shipped to ``route_decisions`` across recent merges without
|
|
any of them reaching the dashboard, and ``tests/test_route_decisions.py``'s own
|
|
hand-maintained column list had ALREADY drifted (missing ``request_id``) — the
|
|
drift recurs, so this file is the enforcement, not the catch-up.
|
|
|
|
Two registries, not a magic diff:
|
|
|
|
- ``SCHEMA_TO_MODEL_KEYS`` — schema column -> the ``tui_model.decision_row``
|
|
key that surfaces it. This is the recorded DECISION about surfacing. Three
|
|
keys are intentionally renamed (``task_category`` -> ``category``,
|
|
``task_tier`` -> ``tier``, ``selected_model`` -> ``selected``), which is
|
|
exactly why the mapping is explicit and never derived by string identity.
|
|
- ``UNSURFACED_COLUMNS`` — columns deliberately NOT surfaced, each with a
|
|
one-line reason. Currently holds one entry (``parent_key``).
|
|
|
|
Every new schema column must land in one of the two registries or
|
|
``test_schema_columns_all_have_surfacing_decisions`` fails naming the column;
|
|
every registry entry must actually round-trip through ``decision_row`` and the
|
|
``/metrics`` SELECT or the other two tests fail naming the key.
|
|
|
|
Offline: temp SQLite created from ``config/schema.sql``, never the live
|
|
router.db (pattern borrowed from tests/test_route_decisions.py).
|
|
"""
|
|
|
|
import sqlite3
|
|
from pathlib import Path
|
|
|
|
import metrics
|
|
import tui_model
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
|
|
|
# route_decisions column -> the tui_model.decision_row key that surfaces it.
|
|
# Schema order (config/schema.sql route_decisions declaration). Add a new
|
|
# schema column here when the TUI should surface it, or to
|
|
# UNSURFACED_COLUMNS when it deliberately should not.
|
|
SCHEMA_TO_MODEL_KEYS = {
|
|
"id": "id",
|
|
"observed_at": "observed_at",
|
|
"kind": "kind",
|
|
"task_category": "category",
|
|
"task_tier": "tier",
|
|
"required_context_tokens": "required_context_tokens",
|
|
"confidence": "confidence",
|
|
"classifier_ms": "classifier_ms",
|
|
"classification_source": "classification_source",
|
|
"latency_tolerance": "latency_tolerance",
|
|
"candidates_considered": "candidates_considered",
|
|
"selected_model": "selected",
|
|
"selected_provider": "selected_provider",
|
|
"runner_up_models": "runner_up_models",
|
|
"est_cost_usd": "est_cost_usd",
|
|
"est_proficiency": "est_proficiency",
|
|
"rejected_reason": "rejected_reason",
|
|
"session_key": "session_key",
|
|
"tools": "tools",
|
|
"images": "images",
|
|
"json_mode": "json_mode",
|
|
"streamed": "streamed",
|
|
"flex_preference": "flex_preference",
|
|
"flex_swapped": "flex_swapped",
|
|
"flex_forced": "flex_forced",
|
|
"request_id": "request_id",
|
|
"exploration": "exploration",
|
|
"pinch_original_tokens": "pinch_original_tokens",
|
|
"pinch_final_tokens": "pinch_final_tokens",
|
|
"profile": "profile",
|
|
# Prefix-stability probe (pinch.prefix_probe). Surfaced through the detail
|
|
# popup, which renders the whole decision_row dict, rather than as new
|
|
# table columns: the three numbers only mean anything read together.
|
|
"prefix_divergence_index": "prefix_divergence_index",
|
|
"prefix_tokens_after_divergence": "prefix_tokens_after_divergence",
|
|
"prefix_prev_message_count": "prefix_prev_message_count",
|
|
"agent": "agent",
|
|
# The classifier's own attempt. Detail popup, like the prefix probe above.
|
|
"classifier_confidence": "classifier_confidence",
|
|
"classifier_coverage": "classifier_coverage",
|
|
"classifier_reject": "classifier_reject",
|
|
}
|
|
# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a
|
|
# one-line reason (callers must keep the comment). A new schema column that lands
|
|
# in NEITHER registry fails the drift test.
|
|
UNSURFACED_COLUMNS: set[str] = {
|
|
"parent_key", # opaque id, recorded for later lineage queries, not displayed
|
|
}
|
|
|
|
# Every column seeded non-NULL: the single source of truth for the INSERT and
|
|
# the round-trip assertions, so the tests prove the column flows, not that
|
|
# None flows.
|
|
FULL_ROW = {
|
|
"id": 1,
|
|
"observed_at": "2026-09-05T12:34:56+00:00",
|
|
"kind": "chat",
|
|
"task_category": "coding_general",
|
|
"task_tier": 2,
|
|
"required_context_tokens": 123456,
|
|
"confidence": 0.91,
|
|
"classifier_ms": 1800,
|
|
"classification_source": "classifier",
|
|
"latency_tolerance": "interactive",
|
|
"candidates_considered": 7,
|
|
"selected_model": "fixture-model",
|
|
"selected_provider": "neuralwatt",
|
|
"runner_up_models": '[{"model_id": "runner", "provider": "neuralwatt"}]',
|
|
"est_cost_usd": 0.00123,
|
|
"est_proficiency": 0.88,
|
|
# Schema-legal even for a selected row (no CHECK constraint); non-NULL so the
|
|
# round-trip assertion proves the column flows, not that None flows.
|
|
"rejected_reason": "fixture rejected reason",
|
|
"session_key": "fixture-session-key",
|
|
"tools": 0,
|
|
"images": 0,
|
|
"json_mode": 1,
|
|
"streamed": 1,
|
|
"flex_preference": "auto",
|
|
"flex_swapped": 0,
|
|
"flex_forced": 0,
|
|
"request_id": "chatcmpl-fixture-42",
|
|
"exploration": 1,
|
|
"pinch_original_tokens": 120000,
|
|
"pinch_final_tokens": 96122,
|
|
"profile": "default",
|
|
# A destructive divergence: index 19 sits below the 81 messages the
|
|
# previous turn had, so the payload was rewritten rather than appended to.
|
|
"prefix_divergence_index": 19,
|
|
"prefix_tokens_after_divergence": 72091,
|
|
"prefix_prev_message_count": 81,
|
|
"agent": "classifier",
|
|
"parent_key": None,
|
|
"classifier_confidence": 0.431,
|
|
"classifier_coverage": 0.874,
|
|
"classifier_reject": "below_confidence_min",
|
|
}
|
|
|
|
|
|
def _db_from_schema(tmp_path, row_factory=None) -> sqlite3.Connection:
|
|
"""A throwaway DB from config/schema.sql — never the live router.db.
|
|
|
|
``row_factory`` defaults to None (plain tuples) like the PRAGMA test
|
|
needs; the metrics round-trip test passes ``sqlite3.Row`` because
|
|
``metrics.recent_decisions`` builds its rows via ``dict(row)``.
|
|
"""
|
|
conn = sqlite3.connect(tmp_path / "schema-drift.db")
|
|
conn.executescript(SCHEMA_SQL)
|
|
conn.row_factory = row_factory
|
|
return conn
|
|
|
|
|
|
def test_schema_columns_all_have_surfacing_decisions(tmp_path):
|
|
"""Every route_decisions column is either surfaced or consciously not."""
|
|
conn = _db_from_schema(tmp_path)
|
|
schema_cols = {
|
|
row[1] for row in conn.execute("PRAGMA table_info(route_decisions)")
|
|
}
|
|
conn.close()
|
|
|
|
registered = set(SCHEMA_TO_MODEL_KEYS) | UNSURFACED_COLUMNS
|
|
missing = schema_cols - set(SCHEMA_TO_MODEL_KEYS) - UNSURFACED_COLUMNS
|
|
extra = registered - schema_cols
|
|
assert not missing, (
|
|
f"route_decisions column(s) added without a surfacing decision: "
|
|
f"{sorted(missing)}. Add each to SCHEMA_TO_MODEL_KEYS if the TUI "
|
|
f"should surface it (project it in tui_model.decision_row), or to "
|
|
f"UNSURFACED_COLUMNS with a reason comment."
|
|
)
|
|
assert not extra, (
|
|
f"registry registers column(s) that no longer exist in "
|
|
f"route_decisions: {sorted(extra)}."
|
|
)
|
|
|
|
|
|
def test_decision_row_projects_every_tracked_column():
|
|
"""decision_row surfaces every registered column's value faithfully."""
|
|
projected = tui_model.decision_row(dict(FULL_ROW))
|
|
|
|
expected_keys = set(SCHEMA_TO_MODEL_KEYS.values())
|
|
assert set(projected) == expected_keys, (
|
|
"tui_model.decision_row no longer matches the surfacing registry "
|
|
f"(symmetric difference: {sorted(expected_keys ^ set(projected))}).\n"
|
|
f" registry values: {sorted(expected_keys)}\n"
|
|
f" decision_row keys: {sorted(projected)}"
|
|
)
|
|
for col, key in SCHEMA_TO_MODEL_KEYS.items():
|
|
assert projected[key] == FULL_ROW[col], (
|
|
f"route_decisions column {col!r} must surface as decision_row "
|
|
f"key {key!r}: got {projected[key]!r}, want {FULL_ROW[col]!r}"
|
|
)
|
|
|
|
|
|
def test_metrics_recent_decisions_selects_every_tracked_column(tmp_path):
|
|
"""The /metrics path carries every registered column end-to-end.
|
|
|
|
``request_id`` reaches the TUI ONLY through this SELECT (SSE publishes
|
|
before the backfill), so a narrowed SELECT silently erases forensics
|
|
fields a few seconds after a live decision arrives.
|
|
"""
|
|
conn = _db_from_schema(tmp_path, row_factory=sqlite3.Row)
|
|
cols = ",".join(FULL_ROW.keys())
|
|
placeholders = ",".join("?" * len(FULL_ROW))
|
|
conn.execute(
|
|
f"INSERT INTO route_decisions ({cols}) VALUES ({placeholders})",
|
|
tuple(FULL_ROW.values()),
|
|
)
|
|
conn.commit()
|
|
|
|
rows = metrics.recent_decisions(conn, limit=5)
|
|
conn.close()
|
|
assert rows, "the seeded FULL_ROW must come back from recent_decisions"
|
|
projected = tui_model.decision_row(rows[0])
|
|
|
|
unreachable = []
|
|
for col, key in SCHEMA_TO_MODEL_KEYS.items():
|
|
if projected.get(key) != FULL_ROW[col]:
|
|
unreachable.append(
|
|
f"route_decisions column {col!r} does not reach the TUI "
|
|
f"through /metrics — is it selected by "
|
|
f"metrics.recent_decisions()?"
|
|
)
|
|
assert not unreachable, "\n".join(unreachable)
|