Files
6krrt/tests/test_tui_schema_drift.py
adlee-was-taken 92031dbd0b feat(classifier): record the classifier's own attempt on route_decisions
A declined answer left one number behind and it was in a log line, so
confidence_min could not be tuned from data. route_decisions.confidence cannot
say it either: the chat path re-routes through the override branch, which
hard-codes 1.0, and 43,804 of 43,856 live rows hold exactly that.

Measured on 2026-10-04 under local_decision (qwen3.5:4b): 769 of 2,557 turns
(30%) had no fresh classification, against 0% for local_llm and local_encoder.
The cause was recorded in only 4 of them.

- classifier_confidence, classifier_coverage and classifier_reject on
  route_decisions, filled from a ClassifierAttempt carried on Classification.
  Both accepted and declined answers carry one, so the two distributions can be
  compared around the floor. Reason codes are listed in docs/data-model.md.
- ClassifierRejected (a RuntimeError subclass, messages unchanged) replaces the
  plain RuntimeErrors at the six floor-miss sites and the two local_decision
  refusals, so the number and reason travel out of the raise site.
- The chat path captures the classifier's verdict before the re-route and passes
  it to persist_route_decision (attempt_of), like it already does for source.
- Admin decisions page: the source badge's tooltip shows the attempt, and
  session_history / session_stale get an amber badge instead of neutral grey.
  TUI detail popup and the live event carry the same three keys.
- The /metrics degraded-share warning lists the recorded reasons instead of
  claiming the classifier "has been failing", which was wrong for a classifier
  that answers and is declined.
- degraded_warn_threshold must be in (0, 1] and degraded_warn_min at least 1,
  refused at load: a value above 1 could never fire.

The three columns arrive by ALTER and are NULL on every earlier row; metrics
selects them only when present, so the live DB reads NULL until its restart.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01KkCGRantZsSwmcFpet6FTa
2026-10-04 22:06:57 -04:00

223 lines
9.0 KiB
Python

"""Schema-drift guard: every route_decisions column reaches the TUI on purpose.
Five-plus columns shipped to ``route_decisions`` across recent merges without
any of them reaching the dashboard, and ``tests/test_route_decisions.py``'s own
hand-maintained column list had ALREADY drifted (missing ``request_id``) — the
drift recurs, so this file is the enforcement, not the catch-up.
Two registries, not a magic diff:
- ``SCHEMA_TO_MODEL_KEYS`` — schema column -> the ``tui_model.decision_row``
key that surfaces it. This is the recorded DECISION about surfacing. Three
keys are intentionally renamed (``task_category`` -> ``category``,
``task_tier`` -> ``tier``, ``selected_model`` -> ``selected``), which is
exactly why the mapping is explicit and never derived by string identity.
- ``UNSURFACED_COLUMNS`` — columns deliberately NOT surfaced, each with a
one-line reason. Currently holds one entry (``parent_key``).
Every new schema column must land in one of the two registries or
``test_schema_columns_all_have_surfacing_decisions`` fails naming the column;
every registry entry must actually round-trip through ``decision_row`` and the
``/metrics`` SELECT or the other two tests fail naming the key.
Offline: temp SQLite created from ``config/schema.sql``, never the live
router.db (pattern borrowed from tests/test_route_decisions.py).
"""
import sqlite3
from pathlib import Path
import metrics
import tui_model
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
# route_decisions column -> the tui_model.decision_row key that surfaces it.
# Schema order (config/schema.sql route_decisions declaration). Add a new
# schema column here when the TUI should surface it, or to
# UNSURFACED_COLUMNS when it deliberately should not.
SCHEMA_TO_MODEL_KEYS = {
"id": "id",
"observed_at": "observed_at",
"kind": "kind",
"task_category": "category",
"task_tier": "tier",
"required_context_tokens": "required_context_tokens",
"confidence": "confidence",
"classifier_ms": "classifier_ms",
"classification_source": "classification_source",
"latency_tolerance": "latency_tolerance",
"candidates_considered": "candidates_considered",
"selected_model": "selected",
"selected_provider": "selected_provider",
"runner_up_models": "runner_up_models",
"est_cost_usd": "est_cost_usd",
"est_proficiency": "est_proficiency",
"rejected_reason": "rejected_reason",
"session_key": "session_key",
"tools": "tools",
"images": "images",
"json_mode": "json_mode",
"streamed": "streamed",
"flex_preference": "flex_preference",
"flex_swapped": "flex_swapped",
"flex_forced": "flex_forced",
"request_id": "request_id",
"exploration": "exploration",
"pinch_original_tokens": "pinch_original_tokens",
"pinch_final_tokens": "pinch_final_tokens",
"profile": "profile",
# Prefix-stability probe (pinch.prefix_probe). Surfaced through the detail
# popup, which renders the whole decision_row dict, rather than as new
# table columns: the three numbers only mean anything read together.
"prefix_divergence_index": "prefix_divergence_index",
"prefix_tokens_after_divergence": "prefix_tokens_after_divergence",
"prefix_prev_message_count": "prefix_prev_message_count",
"agent": "agent",
# The classifier's own attempt. Detail popup, like the prefix probe above.
"classifier_confidence": "classifier_confidence",
"classifier_coverage": "classifier_coverage",
"classifier_reject": "classifier_reject",
}
# Columns deliberately NOT surfaced anywhere in the TUI get recorded here with a
# one-line reason (callers must keep the comment). A new schema column that lands
# in NEITHER registry fails the drift test.
UNSURFACED_COLUMNS: set[str] = {
"parent_key", # opaque id, recorded for later lineage queries, not displayed
}
# Every column seeded non-NULL: the single source of truth for the INSERT and
# the round-trip assertions, so the tests prove the column flows, not that
# None flows.
FULL_ROW = {
"id": 1,
"observed_at": "2026-09-05T12:34:56+00:00",
"kind": "chat",
"task_category": "coding_general",
"task_tier": 2,
"required_context_tokens": 123456,
"confidence": 0.91,
"classifier_ms": 1800,
"classification_source": "classifier",
"latency_tolerance": "interactive",
"candidates_considered": 7,
"selected_model": "fixture-model",
"selected_provider": "neuralwatt",
"runner_up_models": '[{"model_id": "runner", "provider": "neuralwatt"}]',
"est_cost_usd": 0.00123,
"est_proficiency": 0.88,
# Schema-legal even for a selected row (no CHECK constraint); non-NULL so the
# round-trip assertion proves the column flows, not that None flows.
"rejected_reason": "fixture rejected reason",
"session_key": "fixture-session-key",
"tools": 0,
"images": 0,
"json_mode": 1,
"streamed": 1,
"flex_preference": "auto",
"flex_swapped": 0,
"flex_forced": 0,
"request_id": "chatcmpl-fixture-42",
"exploration": 1,
"pinch_original_tokens": 120000,
"pinch_final_tokens": 96122,
"profile": "default",
# A destructive divergence: index 19 sits below the 81 messages the
# previous turn had, so the payload was rewritten rather than appended to.
"prefix_divergence_index": 19,
"prefix_tokens_after_divergence": 72091,
"prefix_prev_message_count": 81,
"agent": "classifier",
"parent_key": None,
"classifier_confidence": 0.431,
"classifier_coverage": 0.874,
"classifier_reject": "below_confidence_min",
}
def _db_from_schema(tmp_path, row_factory=None) -> sqlite3.Connection:
"""A throwaway DB from config/schema.sql — never the live router.db.
``row_factory`` defaults to None (plain tuples) like the PRAGMA test
needs; the metrics round-trip test passes ``sqlite3.Row`` because
``metrics.recent_decisions`` builds its rows via ``dict(row)``.
"""
conn = sqlite3.connect(tmp_path / "schema-drift.db")
conn.executescript(SCHEMA_SQL)
conn.row_factory = row_factory
return conn
def test_schema_columns_all_have_surfacing_decisions(tmp_path):
"""Every route_decisions column is either surfaced or consciously not."""
conn = _db_from_schema(tmp_path)
schema_cols = {
row[1] for row in conn.execute("PRAGMA table_info(route_decisions)")
}
conn.close()
registered = set(SCHEMA_TO_MODEL_KEYS) | UNSURFACED_COLUMNS
missing = schema_cols - set(SCHEMA_TO_MODEL_KEYS) - UNSURFACED_COLUMNS
extra = registered - schema_cols
assert not missing, (
f"route_decisions column(s) added without a surfacing decision: "
f"{sorted(missing)}. Add each to SCHEMA_TO_MODEL_KEYS if the TUI "
f"should surface it (project it in tui_model.decision_row), or to "
f"UNSURFACED_COLUMNS with a reason comment."
)
assert not extra, (
f"registry registers column(s) that no longer exist in "
f"route_decisions: {sorted(extra)}."
)
def test_decision_row_projects_every_tracked_column():
"""decision_row surfaces every registered column's value faithfully."""
projected = tui_model.decision_row(dict(FULL_ROW))
expected_keys = set(SCHEMA_TO_MODEL_KEYS.values())
assert set(projected) == expected_keys, (
"tui_model.decision_row no longer matches the surfacing registry "
f"(symmetric difference: {sorted(expected_keys ^ set(projected))}).\n"
f" registry values: {sorted(expected_keys)}\n"
f" decision_row keys: {sorted(projected)}"
)
for col, key in SCHEMA_TO_MODEL_KEYS.items():
assert projected[key] == FULL_ROW[col], (
f"route_decisions column {col!r} must surface as decision_row "
f"key {key!r}: got {projected[key]!r}, want {FULL_ROW[col]!r}"
)
def test_metrics_recent_decisions_selects_every_tracked_column(tmp_path):
"""The /metrics path carries every registered column end-to-end.
``request_id`` reaches the TUI ONLY through this SELECT (SSE publishes
before the backfill), so a narrowed SELECT silently erases forensics
fields a few seconds after a live decision arrives.
"""
conn = _db_from_schema(tmp_path, row_factory=sqlite3.Row)
cols = ",".join(FULL_ROW.keys())
placeholders = ",".join("?" * len(FULL_ROW))
conn.execute(
f"INSERT INTO route_decisions ({cols}) VALUES ({placeholders})",
tuple(FULL_ROW.values()),
)
conn.commit()
rows = metrics.recent_decisions(conn, limit=5)
conn.close()
assert rows, "the seeded FULL_ROW must come back from recent_decisions"
projected = tui_model.decision_row(rows[0])
unreachable = []
for col, key in SCHEMA_TO_MODEL_KEYS.items():
if projected.get(key) != FULL_ROW[col]:
unreachable.append(
f"route_decisions column {col!r} does not reach the TUI "
f"through /metrics — is it selected by "
f"metrics.recent_decisions()?"
)
assert not unreachable, "\n".join(unreachable)