Foundation for a selectable classifier primary, generalizing "which implementation answers a classification request" into config rather than always assuming the local Ollama model: - classifier.mode (default local_llm, unchanged behavior) plus cloud_primary / cloud_primary_auto for cloud_llm and an encoder block for local_encoder. Two new RouterConfig validators reject an incomplete combination at load time -- cloud_llm with neither/both primaries set, local_encoder with no encoder block -- the same model_validator(mode= "after") pattern this project already uses elsewhere. - routing.cheapest_classifier_candidate: the "auto_classifier" resolver. Reuses select_candidates + estimated_cost -- the same functions real dispatch ranking uses -- rather than a second cost model, priced for the classifier's own short-prompt/short-completion call shape (500/50 tokens, cache_rate 0) instead of the task's. required_tier=1 is a floor, not a ceiling, so a tier-3 model can still win on price -- a test pins this after an initial wrong assumption in the test itself. - local_encoder.py: zero-shot category classification via a non-generative encoder (default MoritzLaurer/deberta-v3-base-zeroshot-v2). Structurally immune to the one failure mode that has cost this project two prior classifier generations (docs/local-models.md): a generative model spending its budget on an unbounded reasoning trace. Zero-shot rather than fine-tuned, deliberately -- this router never stores raw task text anywhere, so there is no training corpus without a new, separate opt-in capture feature (scoped, not built). transformers/torch imported lazily inside the function, the same rule tui.py already follows for textual, so a deployment that never selects this mode needs neither installed. requirements-encoder.txt keeps them out of the main, pinned requirements file. Every test here was run against unmodified main first and observed to fail for the right reason before this commit made it pass. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
119 lines
4.3 KiB
Python
119 lines
4.3 KiB
Python
"""routing.cheapest_classifier_candidate -- backs classifier.cloud_primary_auto.
|
|
|
|
Reuses select_candidates + estimated_cost rather than a second cost model, so
|
|
these tests are mostly about the FILTER choices (tier 1, interactive, public,
|
|
non-stale/deprecated) and the tie-break (cheapest wins), not about pricing
|
|
arithmetic -- estimated_cost already has its own tests.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from routing import cheapest_classifier_candidate
|
|
|
|
|
|
def _row(**overrides) -> dict:
|
|
row = {
|
|
"model_id": "m",
|
|
"provider": "neuralwatt",
|
|
"tier": 1,
|
|
"effective_context_window": 100_000,
|
|
"availability": "active",
|
|
"deprecated": 0,
|
|
"access_level": "public",
|
|
"latency_class": "standard",
|
|
"reasoning_mode": "default",
|
|
"context_variant": "full",
|
|
"supports_vision": 1,
|
|
"supports_json_mode": 1,
|
|
"cost_per_1m_prompt": 1.0,
|
|
"cost_per_1m_completion": 1.0,
|
|
}
|
|
row.update(overrides)
|
|
return row
|
|
|
|
|
|
def test_picks_the_cheapest_of_several_candidates():
|
|
rows = [
|
|
_row(model_id="expensive", cost_per_1m_completion=10.0),
|
|
_row(model_id="cheap", cost_per_1m_completion=0.5),
|
|
_row(model_id="middle", cost_per_1m_completion=2.0),
|
|
]
|
|
picked = cheapest_classifier_candidate(rows)
|
|
assert picked["model_id"] == "cheap"
|
|
|
|
|
|
def test_tier_is_a_floor_not_a_ceiling_so_a_frontier_model_can_still_win():
|
|
"""required_tier=1 means "at least tier-1 capable", which every tier
|
|
satisfies (tier is a capability floor -- see CLAUDE.md). It does not
|
|
restrict candidates to cheap models; cost alone decides among them, and
|
|
a tier-3 model can still be cheapest for a short classification call."""
|
|
rows = [
|
|
_row(model_id="tier1", tier=1, cost_per_1m_completion=5.0),
|
|
_row(model_id="tier3-cheaper", tier=3, cost_per_1m_completion=0.1),
|
|
]
|
|
picked = cheapest_classifier_candidate(rows)
|
|
assert picked["model_id"] == "tier3-cheaper"
|
|
|
|
|
|
def test_flex_rows_are_excluded():
|
|
"""INTERACTIVE tolerance: a request already waiting on classification
|
|
should not also wait out a flex capacity gap."""
|
|
rows = [
|
|
_row(model_id="standard", cost_per_1m_completion=5.0),
|
|
_row(model_id="flex-cheaper", latency_class="flex", cost_per_1m_completion=0.1),
|
|
]
|
|
picked = cheapest_classifier_candidate(rows)
|
|
assert picked["model_id"] == "standard"
|
|
|
|
|
|
def test_stale_and_deprecated_rows_are_excluded():
|
|
rows = [
|
|
_row(model_id="active", cost_per_1m_completion=5.0),
|
|
_row(model_id="stale-cheaper", availability="stale", cost_per_1m_completion=0.1),
|
|
_row(model_id="deprecated-cheaper", deprecated=1, cost_per_1m_completion=0.1),
|
|
]
|
|
picked = cheapest_classifier_candidate(rows)
|
|
assert picked["model_id"] == "active"
|
|
|
|
|
|
def test_non_public_access_level_is_excluded_by_default():
|
|
rows = [
|
|
_row(model_id="public", cost_per_1m_completion=5.0),
|
|
_row(model_id="preview-cheaper", access_level="preview", cost_per_1m_completion=0.1),
|
|
]
|
|
picked = cheapest_classifier_candidate(rows)
|
|
assert picked["model_id"] == "public"
|
|
|
|
|
|
def test_allowed_access_levels_is_a_real_parameter_not_a_hardcoded_default():
|
|
"""Pure function per this module's own rule: thresholds are arguments,
|
|
not baked-in assumptions."""
|
|
rows = [_row(model_id="preview-only", access_level="preview")]
|
|
assert cheapest_classifier_candidate(rows) is None
|
|
picked = cheapest_classifier_candidate(
|
|
rows, allowed_access_levels=["public", "preview"]
|
|
)
|
|
assert picked["model_id"] == "preview-only"
|
|
|
|
|
|
def test_rows_with_no_catalog_price_are_ignored():
|
|
rows = [
|
|
_row(model_id="unpriced", cost_per_1m_prompt=None, cost_per_1m_completion=None),
|
|
_row(model_id="priced", cost_per_1m_completion=3.0),
|
|
]
|
|
picked = cheapest_classifier_candidate(rows)
|
|
assert picked["model_id"] == "priced"
|
|
|
|
|
|
def test_empty_catalog_returns_none():
|
|
assert cheapest_classifier_candidate([]) is None
|
|
|
|
|
|
def test_no_routable_candidates_returns_none():
|
|
rows = [_row(tier=None)] # unknown tier fails closed, per is_eligible
|
|
assert cheapest_classifier_candidate(rows) is None
|
|
|
|
|
|
def test_no_priced_candidates_returns_none():
|
|
rows = [_row(cost_per_1m_prompt=None, cost_per_1m_completion=None)]
|
|
assert cheapest_classifier_candidate(rows) is None
|