Both endpoints resolved `req.profile or "default"` -- the literal built-in profile named "default" -- and passed a non-None profile_obj, so route()'s cfg.routing.default_profile fallback was dead code on those paths. Only `auto` went through _resolve_profile, which does honour it. Those were the same thing until an operator set default_profile. After that, /route -- the documented no-spend probe of "what would the router pick" -- answered for a profile the router was not using, and wrote that profile's name into route_decisions. Found live, and the way it presented is the reason to fix it rather than document it. With default_profile = "MiMo Test": POST /route -> profile "default", deepseek-v4-flash POST /v1/chat/completions -> profile "MiMo Test", xiaomi/mimo-v2.5 (same task, same instant) Read back from the decisions history that looks like a broken profile overlay. The overlay was fine; the probe was lying, and it briefly convinced me too. An explicit profile still wins, so a caller can still probe one deliberately, including the built-in called "default" -- test_an_explicit_profile_still_wins_over_the_configured_default pins that the narrowing stays narrow. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
815 lines
30 KiB
Python
815 lines
30 KiB
Python
"""Integration pins for the five preserved behaviors of named routing profiles.
|
|
|
|
This file covers Todo 7 of the named-routing-profiles plan. These tests cut
|
|
across the dispatcher / routing boundary: `route` is invoked directly with a
|
|
real routing profile, against a monkeypatched SQLite catalog in a temp file.
|
|
No classifier, provider, or local model is ever called.
|
|
|
|
The five planks:
|
|
|
|
(a) ``auto`` and ``auto:batch`` behavior is byte-identical to the old ternary:
|
|
route() with the batch profile must return the same selection and
|
|
runner-up ordering as the default profile with request
|
|
``latency_tolerance=batch``.
|
|
(b) Unknown profiles return 422 naming valid profiles. Coverage for
|
|
``/v1/chat/completions`` unknown profiles exists in
|
|
tests/test_chat_completions.py; we do not duplicate it here.
|
|
(c) Empty profile allowlist fails with a profile-specific reason
|
|
("profile_excluded") and the reason is persisted to route_decisions
|
|
when no candidate survives.
|
|
(d) Profiles AND with circuit-breaker exclusions, admin overrides, and
|
|
``eligible_categories``.
|
|
(e) Exploration picks only from profile-filtered candidates.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import random
|
|
import sqlite3
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
import pytest
|
|
from starlette.testclient import TestClient
|
|
|
|
import circuit_breaker
|
|
import dispatcher
|
|
from config import RoutingProfile
|
|
from dispatcher import (
|
|
BUILTIN_PROFILES,
|
|
TaskRequest,
|
|
app,
|
|
persist_route_decision,
|
|
route,
|
|
)
|
|
from routing import rejection_reason, select_candidates
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
|
ADMIN_SCHEMA_SQL = (ROOT / "config" / "admin_schema.sql").read_text()
|
|
|
|
M1 = "m1-cloud"
|
|
M1_FLEX = "m1-cloud-flex"
|
|
M2 = "m2-cloud"
|
|
LOCAL = "m3-local"
|
|
LOCAL2 = "m3b-local"
|
|
|
|
|
|
def _model_row(
|
|
model_id: str,
|
|
*,
|
|
provider: str = "neuralwatt",
|
|
tier: int = 2,
|
|
latency_class: str = "standard",
|
|
completion_price: float = 1.0,
|
|
prompt_price: Optional[float] = None,
|
|
context_window: int = 262128,
|
|
effective_context_window: int = 192500,
|
|
supports_vision: int = 0,
|
|
supports_json_mode: int = 1,
|
|
eligible_categories: Optional[str] = None,
|
|
) -> dict:
|
|
"""A models-table row, shaped for the fixture's INSERT."""
|
|
return {
|
|
"model_id": model_id,
|
|
"provider": provider,
|
|
"base_model_id": model_id,
|
|
"tier": tier,
|
|
"context_window": context_window,
|
|
"effective_context_window": effective_context_window,
|
|
"max_output_tokens": 16384,
|
|
"cost_per_1m_prompt": (
|
|
prompt_price if prompt_price is not None else completion_price / 3
|
|
),
|
|
"cost_per_1m_completion": completion_price,
|
|
"cost_per_1m_prompt_cached": (
|
|
prompt_price if prompt_price is not None else completion_price / 3
|
|
),
|
|
"supports_vision": supports_vision,
|
|
"supports_json_mode": supports_json_mode,
|
|
"latency_class": latency_class,
|
|
"reasoning_mode": "default",
|
|
"context_variant": "full",
|
|
"access_level": "public",
|
|
"availability": "active",
|
|
"eligible_categories": eligible_categories,
|
|
"last_updated": "2026-08-22T00:00:00+00:00",
|
|
}
|
|
|
|
|
|
def _insert_model(conn: sqlite3.Connection, row: dict) -> None:
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, tier, context_window,
|
|
effective_context_window, max_output_tokens,
|
|
cost_per_1m_prompt, cost_per_1m_completion, cost_per_1m_prompt_cached,
|
|
supports_vision, supports_json_mode,
|
|
latency_class, reasoning_mode, context_variant,
|
|
access_level, availability, eligible_categories, last_updated
|
|
) VALUES (
|
|
:model_id, :provider, :base_model_id, :tier, :context_window,
|
|
:effective_context_window, :max_output_tokens,
|
|
:cost_per_1m_prompt, :cost_per_1m_completion, :cost_per_1m_prompt_cached,
|
|
:supports_vision, :supports_json_mode,
|
|
:latency_class, :reasoning_mode, :context_variant,
|
|
:access_level, :availability, :eligible_categories, :last_updated
|
|
)
|
|
""",
|
|
row,
|
|
)
|
|
|
|
|
|
def _insert_proficiency(
|
|
conn: sqlite3.Connection,
|
|
model_id: str,
|
|
provider: str,
|
|
*,
|
|
blended_score: float,
|
|
outcome_samples: int = 0,
|
|
) -> None:
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO proficiency (
|
|
model_id, provider, category, blended_score, source,
|
|
outcome_samples, last_updated
|
|
) VALUES (?, ?, 'coding_general', ?, 'leaderboard', ?,
|
|
'2026-08-22T00:00:00+00:00')
|
|
""",
|
|
(model_id, provider, blended_score, outcome_samples),
|
|
)
|
|
|
|
|
|
class FakeResponse:
|
|
"""Just enough of requests.Response for the stubbed provider POST."""
|
|
|
|
def __init__(self, payload=None):
|
|
self.status_code = 200
|
|
self._payload = payload or {}
|
|
self.text = json.dumps(self._payload)
|
|
self.headers = {}
|
|
self.closed = False
|
|
|
|
def json(self):
|
|
return self._payload
|
|
|
|
def iter_lines(self, decode_unicode=False):
|
|
return iter([])
|
|
|
|
def close(self):
|
|
self.closed = True
|
|
|
|
|
|
@pytest.fixture
|
|
def profile_router(tmp_path, monkeypatch):
|
|
"""The dispatcher pointed at a curated throwaway catalog, nothing dialled out.
|
|
|
|
Candidate economics are chosen so the routing outcome is fully determined:
|
|
|
|
- M1 / M1_FLEX tie on proficiency (0.9); the flex twin is cheaper, so under
|
|
batch latency it outranks standard, and under interactive it is filtered
|
|
out entirely by the latency_class hard filter.
|
|
- In exploration tests M2 has zero outcome samples, so an epsilon=1.0 run
|
|
with the default profile picks it (cloud), while the locality profile's
|
|
candidate set contains only the two ollama-local rows.
|
|
"""
|
|
db_path = tmp_path / "profiles.db"
|
|
conn = sqlite3.connect(db_path)
|
|
conn.executescript(SCHEMA_SQL)
|
|
conn.executescript(ADMIN_SCHEMA_SQL)
|
|
|
|
_insert_model(conn, _model_row(M1, completion_price=3.0))
|
|
_insert_model(
|
|
conn, _model_row(M1_FLEX, completion_price=2.0, latency_class="flex")
|
|
)
|
|
_insert_model(conn, _model_row(M2, completion_price=9.0))
|
|
_insert_model(
|
|
conn,
|
|
_model_row(
|
|
LOCAL, provider="ollama-local", completion_price=5.0, prompt_price=1.0
|
|
),
|
|
)
|
|
_insert_model(
|
|
conn,
|
|
_model_row(
|
|
LOCAL2, provider="ollama-local", completion_price=1.0, prompt_price=0.33
|
|
),
|
|
)
|
|
|
|
_insert_proficiency(conn, M1, "neuralwatt", blended_score=0.90, outcome_samples=50)
|
|
_insert_proficiency(conn, M1_FLEX, "neuralwatt", blended_score=0.90, outcome_samples=50)
|
|
_insert_proficiency(conn, M2, "neuralwatt", blended_score=0.85, outcome_samples=0)
|
|
_insert_proficiency(conn, LOCAL, "ollama-local", blended_score=0.88, outcome_samples=50)
|
|
# 0.70 sits a full quality-tolerance band below the 0.88/0.90 rows
|
|
# (0.9 - 0.8 would land there only by float luck: 0.099999... rounds to
|
|
# band 0). Band 1 keeps LOCAL2 eligible-but-ranked-last under the default
|
|
# profile while still being the cheapest local row under locality.
|
|
_insert_proficiency(conn, LOCAL2, "ollama-local", blended_score=0.70, outcome_samples=50)
|
|
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
monkeypatch.setattr(dispatcher.cfg.database, "path", str(db_path))
|
|
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
|
|
monkeypatch.setattr(dispatcher.cfg.local_vision, "enabled", False)
|
|
monkeypatch.setattr(dispatcher.cfg.session_cache, "enabled", False)
|
|
# Exploration is pinned off for determinism; the exploration tests
|
|
# re-enable it explicitly.
|
|
monkeypatch.setattr(dispatcher.cfg.exploration, "enabled", False)
|
|
monkeypatch.setattr(dispatcher.cfg.exploration, "epsilon", 0.0)
|
|
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
|
|
|
|
calls = []
|
|
|
|
def fake_post(url, headers=None, json=None, stream=False, timeout=None):
|
|
calls.append({"url": url, "body": json, "stream": stream})
|
|
return FakeResponse(
|
|
{
|
|
"id": "chatcmpl-test-1",
|
|
"model": json["model"],
|
|
"choices": [
|
|
{"message": {"role": "assistant", "content": "ok"},
|
|
"finish_reason": "stop"}
|
|
],
|
|
"usage": {"prompt_tokens": 31, "completion_tokens": 12},
|
|
"energy": {"energy_kwh": 5e-05},
|
|
"cost": {"request_cost_usd": 4e-04},
|
|
}
|
|
)
|
|
|
|
monkeypatch.setattr(dispatcher.requests, "post", fake_post)
|
|
monkeypatch.setattr(
|
|
dispatcher, "classify",
|
|
lambda task, context: dispatcher.Classification(
|
|
task_category="coding_general", task_tier=2,
|
|
required_context_tokens=100, confidence=0.9,
|
|
),
|
|
)
|
|
|
|
yield TestClient(app), db_path, calls
|
|
|
|
circuit_breaker.clear()
|
|
|
|
|
|
@pytest.fixture
|
|
def logbuf():
|
|
"""Capture what would reach the journal, through the real formatter."""
|
|
import io
|
|
|
|
import logs
|
|
|
|
buf = io.StringIO()
|
|
logs.configure("debug", stream=buf, journald=False)
|
|
yield buf
|
|
for handler in list(logs.log.handlers):
|
|
logs.log.removeHandler(handler)
|
|
|
|
|
|
def _route_req(
|
|
*,
|
|
latency_tolerance: Optional[str] = None,
|
|
task_category: str = "coding_general",
|
|
):
|
|
return TaskRequest(
|
|
task="write me a function",
|
|
task_category=task_category,
|
|
task_tier=2,
|
|
required_context_tokens=100,
|
|
latency_tolerance=latency_tolerance,
|
|
)
|
|
|
|
|
|
def _hard_filters(exclude_models: set[str] = frozenset()) -> dict:
|
|
"""The hard-filter kwargs every direct rejection_reason/select_candidates
|
|
call in this file shares, matching what route() itself passes."""
|
|
return dict(
|
|
required_context_tokens=100,
|
|
required_tier=2,
|
|
latency_tolerance="interactive",
|
|
allowed_access_levels=["public"],
|
|
exclude_stale=True,
|
|
exclude_deprecated=True,
|
|
exclude_models=exclude_models,
|
|
)
|
|
|
|
|
|
# =============================================================================
|
|
# Plank (a): auto and auto:batch behavior is byte-identical to the old ternary.
|
|
# =============================================================================
|
|
|
|
|
|
def test_auto_batch_matches_default_with_batch_latency(profile_router):
|
|
"""route() under the batch profile must equal the default profile plus
|
|
request latency_tolerance=batch, byte for byte.
|
|
|
|
The batch profile's only field is latency_tolerance, and a request-level
|
|
batch tolerance on the default profile reaches the same effective latency;
|
|
both must produce the same selection, runner-up ordering, and candidate
|
|
count. A leftover special-case ternary that treated ``auto:batch``
|
|
differently from an explicit batch request would break this.
|
|
"""
|
|
_, _, _ = profile_router
|
|
|
|
via_request = route(
|
|
_route_req(latency_tolerance="batch"),
|
|
profile="default",
|
|
profile_obj=BUILTIN_PROFILES["default"],
|
|
)
|
|
via_profile = route(
|
|
_route_req(latency_tolerance="batch"),
|
|
profile="batch",
|
|
profile_obj=BUILTIN_PROFILES["batch"],
|
|
)
|
|
|
|
assert via_request.selected is not None
|
|
assert via_profile.selected is not None
|
|
assert via_request.selected.model_id == via_profile.selected.model_id
|
|
assert [r.model_id for r in via_request.runners_up] == [
|
|
r.model_id for r in via_profile.runners_up
|
|
]
|
|
assert via_request.candidates_considered == via_profile.candidates_considered
|
|
assert via_request.latency_tolerance == via_profile.latency_tolerance == "batch"
|
|
# The flex twin is cheaper at equal proficiency, so batch must admit it and
|
|
# rank it first: this is what makes the batch equivalence observable.
|
|
assert via_profile.selected.model_id == M1_FLEX
|
|
assert via_profile.selected.latency_class == "flex"
|
|
|
|
|
|
def test_interactive_default_excludes_flex_rows_batch_admits_them(profile_router):
|
|
"""Without batch tolerance, the flex twin is latency-filtered out; the
|
|
selection therefore differs, proving the flex scenario is live."""
|
|
_, _, _ = profile_router
|
|
|
|
interactive = route(
|
|
_route_req(), profile="default", profile_obj=BUILTIN_PROFILES["default"]
|
|
)
|
|
batch = route(
|
|
_route_req(latency_tolerance="batch"),
|
|
profile="batch",
|
|
profile_obj=BUILTIN_PROFILES["batch"],
|
|
)
|
|
|
|
assert interactive.selected is not None
|
|
assert batch.selected is not None
|
|
assert interactive.selected.model_id == M1
|
|
assert interactive.selected.latency_class == "standard"
|
|
assert batch.selected.model_id == M1_FLEX
|
|
assert batch.selected.latency_class == "flex"
|
|
# The flex twin only enters the candidate set under batch tolerance.
|
|
assert interactive.candidates_considered < batch.candidates_considered
|
|
|
|
|
|
def test_chat_auto_batch_selects_same_model_as_auto_under_batch_default(
|
|
profile_router, monkeypatch
|
|
):
|
|
"""``POST /v1/chat/completions`` with model=auto:batch returns the same
|
|
selected model as model=auto when the configured default profile is batch.
|
|
|
|
Since Wave A config profiles may not shadow built-in names, the test sets
|
|
``routing.default_profile`` to a new config profile named ``batch-local``
|
|
whose only override is ``latency_tolerance=batch``. Both ``auto`` and
|
|
``auto:batch`` must then pick the flex twin through the profile layer.
|
|
"""
|
|
client, _, _ = profile_router
|
|
monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch-local")
|
|
monkeypatch.setitem(
|
|
dispatcher.cfg.profiles, "batch-local", RoutingProfile(latency_tolerance="batch")
|
|
)
|
|
|
|
auto = client.post(
|
|
"/v1/chat/completions",
|
|
json={"model": "auto", "messages": [{"role": "user", "content": "hi"}]},
|
|
)
|
|
auto_batch = client.post(
|
|
"/v1/chat/completions",
|
|
json={"model": "auto:batch", "messages": [{"role": "user", "content": "hi"}]},
|
|
)
|
|
|
|
assert auto.status_code == 200
|
|
assert auto_batch.status_code == 200
|
|
assert auto.headers["X-Router-Model"] == M1_FLEX
|
|
assert auto_batch.headers["X-Router-Model"] == M1_FLEX
|
|
assert auto.json()["model"] == auto_batch.json()["model"] == M1_FLEX
|
|
|
|
|
|
# =============================================================================
|
|
# Plank (b): Unknown profile returns 422 naming valid profiles.
|
|
# =============================================================================
|
|
|
|
|
|
def test_route_unknown_profile_returns_422_naming_valid_profiles(profile_router):
|
|
"""A misspelled bare profile on /route must name the valid choices."""
|
|
client, _, _ = profile_router
|
|
resp = client.post("/route", json={"task": "x", "profile": "nosuchprofile"})
|
|
assert resp.status_code == 422
|
|
detail = resp.json()["detail"]
|
|
assert "nosuchprofile" in detail
|
|
assert "default" in detail
|
|
assert "batch" in detail
|
|
assert "locality" in detail
|
|
|
|
|
|
# ``/v1/chat/completions`` unknown-profile coverage is NOT duplicated here.
|
|
# See tests/test_chat_completions.py::test_unknown_profile_returns_422_naming_valid_profiles
|
|
# for the auto:<unknown> 422 naming default/batch, added in Todos 3/4.
|
|
|
|
|
|
# =============================================================================
|
|
# Plank (c): Empty profile allowlist fails with a profile-specific reason.
|
|
# =============================================================================
|
|
|
|
|
|
def test_ghost_allowlist_selects_nothing_and_persists_profile_excluded(profile_router):
|
|
"""A profile whose allowed_model_ids intersect no catalog row selects
|
|
nothing, debug-logs ``profile_excluded`` for every candidate, and can
|
|
persist that reason through the persist_route_decision seam."""
|
|
_, db_path, _ = profile_router
|
|
|
|
profile = RoutingProfile(allowed_model_ids={"nonexistent-model"})
|
|
decision = route(_route_req(), profile="ghost", profile_obj=profile)
|
|
|
|
assert decision.selected is None
|
|
assert decision.candidates_considered == 0
|
|
assert decision.profile == "ghost"
|
|
|
|
# The allowlist resolves to an empty restrict_to set: nothing intersects.
|
|
conn = sqlite3.connect(db_path)
|
|
conn.row_factory = sqlite3.Row
|
|
rows = dispatcher.load_candidates(conn, "coding_general")
|
|
admin_deprecated = dispatcher._admin_excluded_models(conn)
|
|
conn.close()
|
|
restrict = dispatcher._restrict_to_from_profile(profile, rows)
|
|
assert restrict == set()
|
|
assert admin_deprecated == set()
|
|
|
|
# Every catalog row carries a profile-era rejection reason. The flex twin
|
|
# is caught one filter earlier (latency precedes the profile check in the
|
|
# documented rejection_reason ordering); everything else lands on the
|
|
# profile-specific reason.
|
|
for row in rows:
|
|
reason = rejection_reason(
|
|
row, restrict_to=restrict, **_hard_filters()
|
|
)
|
|
if row["model_id"] == M1_FLEX:
|
|
assert reason == "latency_class(flex)"
|
|
else:
|
|
assert reason == "profile_excluded"
|
|
|
|
# And that reason, persisted through the production write path, lands on
|
|
# the route_decisions row for the no-candidate decision.
|
|
persist_route_decision(
|
|
"route", classification=decision, rejected_reason="profile_excluded"
|
|
)
|
|
|
|
conn = sqlite3.connect(db_path)
|
|
conn.row_factory = sqlite3.Row
|
|
row = conn.execute(
|
|
"SELECT rejected_reason, selected_model, profile FROM route_decisions "
|
|
"ORDER BY id DESC LIMIT 1"
|
|
).fetchone()
|
|
conn.close()
|
|
assert row is not None
|
|
assert row["selected_model"] is None
|
|
assert row["profile"] == "ghost"
|
|
assert row["rejected_reason"] is not None
|
|
assert row["rejected_reason"].startswith("profile_excluded")
|
|
|
|
|
|
def test_debug_filter_lines_exist_for_profile_exclusions(profile_router, logbuf):
|
|
"""Every catalog row dropped by the allowlist gets a ``filter`` line.
|
|
|
|
Known limitation, pinned as-is: route()'s debug loop calls
|
|
``rejection_reason(row, **filters)`` WITHOUT ``restrict_to``, so the
|
|
logged reason for a profile-excluded row is the placeholder ``-`` rather
|
|
than ``profile_excluded``. The authoritative reason (via
|
|
``rejection_reason(..., restrict_to=...)``) is pinned in the ghost
|
|
test above; this test pins that the dropped rows at least surface in
|
|
the debug log at all. A fix that threads restrict_to into the loop
|
|
should update the ``== "-"`` expectations to ``== "profile_excluded"``.
|
|
"""
|
|
_, _, _ = profile_router
|
|
route(
|
|
_route_req(),
|
|
profile="ghost",
|
|
profile_obj=RoutingProfile(allowed_model_ids={"nonexistent-model"}),
|
|
)
|
|
|
|
reasons: dict[str, str] = {}
|
|
for line in logbuf.getvalue().splitlines():
|
|
if line.startswith("filter "):
|
|
fields = dict(
|
|
token.split("=", 1) for token in line.split(" ")[1:] if "=" in token
|
|
)
|
|
reasons[fields.get("model", "")] = fields.get("reason", "")
|
|
|
|
# Nothing survived, so every catalog row produced a filter line.
|
|
for model_id in (M1, M1_FLEX, M2, LOCAL, LOCAL2):
|
|
assert model_id in reasons, model_id
|
|
# The flex row is dropped by the latency filter (reason visible), the
|
|
# rest by the allowlist (reason currently '-').
|
|
assert reasons[M1_FLEX] == "latency_class(flex)"
|
|
for model_id in (M1, M2, LOCAL, LOCAL2):
|
|
assert reasons[model_id] == "-"
|
|
|
|
|
|
def test_restriction_set_passes_rows_inside_the_allowlist(profile_router):
|
|
"""The same restrict_to set keeps rows it names — the set is precise, not
|
|
an unconditional kill switch."""
|
|
_, db_path, _ = profile_router
|
|
|
|
profile = RoutingProfile(allowed_model_ids={M1, M2})
|
|
conn = sqlite3.connect(db_path)
|
|
conn.row_factory = sqlite3.Row
|
|
rows = dispatcher.load_candidates(conn, "coding_general")
|
|
conn.close()
|
|
|
|
restrict = dispatcher._restrict_to_from_profile(profile, rows)
|
|
assert restrict == {M1, M2}
|
|
|
|
survivors = select_candidates(rows, restrict_to=restrict, **_hard_filters())
|
|
assert {r["model_id"] for r in survivors} == {M1, M2}
|
|
|
|
|
|
# =============================================================================
|
|
# Plank (d): Profiles AND with circuit-breaker, admin overrides,
|
|
# and eligible_categories.
|
|
# =============================================================================
|
|
|
|
|
|
def test_profile_excludes_circuit_open_model(profile_router, monkeypatch):
|
|
"""A profile that allows only m1 must still drop it when the circuit
|
|
breaker has it open, and the rejection reason is the exclusion reason."""
|
|
_, _, _ = profile_router
|
|
|
|
monkeypatch.setattr(dispatcher.cfg.circuit_breaker, "enabled", True)
|
|
circuit_breaker.record_failure(
|
|
M1,
|
|
"neuralwatt",
|
|
initial_cooldown=dispatcher.cfg.circuit_breaker.initial_cooldown_seconds,
|
|
max_cooldown=dispatcher.cfg.circuit_breaker.max_cooldown_seconds,
|
|
backoff_multiplier=dispatcher.cfg.circuit_breaker.backoff_multiplier,
|
|
)
|
|
|
|
profile = RoutingProfile(allowed_model_ids={M1})
|
|
decision = route(_route_req(), profile="m1-only", profile_obj=profile)
|
|
assert decision.selected is None
|
|
assert decision.candidates_considered == 0
|
|
|
|
# The composition is visible at the primitive level too: the exclude_models
|
|
# set (circuit-open rows) drops m1 even though the profile admits it.
|
|
row = _model_row(M1)
|
|
assert (
|
|
rejection_reason(row, restrict_to={M1}, **_hard_filters(exclude_models={M1}))
|
|
== "circuit_open"
|
|
)
|
|
|
|
|
|
def test_profile_excludes_admin_deprecated_model(profile_router):
|
|
"""A profile that allows only m1 must drop it when an admin override
|
|
deprecates it; the survivor set is empty."""
|
|
_, db_path, _ = profile_router
|
|
|
|
conn = sqlite3.connect(db_path)
|
|
conn.execute(
|
|
"INSERT INTO admin_model_overrides "
|
|
"(model_id, provider, availability, reason, updated_at) "
|
|
"VALUES (?, 'neuralwatt', 'deprecated', 'test', "
|
|
"'2026-08-22T00:00:00+00:00')",
|
|
(M1,),
|
|
)
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
profile = RoutingProfile(allowed_model_ids={M1})
|
|
decision = route(_route_req(), profile="m1-only", profile_obj=profile)
|
|
|
|
assert decision.selected is None
|
|
assert decision.candidates_considered == 0
|
|
|
|
|
|
def test_profile_excludes_category_ineligible_model(profile_router):
|
|
"""A profile that allows only m1 must drop it when m1's
|
|
eligible_categories excludes the task's category."""
|
|
_, db_path, _ = profile_router
|
|
|
|
conn = sqlite3.connect(db_path)
|
|
conn.execute(
|
|
"UPDATE models SET eligible_categories = 'coding_refactor' "
|
|
"WHERE model_id = ?",
|
|
(M1,),
|
|
)
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
profile = RoutingProfile(allowed_model_ids={M1})
|
|
req = TaskRequest(
|
|
task="summarize this",
|
|
task_category="summarization",
|
|
task_tier=2,
|
|
required_context_tokens=100,
|
|
)
|
|
decision = route(req, profile="m1-only", profile_obj=profile)
|
|
|
|
assert decision.selected is None
|
|
assert decision.candidates_considered == 0
|
|
|
|
# And the primitive composes: the category gate fires even though the
|
|
# profile allowlist names the row.
|
|
row = _model_row(M1, eligible_categories="coding_refactor")
|
|
assert (
|
|
rejection_reason(
|
|
row, task_category="summarization", restrict_to={M1}, **_hard_filters()
|
|
)
|
|
== "category_ineligible"
|
|
)
|
|
|
|
|
|
def test_profile_of_two_models_selects_the_survivor(profile_router):
|
|
"""When the profile allows both m1 and m2 and an admin override kills m1,
|
|
the other candidate still wins — the filters compose without wiping the
|
|
whole profile set."""
|
|
_, db_path, _ = profile_router
|
|
|
|
conn = sqlite3.connect(db_path)
|
|
conn.execute(
|
|
"INSERT INTO admin_model_overrides "
|
|
"(model_id, provider, availability, reason, updated_at) "
|
|
"VALUES (?, 'neuralwatt', 'deprecated', 'test', "
|
|
"'2026-08-22T00:00:00+00:00')",
|
|
(M1,),
|
|
)
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
profile = RoutingProfile(allowed_model_ids={M1, M2})
|
|
decision = route(_route_req(), profile="m1-or-m2", profile_obj=profile)
|
|
|
|
assert decision.selected is not None
|
|
assert decision.selected.model_id == M2
|
|
assert decision.candidates_considered == 1
|
|
|
|
|
|
# =============================================================================
|
|
# Plank (e): Exploration picks only from profile-filtered candidates.
|
|
# =============================================================================
|
|
|
|
|
|
def _force_exploration(monkeypatch, *, seed: int = 0) -> None:
|
|
monkeypatch.setattr(dispatcher.cfg.exploration, "enabled", True)
|
|
monkeypatch.setattr(dispatcher.cfg.exploration, "epsilon", 1.0)
|
|
monkeypatch.setattr(dispatcher.cfg.exploration, "max_tier", 3)
|
|
# The local rows price above the cloud winner by more than the default
|
|
# cap of 4.0; widen it so the explore path is reachable at all.
|
|
monkeypatch.setattr(dispatcher.cfg.exploration, "max_cost_ratio", 1_000.0)
|
|
monkeypatch.setattr(dispatcher, "_router_rng", random.Random(seed))
|
|
|
|
|
|
def test_exploration_never_leaves_the_locality_profile(profile_router, monkeypatch):
|
|
"""With exploration forced on, the locality profile restricts candidates
|
|
to ollama-local rows, so every explored pick is local — exploration
|
|
cannot reach a cloud row it never sees."""
|
|
_, _, _ = profile_router
|
|
_force_exploration(monkeypatch)
|
|
|
|
profile = BUILTIN_PROFILES["locality"]
|
|
seen: set[tuple[str, str]] = set()
|
|
explored_any = False
|
|
for _ in range(30):
|
|
decision = route(_route_req(), profile="locality", profile_obj=profile)
|
|
assert decision.selected is not None
|
|
seen.add((decision.selected.model_id, decision.selected.provider))
|
|
explored_any = explored_any or decision.explored
|
|
|
|
assert seen, "the locality profile must route to something"
|
|
assert explored_any, "with two local candidates and epsilon=1 the explore path runs"
|
|
assert {model_id for model_id, _ in seen} <= {LOCAL, LOCAL2}
|
|
assert all(provider == "ollama-local" for _, provider in seen)
|
|
|
|
|
|
def test_exploration_default_profile_can_reach_a_cloud_row(
|
|
profile_router, monkeypatch
|
|
):
|
|
"""Positive control: same rng, same epsilon, default profile — exploration
|
|
does pick a cloud row (the zero-sample one). This is what makes the
|
|
locality assertion above meaningful rather than vacuous."""
|
|
_, _, _ = profile_router
|
|
_force_exploration(monkeypatch)
|
|
|
|
seen: set[tuple[str, str]] = set()
|
|
for _ in range(30):
|
|
decision = route(
|
|
_route_req(), profile="default", profile_obj=BUILTIN_PROFILES["default"]
|
|
)
|
|
assert decision.selected is not None
|
|
seen.add((decision.selected.model_id, decision.selected.provider))
|
|
|
|
assert any(provider == "neuralwatt" for _, provider in seen), seen
|
|
assert any(model_id == M2 for model_id, _ in seen), (
|
|
"the zero-sample cloud row is exactly what exploration targets"
|
|
)
|
|
|
|
|
|
def test_exploration_stays_inside_an_allowed_model_ids_profile(
|
|
profile_router, monkeypatch
|
|
):
|
|
"""An allowed_model_ids profile pins exploration to its allowlist."""
|
|
_, _, _ = profile_router
|
|
_force_exploration(monkeypatch)
|
|
|
|
allowlist = {M1, M2}
|
|
profile = RoutingProfile(allowed_model_ids=allowlist)
|
|
for _ in range(30):
|
|
decision = route(_route_req(), profile="m1-or-m2", profile_obj=profile)
|
|
assert decision.selected is not None
|
|
assert decision.selected.model_id in allowlist
|
|
|
|
|
|
# =============================================================================
|
|
# Route endpoint round-trip: profiles resolve and persist end to end.
|
|
# =============================================================================
|
|
|
|
|
|
def test_route_endpoint_batch_profile_round_trip(profile_router):
|
|
"""/route with profile=batch stores profile and batch latency on the row."""
|
|
client, db_path, _ = profile_router
|
|
|
|
resp = client.post(
|
|
"/route", json={"task": "write me a function", "profile": "batch"}
|
|
)
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
assert data["profile"] == "batch"
|
|
assert data["latency_tolerance"] == "batch"
|
|
assert data["selected"]["model_id"] == M1_FLEX
|
|
|
|
conn = sqlite3.connect(db_path)
|
|
conn.row_factory = sqlite3.Row
|
|
row = conn.execute(
|
|
"SELECT profile, latency_tolerance, selected_model "
|
|
"FROM route_decisions ORDER BY id DESC LIMIT 1"
|
|
).fetchone()
|
|
conn.close()
|
|
assert row is not None
|
|
assert row["profile"] == "batch"
|
|
assert row["latency_tolerance"] == "batch"
|
|
assert row["selected_model"] == M1_FLEX
|
|
|
|
|
|
def test_bare_auto_resolves_to_configured_default_profile(profile_router, monkeypatch):
|
|
"""A bare ``auto`` model fills from ``routing.default_profile``, not a hard-coded "default".
|
|
|
|
Monkeypatching ``cfg.routing.default_profile = "batch"`` causes a route
|
|
whose profile kwarg is the default to behave exactly like ``auto:batch``.
|
|
"""
|
|
_, _, _ = profile_router
|
|
monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch")
|
|
|
|
decision = route(_route_req())
|
|
assert decision.latency_tolerance == "batch"
|
|
assert decision.selected is not None
|
|
assert decision.selected.model_id == M1_FLEX
|
|
assert decision.selected.latency_class == "flex"
|
|
|
|
|
|
def test_the_route_endpoint_honours_the_configured_default_profile(
|
|
profile_router, monkeypatch
|
|
):
|
|
"""/route is the documented no-spend probe of "what would the router
|
|
pick". It resolved the literal built-in named "default" rather than
|
|
routing.default_profile, so it answered a different question than the
|
|
router answers -- and wrote the wrong name into route_decisions.
|
|
|
|
Found live: with default_profile = "MiMo Test", /route logged
|
|
profile="default" and picked deepseek-v4-flash, while the same task
|
|
through ``auto`` logged "MiMo Test" and picked xiaomi/mimo-v2.5. Read
|
|
back from the decisions history, that looks like a broken profile
|
|
overlay rather than a broken probe.
|
|
"""
|
|
client, _, _ = profile_router
|
|
monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch")
|
|
|
|
resp = client.post("/route", json={"task": "write me a function"})
|
|
|
|
assert resp.status_code == 200
|
|
body = resp.json()
|
|
assert body["profile"] == "batch"
|
|
assert body["latency_tolerance"] == "batch"
|
|
assert body["selected"]["model_id"] == M1_FLEX
|
|
|
|
|
|
def test_an_explicit_profile_still_wins_over_the_configured_default(
|
|
profile_router, monkeypatch
|
|
):
|
|
"""The narrowing stays narrow: a caller naming a profile is deliberately
|
|
probing that one, including the built-in called "default"."""
|
|
client, _, _ = profile_router
|
|
monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch")
|
|
|
|
resp = client.post(
|
|
"/route", json={"task": "write me a function", "profile": "default"}
|
|
)
|
|
|
|
assert resp.status_code == 200
|
|
assert resp.json()["profile"] == "default"
|