Files
6krrt/tests/test_routing_profiles_integration.py
adlee-was-taken 6ecd134e43 fix(api): /route and /dispatch honour routing.default_profile
Both endpoints resolved `req.profile or "default"` -- the literal
built-in profile named "default" -- and passed a non-None profile_obj,
so route()'s cfg.routing.default_profile fallback was dead code on those
paths. Only `auto` went through _resolve_profile, which does honour it.

Those were the same thing until an operator set default_profile. After
that, /route -- the documented no-spend probe of "what would the router
pick" -- answered for a profile the router was not using, and wrote that
profile's name into route_decisions.

Found live, and the way it presented is the reason to fix it rather than
document it. With default_profile = "MiMo Test":

  POST /route                 -> profile "default",   deepseek-v4-flash
  POST /v1/chat/completions   -> profile "MiMo Test", xiaomi/mimo-v2.5
  (same task, same instant)

Read back from the decisions history that looks like a broken profile
overlay. The overlay was fine; the probe was lying, and it briefly
convinced me too.

An explicit profile still wins, so a caller can still probe one
deliberately, including the built-in called "default" --
test_an_explicit_profile_still_wins_over_the_configured_default pins
that the narrowing stays narrow.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
2026-09-09 21:19:09 -04:00

815 lines
30 KiB
Python

"""Integration pins for the five preserved behaviors of named routing profiles.
This file covers Todo 7 of the named-routing-profiles plan. These tests cut
across the dispatcher / routing boundary: `route` is invoked directly with a
real routing profile, against a monkeypatched SQLite catalog in a temp file.
No classifier, provider, or local model is ever called.
The five planks:
(a) ``auto`` and ``auto:batch`` behavior is byte-identical to the old ternary:
route() with the batch profile must return the same selection and
runner-up ordering as the default profile with request
``latency_tolerance=batch``.
(b) Unknown profiles return 422 naming valid profiles. Coverage for
``/v1/chat/completions`` unknown profiles exists in
tests/test_chat_completions.py; we do not duplicate it here.
(c) Empty profile allowlist fails with a profile-specific reason
("profile_excluded") and the reason is persisted to route_decisions
when no candidate survives.
(d) Profiles AND with circuit-breaker exclusions, admin overrides, and
``eligible_categories``.
(e) Exploration picks only from profile-filtered candidates.
"""
from __future__ import annotations
import json
import random
import sqlite3
from pathlib import Path
from typing import Optional
import pytest
from starlette.testclient import TestClient
import circuit_breaker
import dispatcher
from config import RoutingProfile
from dispatcher import (
BUILTIN_PROFILES,
TaskRequest,
app,
persist_route_decision,
route,
)
from routing import rejection_reason, select_candidates
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
ADMIN_SCHEMA_SQL = (ROOT / "config" / "admin_schema.sql").read_text()
M1 = "m1-cloud"
M1_FLEX = "m1-cloud-flex"
M2 = "m2-cloud"
LOCAL = "m3-local"
LOCAL2 = "m3b-local"
def _model_row(
model_id: str,
*,
provider: str = "neuralwatt",
tier: int = 2,
latency_class: str = "standard",
completion_price: float = 1.0,
prompt_price: Optional[float] = None,
context_window: int = 262128,
effective_context_window: int = 192500,
supports_vision: int = 0,
supports_json_mode: int = 1,
eligible_categories: Optional[str] = None,
) -> dict:
"""A models-table row, shaped for the fixture's INSERT."""
return {
"model_id": model_id,
"provider": provider,
"base_model_id": model_id,
"tier": tier,
"context_window": context_window,
"effective_context_window": effective_context_window,
"max_output_tokens": 16384,
"cost_per_1m_prompt": (
prompt_price if prompt_price is not None else completion_price / 3
),
"cost_per_1m_completion": completion_price,
"cost_per_1m_prompt_cached": (
prompt_price if prompt_price is not None else completion_price / 3
),
"supports_vision": supports_vision,
"supports_json_mode": supports_json_mode,
"latency_class": latency_class,
"reasoning_mode": "default",
"context_variant": "full",
"access_level": "public",
"availability": "active",
"eligible_categories": eligible_categories,
"last_updated": "2026-08-22T00:00:00+00:00",
}
def _insert_model(conn: sqlite3.Connection, row: dict) -> None:
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, tier, context_window,
effective_context_window, max_output_tokens,
cost_per_1m_prompt, cost_per_1m_completion, cost_per_1m_prompt_cached,
supports_vision, supports_json_mode,
latency_class, reasoning_mode, context_variant,
access_level, availability, eligible_categories, last_updated
) VALUES (
:model_id, :provider, :base_model_id, :tier, :context_window,
:effective_context_window, :max_output_tokens,
:cost_per_1m_prompt, :cost_per_1m_completion, :cost_per_1m_prompt_cached,
:supports_vision, :supports_json_mode,
:latency_class, :reasoning_mode, :context_variant,
:access_level, :availability, :eligible_categories, :last_updated
)
""",
row,
)
def _insert_proficiency(
conn: sqlite3.Connection,
model_id: str,
provider: str,
*,
blended_score: float,
outcome_samples: int = 0,
) -> None:
conn.execute(
"""
INSERT INTO proficiency (
model_id, provider, category, blended_score, source,
outcome_samples, last_updated
) VALUES (?, ?, 'coding_general', ?, 'leaderboard', ?,
'2026-08-22T00:00:00+00:00')
""",
(model_id, provider, blended_score, outcome_samples),
)
class FakeResponse:
"""Just enough of requests.Response for the stubbed provider POST."""
def __init__(self, payload=None):
self.status_code = 200
self._payload = payload or {}
self.text = json.dumps(self._payload)
self.headers = {}
self.closed = False
def json(self):
return self._payload
def iter_lines(self, decode_unicode=False):
return iter([])
def close(self):
self.closed = True
@pytest.fixture
def profile_router(tmp_path, monkeypatch):
"""The dispatcher pointed at a curated throwaway catalog, nothing dialled out.
Candidate economics are chosen so the routing outcome is fully determined:
- M1 / M1_FLEX tie on proficiency (0.9); the flex twin is cheaper, so under
batch latency it outranks standard, and under interactive it is filtered
out entirely by the latency_class hard filter.
- In exploration tests M2 has zero outcome samples, so an epsilon=1.0 run
with the default profile picks it (cloud), while the locality profile's
candidate set contains only the two ollama-local rows.
"""
db_path = tmp_path / "profiles.db"
conn = sqlite3.connect(db_path)
conn.executescript(SCHEMA_SQL)
conn.executescript(ADMIN_SCHEMA_SQL)
_insert_model(conn, _model_row(M1, completion_price=3.0))
_insert_model(
conn, _model_row(M1_FLEX, completion_price=2.0, latency_class="flex")
)
_insert_model(conn, _model_row(M2, completion_price=9.0))
_insert_model(
conn,
_model_row(
LOCAL, provider="ollama-local", completion_price=5.0, prompt_price=1.0
),
)
_insert_model(
conn,
_model_row(
LOCAL2, provider="ollama-local", completion_price=1.0, prompt_price=0.33
),
)
_insert_proficiency(conn, M1, "neuralwatt", blended_score=0.90, outcome_samples=50)
_insert_proficiency(conn, M1_FLEX, "neuralwatt", blended_score=0.90, outcome_samples=50)
_insert_proficiency(conn, M2, "neuralwatt", blended_score=0.85, outcome_samples=0)
_insert_proficiency(conn, LOCAL, "ollama-local", blended_score=0.88, outcome_samples=50)
# 0.70 sits a full quality-tolerance band below the 0.88/0.90 rows
# (0.9 - 0.8 would land there only by float luck: 0.099999... rounds to
# band 0). Band 1 keeps LOCAL2 eligible-but-ranked-last under the default
# profile while still being the cheapest local row under locality.
_insert_proficiency(conn, LOCAL2, "ollama-local", blended_score=0.70, outcome_samples=50)
conn.commit()
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(db_path))
monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False)
monkeypatch.setattr(dispatcher.cfg.local_vision, "enabled", False)
monkeypatch.setattr(dispatcher.cfg.session_cache, "enabled", False)
# Exploration is pinned off for determinism; the exploration tests
# re-enable it explicitly.
monkeypatch.setattr(dispatcher.cfg.exploration, "enabled", False)
monkeypatch.setattr(dispatcher.cfg.exploration, "epsilon", 0.0)
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
calls = []
def fake_post(url, headers=None, json=None, stream=False, timeout=None):
calls.append({"url": url, "body": json, "stream": stream})
return FakeResponse(
{
"id": "chatcmpl-test-1",
"model": json["model"],
"choices": [
{"message": {"role": "assistant", "content": "ok"},
"finish_reason": "stop"}
],
"usage": {"prompt_tokens": 31, "completion_tokens": 12},
"energy": {"energy_kwh": 5e-05},
"cost": {"request_cost_usd": 4e-04},
}
)
monkeypatch.setattr(dispatcher.requests, "post", fake_post)
monkeypatch.setattr(
dispatcher, "classify",
lambda task, context: dispatcher.Classification(
task_category="coding_general", task_tier=2,
required_context_tokens=100, confidence=0.9,
),
)
yield TestClient(app), db_path, calls
circuit_breaker.clear()
@pytest.fixture
def logbuf():
"""Capture what would reach the journal, through the real formatter."""
import io
import logs
buf = io.StringIO()
logs.configure("debug", stream=buf, journald=False)
yield buf
for handler in list(logs.log.handlers):
logs.log.removeHandler(handler)
def _route_req(
*,
latency_tolerance: Optional[str] = None,
task_category: str = "coding_general",
):
return TaskRequest(
task="write me a function",
task_category=task_category,
task_tier=2,
required_context_tokens=100,
latency_tolerance=latency_tolerance,
)
def _hard_filters(exclude_models: set[str] = frozenset()) -> dict:
"""The hard-filter kwargs every direct rejection_reason/select_candidates
call in this file shares, matching what route() itself passes."""
return dict(
required_context_tokens=100,
required_tier=2,
latency_tolerance="interactive",
allowed_access_levels=["public"],
exclude_stale=True,
exclude_deprecated=True,
exclude_models=exclude_models,
)
# =============================================================================
# Plank (a): auto and auto:batch behavior is byte-identical to the old ternary.
# =============================================================================
def test_auto_batch_matches_default_with_batch_latency(profile_router):
"""route() under the batch profile must equal the default profile plus
request latency_tolerance=batch, byte for byte.
The batch profile's only field is latency_tolerance, and a request-level
batch tolerance on the default profile reaches the same effective latency;
both must produce the same selection, runner-up ordering, and candidate
count. A leftover special-case ternary that treated ``auto:batch``
differently from an explicit batch request would break this.
"""
_, _, _ = profile_router
via_request = route(
_route_req(latency_tolerance="batch"),
profile="default",
profile_obj=BUILTIN_PROFILES["default"],
)
via_profile = route(
_route_req(latency_tolerance="batch"),
profile="batch",
profile_obj=BUILTIN_PROFILES["batch"],
)
assert via_request.selected is not None
assert via_profile.selected is not None
assert via_request.selected.model_id == via_profile.selected.model_id
assert [r.model_id for r in via_request.runners_up] == [
r.model_id for r in via_profile.runners_up
]
assert via_request.candidates_considered == via_profile.candidates_considered
assert via_request.latency_tolerance == via_profile.latency_tolerance == "batch"
# The flex twin is cheaper at equal proficiency, so batch must admit it and
# rank it first: this is what makes the batch equivalence observable.
assert via_profile.selected.model_id == M1_FLEX
assert via_profile.selected.latency_class == "flex"
def test_interactive_default_excludes_flex_rows_batch_admits_them(profile_router):
"""Without batch tolerance, the flex twin is latency-filtered out; the
selection therefore differs, proving the flex scenario is live."""
_, _, _ = profile_router
interactive = route(
_route_req(), profile="default", profile_obj=BUILTIN_PROFILES["default"]
)
batch = route(
_route_req(latency_tolerance="batch"),
profile="batch",
profile_obj=BUILTIN_PROFILES["batch"],
)
assert interactive.selected is not None
assert batch.selected is not None
assert interactive.selected.model_id == M1
assert interactive.selected.latency_class == "standard"
assert batch.selected.model_id == M1_FLEX
assert batch.selected.latency_class == "flex"
# The flex twin only enters the candidate set under batch tolerance.
assert interactive.candidates_considered < batch.candidates_considered
def test_chat_auto_batch_selects_same_model_as_auto_under_batch_default(
profile_router, monkeypatch
):
"""``POST /v1/chat/completions`` with model=auto:batch returns the same
selected model as model=auto when the configured default profile is batch.
Since Wave A config profiles may not shadow built-in names, the test sets
``routing.default_profile`` to a new config profile named ``batch-local``
whose only override is ``latency_tolerance=batch``. Both ``auto`` and
``auto:batch`` must then pick the flex twin through the profile layer.
"""
client, _, _ = profile_router
monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch-local")
monkeypatch.setitem(
dispatcher.cfg.profiles, "batch-local", RoutingProfile(latency_tolerance="batch")
)
auto = client.post(
"/v1/chat/completions",
json={"model": "auto", "messages": [{"role": "user", "content": "hi"}]},
)
auto_batch = client.post(
"/v1/chat/completions",
json={"model": "auto:batch", "messages": [{"role": "user", "content": "hi"}]},
)
assert auto.status_code == 200
assert auto_batch.status_code == 200
assert auto.headers["X-Router-Model"] == M1_FLEX
assert auto_batch.headers["X-Router-Model"] == M1_FLEX
assert auto.json()["model"] == auto_batch.json()["model"] == M1_FLEX
# =============================================================================
# Plank (b): Unknown profile returns 422 naming valid profiles.
# =============================================================================
def test_route_unknown_profile_returns_422_naming_valid_profiles(profile_router):
"""A misspelled bare profile on /route must name the valid choices."""
client, _, _ = profile_router
resp = client.post("/route", json={"task": "x", "profile": "nosuchprofile"})
assert resp.status_code == 422
detail = resp.json()["detail"]
assert "nosuchprofile" in detail
assert "default" in detail
assert "batch" in detail
assert "locality" in detail
# ``/v1/chat/completions`` unknown-profile coverage is NOT duplicated here.
# See tests/test_chat_completions.py::test_unknown_profile_returns_422_naming_valid_profiles
# for the auto:<unknown> 422 naming default/batch, added in Todos 3/4.
# =============================================================================
# Plank (c): Empty profile allowlist fails with a profile-specific reason.
# =============================================================================
def test_ghost_allowlist_selects_nothing_and_persists_profile_excluded(profile_router):
"""A profile whose allowed_model_ids intersect no catalog row selects
nothing, debug-logs ``profile_excluded`` for every candidate, and can
persist that reason through the persist_route_decision seam."""
_, db_path, _ = profile_router
profile = RoutingProfile(allowed_model_ids={"nonexistent-model"})
decision = route(_route_req(), profile="ghost", profile_obj=profile)
assert decision.selected is None
assert decision.candidates_considered == 0
assert decision.profile == "ghost"
# The allowlist resolves to an empty restrict_to set: nothing intersects.
conn = sqlite3.connect(db_path)
conn.row_factory = sqlite3.Row
rows = dispatcher.load_candidates(conn, "coding_general")
admin_deprecated = dispatcher._admin_excluded_models(conn)
conn.close()
restrict = dispatcher._restrict_to_from_profile(profile, rows)
assert restrict == set()
assert admin_deprecated == set()
# Every catalog row carries a profile-era rejection reason. The flex twin
# is caught one filter earlier (latency precedes the profile check in the
# documented rejection_reason ordering); everything else lands on the
# profile-specific reason.
for row in rows:
reason = rejection_reason(
row, restrict_to=restrict, **_hard_filters()
)
if row["model_id"] == M1_FLEX:
assert reason == "latency_class(flex)"
else:
assert reason == "profile_excluded"
# And that reason, persisted through the production write path, lands on
# the route_decisions row for the no-candidate decision.
persist_route_decision(
"route", classification=decision, rejected_reason="profile_excluded"
)
conn = sqlite3.connect(db_path)
conn.row_factory = sqlite3.Row
row = conn.execute(
"SELECT rejected_reason, selected_model, profile FROM route_decisions "
"ORDER BY id DESC LIMIT 1"
).fetchone()
conn.close()
assert row is not None
assert row["selected_model"] is None
assert row["profile"] == "ghost"
assert row["rejected_reason"] is not None
assert row["rejected_reason"].startswith("profile_excluded")
def test_debug_filter_lines_exist_for_profile_exclusions(profile_router, logbuf):
"""Every catalog row dropped by the allowlist gets a ``filter`` line.
Known limitation, pinned as-is: route()'s debug loop calls
``rejection_reason(row, **filters)`` WITHOUT ``restrict_to``, so the
logged reason for a profile-excluded row is the placeholder ``-`` rather
than ``profile_excluded``. The authoritative reason (via
``rejection_reason(..., restrict_to=...)``) is pinned in the ghost
test above; this test pins that the dropped rows at least surface in
the debug log at all. A fix that threads restrict_to into the loop
should update the ``== "-"`` expectations to ``== "profile_excluded"``.
"""
_, _, _ = profile_router
route(
_route_req(),
profile="ghost",
profile_obj=RoutingProfile(allowed_model_ids={"nonexistent-model"}),
)
reasons: dict[str, str] = {}
for line in logbuf.getvalue().splitlines():
if line.startswith("filter "):
fields = dict(
token.split("=", 1) for token in line.split(" ")[1:] if "=" in token
)
reasons[fields.get("model", "")] = fields.get("reason", "")
# Nothing survived, so every catalog row produced a filter line.
for model_id in (M1, M1_FLEX, M2, LOCAL, LOCAL2):
assert model_id in reasons, model_id
# The flex row is dropped by the latency filter (reason visible), the
# rest by the allowlist (reason currently '-').
assert reasons[M1_FLEX] == "latency_class(flex)"
for model_id in (M1, M2, LOCAL, LOCAL2):
assert reasons[model_id] == "-"
def test_restriction_set_passes_rows_inside_the_allowlist(profile_router):
"""The same restrict_to set keeps rows it names — the set is precise, not
an unconditional kill switch."""
_, db_path, _ = profile_router
profile = RoutingProfile(allowed_model_ids={M1, M2})
conn = sqlite3.connect(db_path)
conn.row_factory = sqlite3.Row
rows = dispatcher.load_candidates(conn, "coding_general")
conn.close()
restrict = dispatcher._restrict_to_from_profile(profile, rows)
assert restrict == {M1, M2}
survivors = select_candidates(rows, restrict_to=restrict, **_hard_filters())
assert {r["model_id"] for r in survivors} == {M1, M2}
# =============================================================================
# Plank (d): Profiles AND with circuit-breaker, admin overrides,
# and eligible_categories.
# =============================================================================
def test_profile_excludes_circuit_open_model(profile_router, monkeypatch):
"""A profile that allows only m1 must still drop it when the circuit
breaker has it open, and the rejection reason is the exclusion reason."""
_, _, _ = profile_router
monkeypatch.setattr(dispatcher.cfg.circuit_breaker, "enabled", True)
circuit_breaker.record_failure(
M1,
"neuralwatt",
initial_cooldown=dispatcher.cfg.circuit_breaker.initial_cooldown_seconds,
max_cooldown=dispatcher.cfg.circuit_breaker.max_cooldown_seconds,
backoff_multiplier=dispatcher.cfg.circuit_breaker.backoff_multiplier,
)
profile = RoutingProfile(allowed_model_ids={M1})
decision = route(_route_req(), profile="m1-only", profile_obj=profile)
assert decision.selected is None
assert decision.candidates_considered == 0
# The composition is visible at the primitive level too: the exclude_models
# set (circuit-open rows) drops m1 even though the profile admits it.
row = _model_row(M1)
assert (
rejection_reason(row, restrict_to={M1}, **_hard_filters(exclude_models={M1}))
== "circuit_open"
)
def test_profile_excludes_admin_deprecated_model(profile_router):
"""A profile that allows only m1 must drop it when an admin override
deprecates it; the survivor set is empty."""
_, db_path, _ = profile_router
conn = sqlite3.connect(db_path)
conn.execute(
"INSERT INTO admin_model_overrides "
"(model_id, provider, availability, reason, updated_at) "
"VALUES (?, 'neuralwatt', 'deprecated', 'test', "
"'2026-08-22T00:00:00+00:00')",
(M1,),
)
conn.commit()
conn.close()
profile = RoutingProfile(allowed_model_ids={M1})
decision = route(_route_req(), profile="m1-only", profile_obj=profile)
assert decision.selected is None
assert decision.candidates_considered == 0
def test_profile_excludes_category_ineligible_model(profile_router):
"""A profile that allows only m1 must drop it when m1's
eligible_categories excludes the task's category."""
_, db_path, _ = profile_router
conn = sqlite3.connect(db_path)
conn.execute(
"UPDATE models SET eligible_categories = 'coding_refactor' "
"WHERE model_id = ?",
(M1,),
)
conn.commit()
conn.close()
profile = RoutingProfile(allowed_model_ids={M1})
req = TaskRequest(
task="summarize this",
task_category="summarization",
task_tier=2,
required_context_tokens=100,
)
decision = route(req, profile="m1-only", profile_obj=profile)
assert decision.selected is None
assert decision.candidates_considered == 0
# And the primitive composes: the category gate fires even though the
# profile allowlist names the row.
row = _model_row(M1, eligible_categories="coding_refactor")
assert (
rejection_reason(
row, task_category="summarization", restrict_to={M1}, **_hard_filters()
)
== "category_ineligible"
)
def test_profile_of_two_models_selects_the_survivor(profile_router):
"""When the profile allows both m1 and m2 and an admin override kills m1,
the other candidate still wins — the filters compose without wiping the
whole profile set."""
_, db_path, _ = profile_router
conn = sqlite3.connect(db_path)
conn.execute(
"INSERT INTO admin_model_overrides "
"(model_id, provider, availability, reason, updated_at) "
"VALUES (?, 'neuralwatt', 'deprecated', 'test', "
"'2026-08-22T00:00:00+00:00')",
(M1,),
)
conn.commit()
conn.close()
profile = RoutingProfile(allowed_model_ids={M1, M2})
decision = route(_route_req(), profile="m1-or-m2", profile_obj=profile)
assert decision.selected is not None
assert decision.selected.model_id == M2
assert decision.candidates_considered == 1
# =============================================================================
# Plank (e): Exploration picks only from profile-filtered candidates.
# =============================================================================
def _force_exploration(monkeypatch, *, seed: int = 0) -> None:
monkeypatch.setattr(dispatcher.cfg.exploration, "enabled", True)
monkeypatch.setattr(dispatcher.cfg.exploration, "epsilon", 1.0)
monkeypatch.setattr(dispatcher.cfg.exploration, "max_tier", 3)
# The local rows price above the cloud winner by more than the default
# cap of 4.0; widen it so the explore path is reachable at all.
monkeypatch.setattr(dispatcher.cfg.exploration, "max_cost_ratio", 1_000.0)
monkeypatch.setattr(dispatcher, "_router_rng", random.Random(seed))
def test_exploration_never_leaves_the_locality_profile(profile_router, monkeypatch):
"""With exploration forced on, the locality profile restricts candidates
to ollama-local rows, so every explored pick is local — exploration
cannot reach a cloud row it never sees."""
_, _, _ = profile_router
_force_exploration(monkeypatch)
profile = BUILTIN_PROFILES["locality"]
seen: set[tuple[str, str]] = set()
explored_any = False
for _ in range(30):
decision = route(_route_req(), profile="locality", profile_obj=profile)
assert decision.selected is not None
seen.add((decision.selected.model_id, decision.selected.provider))
explored_any = explored_any or decision.explored
assert seen, "the locality profile must route to something"
assert explored_any, "with two local candidates and epsilon=1 the explore path runs"
assert {model_id for model_id, _ in seen} <= {LOCAL, LOCAL2}
assert all(provider == "ollama-local" for _, provider in seen)
def test_exploration_default_profile_can_reach_a_cloud_row(
profile_router, monkeypatch
):
"""Positive control: same rng, same epsilon, default profile — exploration
does pick a cloud row (the zero-sample one). This is what makes the
locality assertion above meaningful rather than vacuous."""
_, _, _ = profile_router
_force_exploration(monkeypatch)
seen: set[tuple[str, str]] = set()
for _ in range(30):
decision = route(
_route_req(), profile="default", profile_obj=BUILTIN_PROFILES["default"]
)
assert decision.selected is not None
seen.add((decision.selected.model_id, decision.selected.provider))
assert any(provider == "neuralwatt" for _, provider in seen), seen
assert any(model_id == M2 for model_id, _ in seen), (
"the zero-sample cloud row is exactly what exploration targets"
)
def test_exploration_stays_inside_an_allowed_model_ids_profile(
profile_router, monkeypatch
):
"""An allowed_model_ids profile pins exploration to its allowlist."""
_, _, _ = profile_router
_force_exploration(monkeypatch)
allowlist = {M1, M2}
profile = RoutingProfile(allowed_model_ids=allowlist)
for _ in range(30):
decision = route(_route_req(), profile="m1-or-m2", profile_obj=profile)
assert decision.selected is not None
assert decision.selected.model_id in allowlist
# =============================================================================
# Route endpoint round-trip: profiles resolve and persist end to end.
# =============================================================================
def test_route_endpoint_batch_profile_round_trip(profile_router):
"""/route with profile=batch stores profile and batch latency on the row."""
client, db_path, _ = profile_router
resp = client.post(
"/route", json={"task": "write me a function", "profile": "batch"}
)
assert resp.status_code == 200
data = resp.json()
assert data["profile"] == "batch"
assert data["latency_tolerance"] == "batch"
assert data["selected"]["model_id"] == M1_FLEX
conn = sqlite3.connect(db_path)
conn.row_factory = sqlite3.Row
row = conn.execute(
"SELECT profile, latency_tolerance, selected_model "
"FROM route_decisions ORDER BY id DESC LIMIT 1"
).fetchone()
conn.close()
assert row is not None
assert row["profile"] == "batch"
assert row["latency_tolerance"] == "batch"
assert row["selected_model"] == M1_FLEX
def test_bare_auto_resolves_to_configured_default_profile(profile_router, monkeypatch):
"""A bare ``auto`` model fills from ``routing.default_profile``, not a hard-coded "default".
Monkeypatching ``cfg.routing.default_profile = "batch"`` causes a route
whose profile kwarg is the default to behave exactly like ``auto:batch``.
"""
_, _, _ = profile_router
monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch")
decision = route(_route_req())
assert decision.latency_tolerance == "batch"
assert decision.selected is not None
assert decision.selected.model_id == M1_FLEX
assert decision.selected.latency_class == "flex"
def test_the_route_endpoint_honours_the_configured_default_profile(
profile_router, monkeypatch
):
"""/route is the documented no-spend probe of "what would the router
pick". It resolved the literal built-in named "default" rather than
routing.default_profile, so it answered a different question than the
router answers -- and wrote the wrong name into route_decisions.
Found live: with default_profile = "MiMo Test", /route logged
profile="default" and picked deepseek-v4-flash, while the same task
through ``auto`` logged "MiMo Test" and picked xiaomi/mimo-v2.5. Read
back from the decisions history, that looks like a broken profile
overlay rather than a broken probe.
"""
client, _, _ = profile_router
monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch")
resp = client.post("/route", json={"task": "write me a function"})
assert resp.status_code == 200
body = resp.json()
assert body["profile"] == "batch"
assert body["latency_tolerance"] == "batch"
assert body["selected"]["model_id"] == M1_FLEX
def test_an_explicit_profile_still_wins_over_the_configured_default(
profile_router, monkeypatch
):
"""The narrowing stays narrow: a caller naming a profile is deliberately
probing that one, including the built-in called "default"."""
client, _, _ = profile_router
monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch")
resp = client.post(
"/route", json={"task": "write me a function", "profile": "default"}
)
assert resp.status_code == 200
assert resp.json()["profile"] == "default"