"""Integration pins for the five preserved behaviors of named routing profiles. This file covers Todo 7 of the named-routing-profiles plan. These tests cut across the dispatcher / routing boundary: `route` is invoked directly with a real routing profile, against a monkeypatched SQLite catalog in a temp file. No classifier, provider, or local model is ever called. The five planks: (a) ``auto`` and ``auto:batch`` behavior is byte-identical to the old ternary: route() with the batch profile must return the same selection and runner-up ordering as the default profile with request ``latency_tolerance=batch``. (b) Unknown profiles return 422 naming valid profiles. Coverage for ``/v1/chat/completions`` unknown profiles exists in tests/test_chat_completions.py; we do not duplicate it here. (c) Empty profile allowlist fails with a profile-specific reason ("profile_excluded") and the reason is persisted to route_decisions when no candidate survives. (d) Profiles AND with circuit-breaker exclusions, admin overrides, and ``eligible_categories``. (e) Exploration picks only from profile-filtered candidates. """ from __future__ import annotations import json import random import sqlite3 from pathlib import Path from typing import Optional import pytest from starlette.testclient import TestClient import circuit_breaker import dispatcher from config import RoutingProfile from dispatcher import ( BUILTIN_PROFILES, TaskRequest, app, persist_route_decision, route, ) from routing import rejection_reason, select_candidates ROOT = Path(__file__).resolve().parent.parent SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text() ADMIN_SCHEMA_SQL = (ROOT / "config" / "admin_schema.sql").read_text() M1 = "m1-cloud" M1_FLEX = "m1-cloud-flex" M2 = "m2-cloud" LOCAL = "m3-local" LOCAL2 = "m3b-local" def _model_row( model_id: str, *, provider: str = "neuralwatt", tier: int = 2, latency_class: str = "standard", completion_price: float = 1.0, prompt_price: Optional[float] = None, context_window: int = 262128, effective_context_window: int = 192500, supports_vision: int = 0, supports_json_mode: int = 1, eligible_categories: Optional[str] = None, ) -> dict: """A models-table row, shaped for the fixture's INSERT.""" return { "model_id": model_id, "provider": provider, "base_model_id": model_id, "tier": tier, "context_window": context_window, "effective_context_window": effective_context_window, "max_output_tokens": 16384, "cost_per_1m_prompt": ( prompt_price if prompt_price is not None else completion_price / 3 ), "cost_per_1m_completion": completion_price, "cost_per_1m_prompt_cached": ( prompt_price if prompt_price is not None else completion_price / 3 ), "supports_vision": supports_vision, "supports_json_mode": supports_json_mode, "latency_class": latency_class, "reasoning_mode": "default", "context_variant": "full", "access_level": "public", "availability": "active", "eligible_categories": eligible_categories, "last_updated": "2026-08-22T00:00:00+00:00", } def _insert_model(conn: sqlite3.Connection, row: dict) -> None: conn.execute( """ INSERT INTO models ( model_id, provider, base_model_id, tier, context_window, effective_context_window, max_output_tokens, cost_per_1m_prompt, cost_per_1m_completion, cost_per_1m_prompt_cached, supports_vision, supports_json_mode, latency_class, reasoning_mode, context_variant, access_level, availability, eligible_categories, last_updated ) VALUES ( :model_id, :provider, :base_model_id, :tier, :context_window, :effective_context_window, :max_output_tokens, :cost_per_1m_prompt, :cost_per_1m_completion, :cost_per_1m_prompt_cached, :supports_vision, :supports_json_mode, :latency_class, :reasoning_mode, :context_variant, :access_level, :availability, :eligible_categories, :last_updated ) """, row, ) def _insert_proficiency( conn: sqlite3.Connection, model_id: str, provider: str, *, blended_score: float, outcome_samples: int = 0, ) -> None: conn.execute( """ INSERT INTO proficiency ( model_id, provider, category, blended_score, source, outcome_samples, last_updated ) VALUES (?, ?, 'coding_general', ?, 'leaderboard', ?, '2026-08-22T00:00:00+00:00') """, (model_id, provider, blended_score, outcome_samples), ) class FakeResponse: """Just enough of requests.Response for the stubbed provider POST.""" def __init__(self, payload=None): self.status_code = 200 self._payload = payload or {} self.text = json.dumps(self._payload) self.headers = {} self.closed = False def json(self): return self._payload def iter_lines(self, decode_unicode=False): return iter([]) def close(self): self.closed = True @pytest.fixture def profile_router(tmp_path, monkeypatch): """The dispatcher pointed at a curated throwaway catalog, nothing dialled out. Candidate economics are chosen so the routing outcome is fully determined: - M1 / M1_FLEX tie on proficiency (0.9); the flex twin is cheaper, so under batch latency it outranks standard, and under interactive it is filtered out entirely by the latency_class hard filter. - In exploration tests M2 has zero outcome samples, so an epsilon=1.0 run with the default profile picks it (cloud), while the locality profile's candidate set contains only the two ollama-local rows. """ db_path = tmp_path / "profiles.db" conn = sqlite3.connect(db_path) conn.executescript(SCHEMA_SQL) conn.executescript(ADMIN_SCHEMA_SQL) _insert_model(conn, _model_row(M1, completion_price=3.0)) _insert_model( conn, _model_row(M1_FLEX, completion_price=2.0, latency_class="flex") ) _insert_model(conn, _model_row(M2, completion_price=9.0)) _insert_model( conn, _model_row( LOCAL, provider="ollama-local", completion_price=5.0, prompt_price=1.0 ), ) _insert_model( conn, _model_row( LOCAL2, provider="ollama-local", completion_price=1.0, prompt_price=0.33 ), ) _insert_proficiency(conn, M1, "neuralwatt", blended_score=0.90, outcome_samples=50) _insert_proficiency(conn, M1_FLEX, "neuralwatt", blended_score=0.90, outcome_samples=50) _insert_proficiency(conn, M2, "neuralwatt", blended_score=0.85, outcome_samples=0) _insert_proficiency(conn, LOCAL, "ollama-local", blended_score=0.88, outcome_samples=50) # 0.70 sits a full quality-tolerance band below the 0.88/0.90 rows # (0.9 - 0.8 would land there only by float luck: 0.099999... rounds to # band 0). Band 1 keeps LOCAL2 eligible-but-ranked-last under the default # profile while still being the cheapest local row under locality. _insert_proficiency(conn, LOCAL2, "ollama-local", blended_score=0.70, outcome_samples=50) conn.commit() conn.close() monkeypatch.setattr(dispatcher.cfg.database, "path", str(db_path)) monkeypatch.setattr(dispatcher.cfg.verification, "local_llm_enabled", False) monkeypatch.setattr(dispatcher.cfg.local_vision, "enabled", False) monkeypatch.setattr(dispatcher.cfg.session_cache, "enabled", False) # Exploration is pinned off for determinism; the exploration tests # re-enable it explicitly. monkeypatch.setattr(dispatcher.cfg.exploration, "enabled", False) monkeypatch.setattr(dispatcher.cfg.exploration, "epsilon", 0.0) monkeypatch.setenv("NEURALWATT_API_KEY", "test-key") calls = [] def fake_post(url, headers=None, json=None, stream=False, timeout=None): calls.append({"url": url, "body": json, "stream": stream}) return FakeResponse( { "id": "chatcmpl-test-1", "model": json["model"], "choices": [ {"message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"} ], "usage": {"prompt_tokens": 31, "completion_tokens": 12}, "energy": {"energy_kwh": 5e-05}, "cost": {"request_cost_usd": 4e-04}, } ) monkeypatch.setattr(dispatcher.requests, "post", fake_post) monkeypatch.setattr( dispatcher, "classify", lambda task, context: dispatcher.Classification( task_category="coding_general", task_tier=2, required_context_tokens=100, confidence=0.9, ), ) yield TestClient(app), db_path, calls circuit_breaker.clear() @pytest.fixture def logbuf(): """Capture what would reach the journal, through the real formatter.""" import io import logs buf = io.StringIO() logs.configure("debug", stream=buf, journald=False) yield buf for handler in list(logs.log.handlers): logs.log.removeHandler(handler) def _route_req( *, latency_tolerance: Optional[str] = None, task_category: str = "coding_general", ): return TaskRequest( task="write me a function", task_category=task_category, task_tier=2, required_context_tokens=100, latency_tolerance=latency_tolerance, ) def _hard_filters(exclude_models: set[str] = frozenset()) -> dict: """The hard-filter kwargs every direct rejection_reason/select_candidates call in this file shares, matching what route() itself passes.""" return dict( required_context_tokens=100, required_tier=2, latency_tolerance="interactive", allowed_access_levels=["public"], exclude_stale=True, exclude_deprecated=True, exclude_models=exclude_models, ) # ============================================================================= # Plank (a): auto and auto:batch behavior is byte-identical to the old ternary. # ============================================================================= def test_auto_batch_matches_default_with_batch_latency(profile_router): """route() under the batch profile must equal the default profile plus request latency_tolerance=batch, byte for byte. The batch profile's only field is latency_tolerance, and a request-level batch tolerance on the default profile reaches the same effective latency; both must produce the same selection, runner-up ordering, and candidate count. A leftover special-case ternary that treated ``auto:batch`` differently from an explicit batch request would break this. """ _, _, _ = profile_router via_request = route( _route_req(latency_tolerance="batch"), profile="default", profile_obj=BUILTIN_PROFILES["default"], ) via_profile = route( _route_req(latency_tolerance="batch"), profile="batch", profile_obj=BUILTIN_PROFILES["batch"], ) assert via_request.selected is not None assert via_profile.selected is not None assert via_request.selected.model_id == via_profile.selected.model_id assert [r.model_id for r in via_request.runners_up] == [ r.model_id for r in via_profile.runners_up ] assert via_request.candidates_considered == via_profile.candidates_considered assert via_request.latency_tolerance == via_profile.latency_tolerance == "batch" # The flex twin is cheaper at equal proficiency, so batch must admit it and # rank it first: this is what makes the batch equivalence observable. assert via_profile.selected.model_id == M1_FLEX assert via_profile.selected.latency_class == "flex" def test_interactive_default_excludes_flex_rows_batch_admits_them(profile_router): """Without batch tolerance, the flex twin is latency-filtered out; the selection therefore differs, proving the flex scenario is live.""" _, _, _ = profile_router interactive = route( _route_req(), profile="default", profile_obj=BUILTIN_PROFILES["default"] ) batch = route( _route_req(latency_tolerance="batch"), profile="batch", profile_obj=BUILTIN_PROFILES["batch"], ) assert interactive.selected is not None assert batch.selected is not None assert interactive.selected.model_id == M1 assert interactive.selected.latency_class == "standard" assert batch.selected.model_id == M1_FLEX assert batch.selected.latency_class == "flex" # The flex twin only enters the candidate set under batch tolerance. assert interactive.candidates_considered < batch.candidates_considered def test_chat_auto_batch_selects_same_model_as_auto_under_batch_default( profile_router, monkeypatch ): """``POST /v1/chat/completions`` with model=auto:batch returns the same selected model as model=auto when the configured default profile is batch. Since Wave A config profiles may not shadow built-in names, the test sets ``routing.default_profile`` to a new config profile named ``batch-local`` whose only override is ``latency_tolerance=batch``. Both ``auto`` and ``auto:batch`` must then pick the flex twin through the profile layer. """ client, _, _ = profile_router monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch-local") monkeypatch.setitem( dispatcher.cfg.profiles, "batch-local", RoutingProfile(latency_tolerance="batch") ) auto = client.post( "/v1/chat/completions", json={"model": "auto", "messages": [{"role": "user", "content": "hi"}]}, ) auto_batch = client.post( "/v1/chat/completions", json={"model": "auto:batch", "messages": [{"role": "user", "content": "hi"}]}, ) assert auto.status_code == 200 assert auto_batch.status_code == 200 assert auto.headers["X-Router-Model"] == M1_FLEX assert auto_batch.headers["X-Router-Model"] == M1_FLEX assert auto.json()["model"] == auto_batch.json()["model"] == M1_FLEX # ============================================================================= # Plank (b): Unknown profile returns 422 naming valid profiles. # ============================================================================= def test_route_unknown_profile_returns_422_naming_valid_profiles(profile_router): """A misspelled bare profile on /route must name the valid choices.""" client, _, _ = profile_router resp = client.post("/route", json={"task": "x", "profile": "nosuchprofile"}) assert resp.status_code == 422 detail = resp.json()["detail"] assert "nosuchprofile" in detail assert "default" in detail assert "batch" in detail assert "locality" in detail # ``/v1/chat/completions`` unknown-profile coverage is NOT duplicated here. # See tests/test_chat_completions.py::test_unknown_profile_returns_422_naming_valid_profiles # for the auto: 422 naming default/batch, added in Todos 3/4. # ============================================================================= # Plank (c): Empty profile allowlist fails with a profile-specific reason. # ============================================================================= def test_ghost_allowlist_selects_nothing_and_persists_profile_excluded(profile_router): """A profile whose allowed_model_ids intersect no catalog row selects nothing, debug-logs ``profile_excluded`` for every candidate, and can persist that reason through the persist_route_decision seam.""" _, db_path, _ = profile_router profile = RoutingProfile(allowed_model_ids={"nonexistent-model"}) decision = route(_route_req(), profile="ghost", profile_obj=profile) assert decision.selected is None assert decision.candidates_considered == 0 assert decision.profile == "ghost" # The allowlist resolves to an empty restrict_to set: nothing intersects. conn = sqlite3.connect(db_path) conn.row_factory = sqlite3.Row rows = dispatcher.load_candidates(conn, "coding_general") admin_deprecated = dispatcher._admin_excluded_models(conn) conn.close() restrict = dispatcher._restrict_to_from_profile(profile, rows) assert restrict == set() assert admin_deprecated == set() # Every catalog row carries a profile-era rejection reason. The flex twin # is caught one filter earlier (latency precedes the profile check in the # documented rejection_reason ordering); everything else lands on the # profile-specific reason. for row in rows: reason = rejection_reason( row, restrict_to=restrict, **_hard_filters() ) if row["model_id"] == M1_FLEX: assert reason == "latency_class(flex)" else: assert reason == "profile_excluded" # And that reason, persisted through the production write path, lands on # the route_decisions row for the no-candidate decision. persist_route_decision( "route", classification=decision, rejected_reason="profile_excluded" ) conn = sqlite3.connect(db_path) conn.row_factory = sqlite3.Row row = conn.execute( "SELECT rejected_reason, selected_model, profile FROM route_decisions " "ORDER BY id DESC LIMIT 1" ).fetchone() conn.close() assert row is not None assert row["selected_model"] is None assert row["profile"] == "ghost" assert row["rejected_reason"] is not None assert row["rejected_reason"].startswith("profile_excluded") def test_debug_filter_lines_exist_for_profile_exclusions(profile_router, logbuf): """Every catalog row dropped by the allowlist gets a ``filter`` line. Known limitation, pinned as-is: route()'s debug loop calls ``rejection_reason(row, **filters)`` WITHOUT ``restrict_to``, so the logged reason for a profile-excluded row is the placeholder ``-`` rather than ``profile_excluded``. The authoritative reason (via ``rejection_reason(..., restrict_to=...)``) is pinned in the ghost test above; this test pins that the dropped rows at least surface in the debug log at all. A fix that threads restrict_to into the loop should update the ``== "-"`` expectations to ``== "profile_excluded"``. """ _, _, _ = profile_router route( _route_req(), profile="ghost", profile_obj=RoutingProfile(allowed_model_ids={"nonexistent-model"}), ) reasons: dict[str, str] = {} for line in logbuf.getvalue().splitlines(): if line.startswith("filter "): fields = dict( token.split("=", 1) for token in line.split(" ")[1:] if "=" in token ) reasons[fields.get("model", "")] = fields.get("reason", "") # Nothing survived, so every catalog row produced a filter line. for model_id in (M1, M1_FLEX, M2, LOCAL, LOCAL2): assert model_id in reasons, model_id # The flex row is dropped by the latency filter (reason visible), the # rest by the allowlist (reason currently '-'). assert reasons[M1_FLEX] == "latency_class(flex)" for model_id in (M1, M2, LOCAL, LOCAL2): assert reasons[model_id] == "-" def test_restriction_set_passes_rows_inside_the_allowlist(profile_router): """The same restrict_to set keeps rows it names — the set is precise, not an unconditional kill switch.""" _, db_path, _ = profile_router profile = RoutingProfile(allowed_model_ids={M1, M2}) conn = sqlite3.connect(db_path) conn.row_factory = sqlite3.Row rows = dispatcher.load_candidates(conn, "coding_general") conn.close() restrict = dispatcher._restrict_to_from_profile(profile, rows) assert restrict == {M1, M2} survivors = select_candidates(rows, restrict_to=restrict, **_hard_filters()) assert {r["model_id"] for r in survivors} == {M1, M2} # ============================================================================= # Plank (d): Profiles AND with circuit-breaker, admin overrides, # and eligible_categories. # ============================================================================= def test_profile_excludes_circuit_open_model(profile_router, monkeypatch): """A profile that allows only m1 must still drop it when the circuit breaker has it open, and the rejection reason is the exclusion reason.""" _, _, _ = profile_router monkeypatch.setattr(dispatcher.cfg.circuit_breaker, "enabled", True) circuit_breaker.record_failure( M1, "neuralwatt", initial_cooldown=dispatcher.cfg.circuit_breaker.initial_cooldown_seconds, max_cooldown=dispatcher.cfg.circuit_breaker.max_cooldown_seconds, backoff_multiplier=dispatcher.cfg.circuit_breaker.backoff_multiplier, ) profile = RoutingProfile(allowed_model_ids={M1}) decision = route(_route_req(), profile="m1-only", profile_obj=profile) assert decision.selected is None assert decision.candidates_considered == 0 # The composition is visible at the primitive level too: the exclude_models # set (circuit-open rows) drops m1 even though the profile admits it. row = _model_row(M1) assert ( rejection_reason(row, restrict_to={M1}, **_hard_filters(exclude_models={M1})) == "circuit_open" ) def test_profile_excludes_admin_deprecated_model(profile_router): """A profile that allows only m1 must drop it when an admin override deprecates it; the survivor set is empty.""" _, db_path, _ = profile_router conn = sqlite3.connect(db_path) conn.execute( "INSERT INTO admin_model_overrides " "(model_id, provider, availability, reason, updated_at) " "VALUES (?, 'neuralwatt', 'deprecated', 'test', " "'2026-08-22T00:00:00+00:00')", (M1,), ) conn.commit() conn.close() profile = RoutingProfile(allowed_model_ids={M1}) decision = route(_route_req(), profile="m1-only", profile_obj=profile) assert decision.selected is None assert decision.candidates_considered == 0 def test_profile_excludes_category_ineligible_model(profile_router): """A profile that allows only m1 must drop it when m1's eligible_categories excludes the task's category.""" _, db_path, _ = profile_router conn = sqlite3.connect(db_path) conn.execute( "UPDATE models SET eligible_categories = 'coding_refactor' " "WHERE model_id = ?", (M1,), ) conn.commit() conn.close() profile = RoutingProfile(allowed_model_ids={M1}) req = TaskRequest( task="summarize this", task_category="summarization", task_tier=2, required_context_tokens=100, ) decision = route(req, profile="m1-only", profile_obj=profile) assert decision.selected is None assert decision.candidates_considered == 0 # And the primitive composes: the category gate fires even though the # profile allowlist names the row. row = _model_row(M1, eligible_categories="coding_refactor") assert ( rejection_reason( row, task_category="summarization", restrict_to={M1}, **_hard_filters() ) == "category_ineligible" ) def test_profile_of_two_models_selects_the_survivor(profile_router): """When the profile allows both m1 and m2 and an admin override kills m1, the other candidate still wins — the filters compose without wiping the whole profile set.""" _, db_path, _ = profile_router conn = sqlite3.connect(db_path) conn.execute( "INSERT INTO admin_model_overrides " "(model_id, provider, availability, reason, updated_at) " "VALUES (?, 'neuralwatt', 'deprecated', 'test', " "'2026-08-22T00:00:00+00:00')", (M1,), ) conn.commit() conn.close() profile = RoutingProfile(allowed_model_ids={M1, M2}) decision = route(_route_req(), profile="m1-or-m2", profile_obj=profile) assert decision.selected is not None assert decision.selected.model_id == M2 assert decision.candidates_considered == 1 # ============================================================================= # Plank (e): Exploration picks only from profile-filtered candidates. # ============================================================================= def _force_exploration(monkeypatch, *, seed: int = 0) -> None: monkeypatch.setattr(dispatcher.cfg.exploration, "enabled", True) monkeypatch.setattr(dispatcher.cfg.exploration, "epsilon", 1.0) monkeypatch.setattr(dispatcher.cfg.exploration, "max_tier", 3) # The local rows price above the cloud winner by more than the default # cap of 4.0; widen it so the explore path is reachable at all. monkeypatch.setattr(dispatcher.cfg.exploration, "max_cost_ratio", 1_000.0) monkeypatch.setattr(dispatcher, "_router_rng", random.Random(seed)) def test_exploration_never_leaves_the_locality_profile(profile_router, monkeypatch): """With exploration forced on, the locality profile restricts candidates to ollama-local rows, so every explored pick is local — exploration cannot reach a cloud row it never sees.""" _, _, _ = profile_router _force_exploration(monkeypatch) profile = BUILTIN_PROFILES["locality"] seen: set[tuple[str, str]] = set() explored_any = False for _ in range(30): decision = route(_route_req(), profile="locality", profile_obj=profile) assert decision.selected is not None seen.add((decision.selected.model_id, decision.selected.provider)) explored_any = explored_any or decision.explored assert seen, "the locality profile must route to something" assert explored_any, "with two local candidates and epsilon=1 the explore path runs" assert {model_id for model_id, _ in seen} <= {LOCAL, LOCAL2} assert all(provider == "ollama-local" for _, provider in seen) def test_exploration_default_profile_can_reach_a_cloud_row( profile_router, monkeypatch ): """Positive control: same rng, same epsilon, default profile — exploration does pick a cloud row (the zero-sample one). This is what makes the locality assertion above meaningful rather than vacuous.""" _, _, _ = profile_router _force_exploration(monkeypatch) seen: set[tuple[str, str]] = set() for _ in range(30): decision = route( _route_req(), profile="default", profile_obj=BUILTIN_PROFILES["default"] ) assert decision.selected is not None seen.add((decision.selected.model_id, decision.selected.provider)) assert any(provider == "neuralwatt" for _, provider in seen), seen assert any(model_id == M2 for model_id, _ in seen), ( "the zero-sample cloud row is exactly what exploration targets" ) def test_exploration_stays_inside_an_allowed_model_ids_profile( profile_router, monkeypatch ): """An allowed_model_ids profile pins exploration to its allowlist.""" _, _, _ = profile_router _force_exploration(monkeypatch) allowlist = {M1, M2} profile = RoutingProfile(allowed_model_ids=allowlist) for _ in range(30): decision = route(_route_req(), profile="m1-or-m2", profile_obj=profile) assert decision.selected is not None assert decision.selected.model_id in allowlist # ============================================================================= # Route endpoint round-trip: profiles resolve and persist end to end. # ============================================================================= def test_route_endpoint_batch_profile_round_trip(profile_router): """/route with profile=batch stores profile and batch latency on the row.""" client, db_path, _ = profile_router resp = client.post( "/route", json={"task": "write me a function", "profile": "batch"} ) assert resp.status_code == 200 data = resp.json() assert data["profile"] == "batch" assert data["latency_tolerance"] == "batch" assert data["selected"]["model_id"] == M1_FLEX conn = sqlite3.connect(db_path) conn.row_factory = sqlite3.Row row = conn.execute( "SELECT profile, latency_tolerance, selected_model " "FROM route_decisions ORDER BY id DESC LIMIT 1" ).fetchone() conn.close() assert row is not None assert row["profile"] == "batch" assert row["latency_tolerance"] == "batch" assert row["selected_model"] == M1_FLEX def test_bare_auto_resolves_to_configured_default_profile(profile_router, monkeypatch): """A bare ``auto`` model fills from ``routing.default_profile``, not a hard-coded "default". Monkeypatching ``cfg.routing.default_profile = "batch"`` causes a route whose profile kwarg is the default to behave exactly like ``auto:batch``. """ _, _, _ = profile_router monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch") decision = route(_route_req()) assert decision.latency_tolerance == "batch" assert decision.selected is not None assert decision.selected.model_id == M1_FLEX assert decision.selected.latency_class == "flex" def test_the_route_endpoint_honours_the_configured_default_profile( profile_router, monkeypatch ): """/route is the documented no-spend probe of "what would the router pick". It resolved the literal built-in named "default" rather than routing.default_profile, so it answered a different question than the router answers -- and wrote the wrong name into route_decisions. Found live: with default_profile = "MiMo Test", /route logged profile="default" and picked deepseek-v4-flash, while the same task through ``auto`` logged "MiMo Test" and picked xiaomi/mimo-v2.5. Read back from the decisions history, that looks like a broken profile overlay rather than a broken probe. """ client, _, _ = profile_router monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch") resp = client.post("/route", json={"task": "write me a function"}) assert resp.status_code == 200 body = resp.json() assert body["profile"] == "batch" assert body["latency_tolerance"] == "batch" assert body["selected"]["model_id"] == M1_FLEX def test_an_explicit_profile_still_wins_over_the_configured_default( profile_router, monkeypatch ): """The narrowing stays narrow: a caller naming a profile is deliberately probing that one, including the built-in called "default".""" client, _, _ = profile_router monkeypatch.setattr(dispatcher.cfg.routing, "default_profile", "batch") resp = client.post( "/route", json={"task": "write me a function", "profile": "default"} ) assert resp.status_code == 200 assert resp.json()["profile"] == "default"