"""The classifier and the local verifier are separate endpoints. They used to be one: the verifier derived its URL by stripping ``/v1`` off ``classifier.base_url``. That silently coupled two unrelated decisions, and the coupling only becomes visible when the classifier is moved off-host — pointing classification at a cloud provider would have sent every local verification to ``/api/chat``, which does not exist. The verifier speaks Ollama's NATIVE API (``/api/chat`` with ``think=False``) because that is the only way to disable the reasoning trace, so it cannot follow the classifier anywhere the classifier can go. """ from __future__ import annotations import copy from pathlib import Path import pytest import yaml from config import RouterConfig ROOT = Path(__file__).resolve().parent.parent @pytest.fixture def raw() -> dict: with open(ROOT / "config" / "config.yaml") as fh: return yaml.safe_load(fh) def test_moving_the_classifier_to_a_cloud_provider_leaves_the_verifier_local(raw): cfg = copy.deepcopy(raw) cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1" cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY" cfg["classifier"]["model"] = "deepseek-v4-flash" loaded = RouterConfig(**cfg) assert loaded.classifier.base_url == "https://api.neuralwatt.com/v1" # The whole point: this did NOT follow the line above. assert "localhost" in loaded.verification.base_url def test_the_verifier_defaults_to_a_local_ollama(raw): cfg = copy.deepcopy(raw) cfg["verification"].pop("base_url", None) assert RouterConfig(**cfg).verification.base_url == "http://localhost:11434" def test_an_absent_api_key_env_means_unauthenticated(raw): # The local Ollama case. It ignores the key entirely, but the SDK requires # one to be set, so the dispatcher substitutes a placeholder. Asserted on # an explicit config rather than on whatever config.yaml currently says, # which is a deployment choice and not a property of the code. cfg = copy.deepcopy(raw) cfg["classifier"]["base_url"] = "http://localhost:11434/v1" cfg["classifier"].pop("api_key_env", None) assert RouterConfig(**cfg).classifier.api_key_env is None cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY" assert RouterConfig(**cfg).classifier.api_key_env == "NEURALWATT_API_KEY" def test_verification_model_may_be_unset_while_both_run_on_one_host(raw): cfg = copy.deepcopy(raw) cfg["classifier"]["base_url"] = "http://localhost:11434/v1" cfg["verification"]["model"] = None loaded = RouterConfig(**cfg) # Resolved by the dispatcher as `verification.model or classifier.model`, # which is right only because the hosts match. assert loaded.verification.model is None def test_an_unset_verifier_model_is_refused_once_the_hosts_differ(raw): # The failure this prevents is SILENT, which is why it is a load-time # error rather than a documented caveat. Observed directly: with the # classifier on NeuralWatt and this left null, the verifier POSTed # `deepseek-v4-flash` to localhost:11434, 404d, caught it, logged "local # verification unavailable" and recorded no sample. Verification looked # enabled while producing nothing. cfg = copy.deepcopy(raw) cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1" cfg["classifier"]["model"] = "deepseek-v4-flash" cfg["verification"]["model"] = None cfg["verification"]["local_llm_enabled"] = True with pytest.raises(ValueError, match="verification.model must be set"): RouterConfig(**cfg) def test_a_split_host_setup_loads_once_the_verifier_model_is_stated(raw): cfg = copy.deepcopy(raw) cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1" cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY" cfg["classifier"]["model"] = "deepseek-v4-flash" cfg["verification"]["model"] = "qwen3.5:latest" loaded = RouterConfig(**cfg) assert loaded.classifier.model == "deepseek-v4-flash" assert loaded.verification.model == "qwen3.5:latest" def test_disabling_the_local_check_lifts_the_requirement(raw): # A host with no local inference at all: nothing to name, nothing to guard. cfg = copy.deepcopy(raw) cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1" cfg["verification"]["model"] = None cfg["verification"]["local_llm_enabled"] = False loaded = RouterConfig(**cfg) assert loaded.verification.local_llm_enabled is False def test_a_host_with_no_local_inference_can_turn_the_local_check_off(raw): # An RPi has no usable local model. Structural verification is pure Python # and keeps running; only the LLM check goes away. cfg = copy.deepcopy(raw) cfg["verification"]["local_llm_enabled"] = False loaded = RouterConfig(**cfg) assert loaded.verification.local_llm_enabled is False # --- an unknown key is an error, not a no-op ------------------------------ def test_a_key_in_the_wrong_section_is_rejected(raw): # Exactly the mistake that shipped: max_input_chars belongs to the # classifier and was written into verification, where pydantic's default # extra="ignore" accepted it, dropped it, and left the code default in # force. It carried the same value, so nothing looked wrong -- but editing # it would have done nothing at all. cfg = copy.deepcopy(raw) cfg["verification"]["max_input_chars"] = 4000 with pytest.raises(ValueError, match="max_input_chars"): RouterConfig(**cfg) def test_a_misspelled_key_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["routing"]["min_tool_proficency"] = 0.5 # sic with pytest.raises(ValueError, match="min_tool_proficency"): RouterConfig(**cfg) def test_the_shipped_config_has_no_unknown_keys(raw): # Guards the whole file, not just the sections a test happens to name. RouterConfig(**raw) def test_the_tool_filter_can_be_turned_off_from_config(raw): # It is a knob to experiment with, so null must be a legal value rather # than something requiring a code change. cfg = copy.deepcopy(raw) cfg["routing"]["min_tool_proficiency"] = None assert RouterConfig(**cfg).routing.min_tool_proficiency is None def test_disabling_the_filter_lifts_the_category_name_check(raw): # With no filter there is nothing to join, so an unused category name # must not block startup. cfg = copy.deepcopy(raw) cfg["routing"]["min_tool_proficiency"] = None cfg["routing"]["tool_use_category"] = "not_a_real_category" assert RouterConfig(**cfg).routing.min_tool_proficiency is None # --- the CLI sanity check ------------------------------------------------- def test_the_config_summary_names_fields_that_exist(raw): """`python config.py` is the documented setup step, and it crashed. It printed cfg.weights, which had been replaced by cfg.objective, so the one command whose job is to prove the config is fine reported "Config loaded OK" and then died with AttributeError. A function plus this test means the names cannot rot silently again. """ from config import summary_lines lines = summary_lines(RouterConfig(**raw)) assert lines[0] == "Config loaded OK" body = "\n".join(lines[1:]) assert "quality_tolerance=" in body assert "classifier:" in body assert "dispatch providers:" in body assert "billing_reset_day=" in body # --- local_energy_call_sites log-once --------------------------------------- @pytest.mark.parametrize( ("site_key","attr","non_loopback","expected_sites"), [ ( "classify","classifier", "https://api.neuralwatt.com/v1", {"classify": False, "verify": True, "local_vision": True}, ), ( "verify","verification", "https://api.neuralwatt.com/v1", {"classify": True, "verify": False, "local_vision": True}, ), ( "local_vision","local_vision", "https://edge-ollama.example.com/v1", {"classify": True, "verify": True, "local_vision": False}, ), ], ids=["classifier","verification","local_vision"], ) def test_local_energy_call_sites_warns_once_per_branch( raw, caplog, site_key, attr, non_loopback, expected_sites, ): """Each non-loopback call-site warns exactly once at config load.""" cfg = copy.deepcopy(raw) cfg["local_energy"]["enabled"] = True cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0 cfg["local_energy"]["grid_intensity_g_per_kwh"] = 475.0 # Patch the specific section's base_url section = cfg[attr] section["base_url"] = non_loopback # For the verification branch the hosts-differ validator fires unless # we also set an explicit model name. if attr == "verification": cfg["classifier"]["model"] = "deepseek-v4-flash" cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY" with caplog.at_level("WARNING", logger="router.config"): loaded = RouterConfig(**cfg) # Touch the property multiple times after load to verify idempotent # logging (warning fires once, property is memoised). _ = loaded.local_energy_call_sites _ = loaded.local_energy_call_sites warning_records = [ r for r in caplog.records if r.levelname == "WARNING" and r.name == "router.config" and f"{site_key}.base_url points to a non-loopback host" in r.message ] assert len(warning_records) == 1 sites = loaded.local_energy_call_sites assert sites == expected_sites def test_an_in_process_encoder_is_metered_whatever_base_url_says(raw, caplog): """A local_encoder classifier places no HTTP call at all. The model runs IN-PROCESS on this host's own GPU, so classifier.base_url describes nothing about where that work happens -- and gating on it silently stopped metering a GPU whose electricity is on this machine's bill, while warning about a URL nothing calls. Not hypothetical: docs/local-models.md recommends pointing classifier.base_url at a VPN address rather than 0.0.0.0, which is exactly the configuration that used to switch encoder metering off. """ cfg = copy.deepcopy(raw) cfg["local_energy"]["enabled"] = True cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0 cfg["classifier"]["mode"] = "local_encoder" cfg["classifier"]["encoder"] = {} cfg["classifier"]["base_url"] = "http://ollama.vpn.example:11434/v1" with caplog.at_level("WARNING", logger="router.config"): loaded = RouterConfig(**cfg) assert loaded.local_energy_call_sites["classify"] is True assert not [ r for r in caplog.records if "classify.base_url points to a non-loopback host" in r.message ], "warning names a URL the encoder never calls" def test_a_remote_ollama_classifier_is_still_not_metered(raw, caplog): """The narrowing must stay narrow: in local_llm mode base_url IS the machine doing the work, and a remote one is genuinely unmeterable.""" cfg = copy.deepcopy(raw) cfg["local_energy"]["enabled"] = True cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0 cfg["classifier"]["mode"] = "local_llm" cfg["classifier"]["base_url"] = "http://ollama.vpn.example:11434/v1" loaded = RouterConfig(**cfg) assert loaded.local_energy_call_sites["classify"] is False def test_local_energy_call_sites_no_warning_when_disabled(raw, caplog): """When local_energy.enabled is false, no warning is emitted at all.""" cfg = copy.deepcopy(raw) cfg["local_energy"]["enabled"] = False with caplog.at_level("WARNING", logger="router.config"): loaded = RouterConfig(**cfg) _ = loaded.local_energy_call_sites _ = loaded.local_energy_call_sites warning_records = [r for r in caplog.records if r.levelname == "WARNING" and "local_energy" in r.message and r.name == "router.config"] assert len(warning_records) == 0 # Values are all False when disabled. assert loaded.local_energy_call_sites == {"classify": False, "verify": False, "local_vision": False} def test_billing_reset_day_accepts_1_through_28(): """Valid ranges 1..28 must be accepted.""" from config import Objective for day in (1, 6, 15, 28): obj = Objective(billing_reset_day=day) assert obj.billing_reset_day == day def test_billing_reset_day_accepts_none(): """null disables the feature.""" from config import Objective obj = Objective(billing_reset_day=None) assert obj.billing_reset_day is None def test_billing_reset_day_rejects_out_of_range(raw): """Values outside 1..28 are rejected.""" import copy cfg = copy.deepcopy(raw) cfg["objective"]["billing_reset_day"] = 0 with pytest.raises(ValueError, match="1\.\.28"): RouterConfig(**cfg) cfg["objective"]["billing_reset_day"] = 29 with pytest.raises(ValueError, match="1\.\.28"): RouterConfig(**cfg) cfg["objective"]["billing_reset_day"] = 32 with pytest.raises(ValueError, match="1\.\.28"): RouterConfig(**cfg) cfg["objective"]["billing_reset_day"] = -1 with pytest.raises(ValueError, match="1\.\.28"): RouterConfig(**cfg) def test_adoption_window_seconds_rejects_zero(): """0 is rejected by adoption_window_positive; null or absent means all time.""" from config import Objective with pytest.raises(ValueError, match="adoption_window_seconds"): Objective(adoption_window_seconds=0) def test_adoption_window_seconds_null_is_accepted(): """None is the documented all-time value.""" from config import Objective obj = Objective(adoption_window_seconds=None) assert obj.adoption_window_seconds is None def test_a_removed_key_is_rejected_rather_than_ignored(raw): """log_path named a file nothing ever wrote; leaving it valid would lie.""" import copy import pytest cfg = copy.deepcopy(raw) cfg["logging"]["log_path"] = "router.log" with pytest.raises(Exception, match="log_path|Extra inputs"): RouterConfig(**cfg) # --- capability gates and the local vision fallback ------------------------ def test_the_shipped_config_loads_with_the_new_keys(raw): loaded = RouterConfig(**raw) assert loaded.routing.require_vision is True assert loaded.routing.require_json_mode is True assert loaded.local_vision.model == "qwen3-vl-router:4b" assert loaded.local_vision.enabled is True def test_a_typo_in_require_vision_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["routing"]["require_visoin"] = True # sic with pytest.raises(ValueError, match="require_visoin"): RouterConfig(**cfg) def test_local_vision_can_be_disabled_from_config(raw): cfg = copy.deepcopy(raw) cfg["local_vision"]["enabled"] = False assert RouterConfig(**cfg).local_vision.enabled is False def test_local_vision_defaults_when_absent(raw): cfg = copy.deepcopy(raw) cfg.pop("local_vision") assert RouterConfig(**cfg).local_vision.enabled is True def test_nonpositive_local_timeout_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["local_vision"]["timeout_seconds"] = 0 with pytest.raises(ValueError, match="timeout_seconds"): RouterConfig(**cfg) def test_the_shipped_config_loads_the_pinch_section(raw): loaded = RouterConfig(**raw) assert loaded.pinch.enabled is True assert loaded.pinch.budget_tokens > 0 assert loaded.pinch.keep_last_turns > 0 def test_pinch_defaults_when_absent(raw): cfg = copy.deepcopy(raw) cfg.pop("pinch") loaded = RouterConfig(**cfg) assert loaded.pinch.enabled is True assert loaded.pinch.budget_tokens == 50000 assert loaded.pinch.keep_last_turns == 4 def test_nonpositive_pinch_budget_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["budget_tokens"] = 0 with pytest.raises(ValueError, match="budget_tokens"): RouterConfig(**cfg) def test_nonpositive_pinch_keep_last_turns_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["keep_last_turns"] = 0 with pytest.raises(ValueError, match="keep_last_turns"): RouterConfig(**cfg) def test_classifier_context_framing_loads(raw): # Defaults on (see config.yaml); the legacy layout is opt-out. assert RouterConfig(**raw).classifier.context_framing is True cfg = copy.deepcopy(raw) cfg["classifier"]["context_framing"] = False assert RouterConfig(**cfg).classifier.context_framing is False def test_nonminimum_pinch_max_summarize_chars_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["max_summarize_chars"] = 100 with pytest.raises(ValueError, match="max_summarize_chars.*>=.*3000"): RouterConfig(**cfg) def test_pinch_max_summarize_chars_at_valid_min_loads(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["max_summarize_chars"] = 3000 loaded = RouterConfig(**cfg) assert loaded.pinch.max_summarize_chars == 3000 def test_pinch_max_summarize_chars_accepts_default(raw): loaded = RouterConfig(**raw) assert loaded.pinch.max_summarize_chars == 4000 def test_nonminimum_pinch_protected_max_chars_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["protected_max_chars"] = 100 with pytest.raises(ValueError, match="protected_max_chars.*>=.*3000"): RouterConfig(**cfg) def test_pinch_protected_max_chars_at_valid_min_loads(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["protected_max_chars"] = 3000 loaded = RouterConfig(**cfg) assert loaded.pinch.protected_max_chars == 3000 def test_pinch_protected_max_chars_accepts_default(raw): loaded = RouterConfig(**raw) assert loaded.pinch.protected_max_chars == 20000 def test_pinch_protected_max_chars_null_disables(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["protected_max_chars"] = None loaded = RouterConfig(**cfg) assert loaded.pinch.protected_max_chars is None def test_pinch_relevance_defaults_load(raw): loaded = RouterConfig(**raw) assert loaded.pinch.relevance.enabled is True assert loaded.pinch.relevance.model == "nomic-embed-text" assert loaded.pinch.relevance.min_candidates == 2 def test_pinch_relevance_defaults_when_pinch_absent(raw): cfg = copy.deepcopy(raw) cfg.pop("pinch") loaded = RouterConfig(**cfg) assert loaded.pinch.relevance.enabled is True assert loaded.pinch.relevance.timeout_seconds == 10 def test_nonpositive_pinch_relevance_timeout_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["relevance"]["timeout_seconds"] = 0 with pytest.raises(ValueError, match="timeout_seconds"): RouterConfig(**cfg) def test_nonpositive_pinch_relevance_min_candidates_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["pinch"]["relevance"]["min_candidates"] = 0 with pytest.raises(ValueError, match="min_candidates"): RouterConfig(**cfg) def test_circuit_breaker_defaults_load(raw): loaded = RouterConfig(**raw) assert loaded.circuit_breaker.enabled is True assert loaded.circuit_breaker.initial_cooldown_seconds == 30 assert loaded.circuit_breaker.max_cooldown_seconds == 600 assert loaded.circuit_breaker.backoff_multiplier == 2.0 def test_nonpositive_circuit_breaker_initial_cooldown_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["circuit_breaker"]["initial_cooldown_seconds"] = 0 with pytest.raises(ValueError, match="initial_cooldown_seconds"): RouterConfig(**cfg) def test_circuit_breaker_max_below_initial_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["circuit_breaker"]["initial_cooldown_seconds"] = 300 cfg["circuit_breaker"]["max_cooldown_seconds"] = 200 with pytest.raises(ValueError, match="max_cooldown_seconds"): RouterConfig(**cfg) def test_circuit_breaker_multiplier_at_or_below_one_is_rejected(raw): cfg = copy.deepcopy(raw) cfg["circuit_breaker"]["backoff_multiplier"] = 1.0 with pytest.raises(ValueError, match="backoff_multiplier"): RouterConfig(**cfg) def test_the_shipped_config_points_classifier_and_verifier_at_the_router_tag(raw): # Context size is not a config-level knob: Ollama's OpenAI-compatible # endpoint (0.22.0, verified live) silently ignores num_ctx as a # per-request field, so it has to be baked into the Ollama model tag via # a Modelfile instead (see classifier.model's comment in config.yaml). # What config CAN still assert is that classifier and verification point # at the same tagged model, so they share one resident instance rather # than two differently-sized copies of the same base model. # # As of 2026-09-03 the SAME tag also serves local dispatch, so one model # covers classification, verification and dispatch and only one instance # stays resident. That co-residency is what lets the vision fallback stay # loaded too (23.2GB of 24GB together) -- see docs/local-models.md. loaded = RouterConfig(**raw) assert loaded.classifier.model == "qwen2.5-coder-router:14b" assert loaded.verification.model == loaded.classifier.model assert loaded.local_vision.model == "qwen3-vl-router:4b" # The consolidation invariant: dispatch runs on the classifier's tag. assert loaded.local_dispatch_models[0].model_id == loaded.classifier.model