Files
6krrt/tests/test_config_endpoints.py

582 lines
21 KiB
Python

"""The classifier and the local verifier are separate endpoints.
They used to be one: the verifier derived its URL by stripping ``/v1`` off
``classifier.base_url``. That silently coupled two unrelated decisions, and
the coupling only becomes visible when the classifier is moved off-host —
pointing classification at a cloud provider would have sent every local
verification to ``<provider>/api/chat``, which does not exist.
The verifier speaks Ollama's NATIVE API (``/api/chat`` with ``think=False``)
because that is the only way to disable the reasoning trace, so it cannot
follow the classifier anywhere the classifier can go.
"""
from __future__ import annotations
import copy
from pathlib import Path
import pytest
import yaml
from config import RouterConfig
ROOT = Path(__file__).resolve().parent.parent
@pytest.fixture
def raw() -> dict:
with open(ROOT / "config" / "config.yaml") as fh:
return yaml.safe_load(fh)
def test_moving_the_classifier_to_a_cloud_provider_leaves_the_verifier_local(raw):
cfg = copy.deepcopy(raw)
cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1"
cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY"
cfg["classifier"]["model"] = "deepseek-v4-flash"
loaded = RouterConfig(**cfg)
assert loaded.classifier.base_url == "https://api.neuralwatt.com/v1"
# The whole point: this did NOT follow the line above.
assert "localhost" in loaded.verification.base_url
def test_the_verifier_defaults_to_a_local_ollama(raw):
cfg = copy.deepcopy(raw)
cfg["verification"].pop("base_url", None)
assert RouterConfig(**cfg).verification.base_url == "http://localhost:11434"
def test_an_absent_api_key_env_means_unauthenticated(raw):
# The local Ollama case. It ignores the key entirely, but the SDK requires
# one to be set, so the dispatcher substitutes a placeholder. Asserted on
# an explicit config rather than on whatever config.yaml currently says,
# which is a deployment choice and not a property of the code.
cfg = copy.deepcopy(raw)
cfg["classifier"]["base_url"] = "http://localhost:11434/v1"
cfg["classifier"].pop("api_key_env", None)
assert RouterConfig(**cfg).classifier.api_key_env is None
cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY"
assert RouterConfig(**cfg).classifier.api_key_env == "NEURALWATT_API_KEY"
def test_verification_model_may_be_unset_while_both_run_on_one_host(raw):
cfg = copy.deepcopy(raw)
cfg["classifier"]["base_url"] = "http://localhost:11434/v1"
cfg["verification"]["model"] = None
loaded = RouterConfig(**cfg)
# Resolved by the dispatcher as `verification.model or classifier.model`,
# which is right only because the hosts match.
assert loaded.verification.model is None
def test_an_unset_verifier_model_is_refused_once_the_hosts_differ(raw):
# The failure this prevents is SILENT, which is why it is a load-time
# error rather than a documented caveat. Observed directly: with the
# classifier on NeuralWatt and this left null, the verifier POSTed
# `deepseek-v4-flash` to localhost:11434, 404d, caught it, logged "local
# verification unavailable" and recorded no sample. Verification looked
# enabled while producing nothing.
cfg = copy.deepcopy(raw)
cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1"
cfg["classifier"]["model"] = "deepseek-v4-flash"
cfg["verification"]["model"] = None
cfg["verification"]["local_llm_enabled"] = True
with pytest.raises(ValueError, match="verification.model must be set"):
RouterConfig(**cfg)
def test_a_split_host_setup_loads_once_the_verifier_model_is_stated(raw):
cfg = copy.deepcopy(raw)
cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1"
cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY"
cfg["classifier"]["model"] = "deepseek-v4-flash"
cfg["verification"]["model"] = "qwen3.5:latest"
loaded = RouterConfig(**cfg)
assert loaded.classifier.model == "deepseek-v4-flash"
assert loaded.verification.model == "qwen3.5:latest"
def test_disabling_the_local_check_lifts_the_requirement(raw):
# A host with no local inference at all: nothing to name, nothing to guard.
cfg = copy.deepcopy(raw)
cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1"
cfg["verification"]["model"] = None
cfg["verification"]["local_llm_enabled"] = False
loaded = RouterConfig(**cfg)
assert loaded.verification.local_llm_enabled is False
def test_a_host_with_no_local_inference_can_turn_the_local_check_off(raw):
# An RPi has no usable local model. Structural verification is pure Python
# and keeps running; only the LLM check goes away.
cfg = copy.deepcopy(raw)
cfg["verification"]["local_llm_enabled"] = False
loaded = RouterConfig(**cfg)
assert loaded.verification.local_llm_enabled is False
# --- an unknown key is an error, not a no-op ------------------------------
def test_a_key_in_the_wrong_section_is_rejected(raw):
# Exactly the mistake that shipped: max_input_chars belongs to the
# classifier and was written into verification, where pydantic's default
# extra="ignore" accepted it, dropped it, and left the code default in
# force. It carried the same value, so nothing looked wrong -- but editing
# it would have done nothing at all.
cfg = copy.deepcopy(raw)
cfg["verification"]["max_input_chars"] = 4000
with pytest.raises(ValueError, match="max_input_chars"):
RouterConfig(**cfg)
def test_a_misspelled_key_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["routing"]["min_tool_proficency"] = 0.5 # sic
with pytest.raises(ValueError, match="min_tool_proficency"):
RouterConfig(**cfg)
def test_the_shipped_config_has_no_unknown_keys(raw):
# Guards the whole file, not just the sections a test happens to name.
RouterConfig(**raw)
def test_the_tool_filter_can_be_turned_off_from_config(raw):
# It is a knob to experiment with, so null must be a legal value rather
# than something requiring a code change.
cfg = copy.deepcopy(raw)
cfg["routing"]["min_tool_proficiency"] = None
assert RouterConfig(**cfg).routing.min_tool_proficiency is None
def test_disabling_the_filter_lifts_the_category_name_check(raw):
# With no filter there is nothing to join, so an unused category name
# must not block startup.
cfg = copy.deepcopy(raw)
cfg["routing"]["min_tool_proficiency"] = None
cfg["routing"]["tool_use_category"] = "not_a_real_category"
assert RouterConfig(**cfg).routing.min_tool_proficiency is None
# --- the CLI sanity check -------------------------------------------------
def test_the_config_summary_names_fields_that_exist(raw):
"""`python config.py` is the documented setup step, and it crashed.
It printed cfg.weights, which had been replaced by cfg.objective, so the
one command whose job is to prove the config is fine reported "Config
loaded OK" and then died with AttributeError. A function plus this test
means the names cannot rot silently again.
"""
from config import summary_lines
lines = summary_lines(RouterConfig(**raw))
assert lines[0] == "Config loaded OK"
body = "\n".join(lines[1:])
assert "quality_tolerance=" in body
assert "classifier:" in body
assert "dispatch providers:" in body
assert "billing_reset_day=" in body
# --- local_energy_call_sites log-once ---------------------------------------
@pytest.mark.parametrize(
("site_key","attr","non_loopback","expected_sites"),
[
(
"classify","classifier",
"https://api.neuralwatt.com/v1",
{"classify": False, "verify": True, "local_vision": True},
),
(
"verify","verification",
"https://api.neuralwatt.com/v1",
{"classify": True, "verify": False, "local_vision": True},
),
(
"local_vision","local_vision",
"https://edge-ollama.example.com/v1",
{"classify": True, "verify": True, "local_vision": False},
),
],
ids=["classifier","verification","local_vision"],
)
def test_local_energy_call_sites_warns_once_per_branch(
raw, caplog, site_key, attr, non_loopback, expected_sites,
):
"""Each non-loopback call-site warns exactly once at config load."""
cfg = copy.deepcopy(raw)
cfg["local_energy"]["enabled"] = True
cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0
cfg["local_energy"]["grid_intensity_g_per_kwh"] = 475.0
# Patch the specific section's base_url
section = cfg[attr]
section["base_url"] = non_loopback
# For the verification branch the hosts-differ validator fires unless
# we also set an explicit model name.
if attr == "verification":
cfg["classifier"]["model"] = "deepseek-v4-flash"
cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY"
with caplog.at_level("WARNING", logger="router.config"):
loaded = RouterConfig(**cfg)
# Touch the property multiple times after load to verify idempotent
# logging (warning fires once, property is memoised).
_ = loaded.local_energy_call_sites
_ = loaded.local_energy_call_sites
warning_records = [
r for r in caplog.records
if r.levelname == "WARNING"
and r.name == "router.config"
and f"{site_key}.base_url points to a non-loopback host" in r.message
]
assert len(warning_records) == 1
sites = loaded.local_energy_call_sites
assert sites == expected_sites
def test_an_in_process_encoder_is_metered_whatever_base_url_says(raw, caplog):
"""A local_encoder classifier places no HTTP call at all.
The model runs IN-PROCESS on this host's own GPU, so classifier.base_url
describes nothing about where that work happens -- and gating on it
silently stopped metering a GPU whose electricity is on this machine's
bill, while warning about a URL nothing calls.
Not hypothetical: docs/local-models.md recommends pointing
classifier.base_url at a VPN address rather than 0.0.0.0, which is exactly
the configuration that used to switch encoder metering off.
"""
cfg = copy.deepcopy(raw)
cfg["local_energy"]["enabled"] = True
cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0
cfg["classifier"]["mode"] = "local_encoder"
cfg["classifier"]["encoder"] = {}
cfg["classifier"]["base_url"] = "http://ollama.vpn.example:11434/v1"
with caplog.at_level("WARNING", logger="router.config"):
loaded = RouterConfig(**cfg)
assert loaded.local_energy_call_sites["classify"] is True
assert not [
r for r in caplog.records
if "classify.base_url points to a non-loopback host" in r.message
], "warning names a URL the encoder never calls"
def test_a_remote_ollama_classifier_is_still_not_metered(raw, caplog):
"""The narrowing must stay narrow: in local_llm mode base_url IS the
machine doing the work, and a remote one is genuinely unmeterable."""
cfg = copy.deepcopy(raw)
cfg["local_energy"]["enabled"] = True
cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0
cfg["classifier"]["mode"] = "local_llm"
cfg["classifier"]["base_url"] = "http://ollama.vpn.example:11434/v1"
loaded = RouterConfig(**cfg)
assert loaded.local_energy_call_sites["classify"] is False
def test_local_energy_call_sites_no_warning_when_disabled(raw, caplog):
"""When local_energy.enabled is false, no warning is emitted at all."""
cfg = copy.deepcopy(raw)
cfg["local_energy"]["enabled"] = False
with caplog.at_level("WARNING", logger="router.config"):
loaded = RouterConfig(**cfg)
_ = loaded.local_energy_call_sites
_ = loaded.local_energy_call_sites
warning_records = [r for r in caplog.records if r.levelname == "WARNING"
and "local_energy" in r.message and r.name == "router.config"]
assert len(warning_records) == 0
# Values are all False when disabled.
assert loaded.local_energy_call_sites == {"classify": False, "verify": False, "local_vision": False}
def test_billing_reset_day_accepts_1_through_28():
"""Valid ranges 1..28 must be accepted."""
from config import Objective
for day in (1, 6, 15, 28):
obj = Objective(billing_reset_day=day)
assert obj.billing_reset_day == day
def test_billing_reset_day_accepts_none():
"""null disables the feature."""
from config import Objective
obj = Objective(billing_reset_day=None)
assert obj.billing_reset_day is None
def test_billing_reset_day_rejects_out_of_range(raw):
"""Values outside 1..28 are rejected."""
import copy
cfg = copy.deepcopy(raw)
cfg["objective"]["billing_reset_day"] = 0
with pytest.raises(ValueError, match="1\.\.28"):
RouterConfig(**cfg)
cfg["objective"]["billing_reset_day"] = 29
with pytest.raises(ValueError, match="1\.\.28"):
RouterConfig(**cfg)
cfg["objective"]["billing_reset_day"] = 32
with pytest.raises(ValueError, match="1\.\.28"):
RouterConfig(**cfg)
cfg["objective"]["billing_reset_day"] = -1
with pytest.raises(ValueError, match="1\.\.28"):
RouterConfig(**cfg)
def test_adoption_window_seconds_rejects_zero():
"""0 is rejected by adoption_window_positive; null or absent means all time."""
from config import Objective
with pytest.raises(ValueError, match="adoption_window_seconds"):
Objective(adoption_window_seconds=0)
def test_adoption_window_seconds_null_is_accepted():
"""None is the documented all-time value."""
from config import Objective
obj = Objective(adoption_window_seconds=None)
assert obj.adoption_window_seconds is None
def test_a_removed_key_is_rejected_rather_than_ignored(raw):
"""log_path named a file nothing ever wrote; leaving it valid would lie."""
import copy
import pytest
cfg = copy.deepcopy(raw)
cfg["logging"]["log_path"] = "router.log"
with pytest.raises(Exception, match="log_path|Extra inputs"):
RouterConfig(**cfg)
# --- capability gates and the local vision fallback ------------------------
def test_the_shipped_config_loads_with_the_new_keys(raw):
loaded = RouterConfig(**raw)
assert loaded.routing.require_vision is True
assert loaded.routing.require_json_mode is True
assert loaded.local_vision.model == "qwen3-vl-router:4b"
assert loaded.local_vision.enabled is True
def test_a_typo_in_require_vision_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["routing"]["require_visoin"] = True # sic
with pytest.raises(ValueError, match="require_visoin"):
RouterConfig(**cfg)
def test_local_vision_can_be_disabled_from_config(raw):
cfg = copy.deepcopy(raw)
cfg["local_vision"]["enabled"] = False
assert RouterConfig(**cfg).local_vision.enabled is False
def test_local_vision_defaults_when_absent(raw):
cfg = copy.deepcopy(raw)
cfg.pop("local_vision")
assert RouterConfig(**cfg).local_vision.enabled is True
def test_nonpositive_local_timeout_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["local_vision"]["timeout_seconds"] = 0
with pytest.raises(ValueError, match="timeout_seconds"):
RouterConfig(**cfg)
def test_the_shipped_config_loads_the_pinch_section(raw):
loaded = RouterConfig(**raw)
assert loaded.pinch.enabled is True
assert loaded.pinch.budget_tokens > 0
assert loaded.pinch.keep_last_turns > 0
def test_pinch_defaults_when_absent(raw):
cfg = copy.deepcopy(raw)
cfg.pop("pinch")
loaded = RouterConfig(**cfg)
assert loaded.pinch.enabled is True
assert loaded.pinch.budget_tokens == 50000
assert loaded.pinch.keep_last_turns == 4
def test_nonpositive_pinch_budget_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["budget_tokens"] = 0
with pytest.raises(ValueError, match="budget_tokens"):
RouterConfig(**cfg)
def test_nonpositive_pinch_keep_last_turns_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["keep_last_turns"] = 0
with pytest.raises(ValueError, match="keep_last_turns"):
RouterConfig(**cfg)
def test_classifier_context_framing_loads(raw):
# Defaults on (see config.yaml); the legacy layout is opt-out.
assert RouterConfig(**raw).classifier.context_framing is True
cfg = copy.deepcopy(raw)
cfg["classifier"]["context_framing"] = False
assert RouterConfig(**cfg).classifier.context_framing is False
def test_nonminimum_pinch_max_summarize_chars_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["max_summarize_chars"] = 100
with pytest.raises(ValueError, match="max_summarize_chars.*>=.*3000"):
RouterConfig(**cfg)
def test_pinch_max_summarize_chars_at_valid_min_loads(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["max_summarize_chars"] = 3000
loaded = RouterConfig(**cfg)
assert loaded.pinch.max_summarize_chars == 3000
def test_pinch_max_summarize_chars_accepts_default(raw):
loaded = RouterConfig(**raw)
assert loaded.pinch.max_summarize_chars == 4000
def test_nonminimum_pinch_protected_max_chars_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["protected_max_chars"] = 100
with pytest.raises(ValueError, match="protected_max_chars.*>=.*3000"):
RouterConfig(**cfg)
def test_pinch_protected_max_chars_at_valid_min_loads(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["protected_max_chars"] = 3000
loaded = RouterConfig(**cfg)
assert loaded.pinch.protected_max_chars == 3000
def test_pinch_protected_max_chars_accepts_default(raw):
loaded = RouterConfig(**raw)
assert loaded.pinch.protected_max_chars == 20000
def test_pinch_protected_max_chars_null_disables(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["protected_max_chars"] = None
loaded = RouterConfig(**cfg)
assert loaded.pinch.protected_max_chars is None
def test_pinch_relevance_defaults_load(raw):
loaded = RouterConfig(**raw)
assert loaded.pinch.relevance.enabled is True
assert loaded.pinch.relevance.model == "nomic-embed-text"
assert loaded.pinch.relevance.min_candidates == 2
def test_pinch_relevance_defaults_when_pinch_absent(raw):
cfg = copy.deepcopy(raw)
cfg.pop("pinch")
loaded = RouterConfig(**cfg)
assert loaded.pinch.relevance.enabled is True
assert loaded.pinch.relevance.timeout_seconds == 10
def test_nonpositive_pinch_relevance_timeout_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["relevance"]["timeout_seconds"] = 0
with pytest.raises(ValueError, match="timeout_seconds"):
RouterConfig(**cfg)
def test_nonpositive_pinch_relevance_min_candidates_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["relevance"]["min_candidates"] = 0
with pytest.raises(ValueError, match="min_candidates"):
RouterConfig(**cfg)
def test_circuit_breaker_defaults_load(raw):
loaded = RouterConfig(**raw)
assert loaded.circuit_breaker.enabled is True
assert loaded.circuit_breaker.initial_cooldown_seconds == 30
assert loaded.circuit_breaker.max_cooldown_seconds == 600
assert loaded.circuit_breaker.backoff_multiplier == 2.0
def test_nonpositive_circuit_breaker_initial_cooldown_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["circuit_breaker"]["initial_cooldown_seconds"] = 0
with pytest.raises(ValueError, match="initial_cooldown_seconds"):
RouterConfig(**cfg)
def test_circuit_breaker_max_below_initial_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["circuit_breaker"]["initial_cooldown_seconds"] = 300
cfg["circuit_breaker"]["max_cooldown_seconds"] = 200
with pytest.raises(ValueError, match="max_cooldown_seconds"):
RouterConfig(**cfg)
def test_circuit_breaker_multiplier_at_or_below_one_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["circuit_breaker"]["backoff_multiplier"] = 1.0
with pytest.raises(ValueError, match="backoff_multiplier"):
RouterConfig(**cfg)
def test_the_shipped_config_points_classifier_and_verifier_at_the_router_tag(raw):
# Context size is not a config-level knob: Ollama's OpenAI-compatible
# endpoint (0.22.0, verified live) silently ignores num_ctx as a
# per-request field, so it has to be baked into the Ollama model tag via
# a Modelfile instead (see classifier.model's comment in config.yaml).
# What config CAN still assert is that classifier and verification point
# at the same tagged model, so they share one resident instance rather
# than two differently-sized copies of the same base model.
#
# As of 2026-09-03 the SAME tag also serves local dispatch, so one model
# covers classification, verification and dispatch and only one instance
# stays resident. That co-residency is what lets the vision fallback stay
# loaded too (23.2GB of 24GB together) -- see docs/local-models.md.
loaded = RouterConfig(**raw)
assert loaded.classifier.model == "qwen2.5-coder-router:14b"
assert loaded.verification.model == loaded.classifier.model
assert loaded.local_vision.model == "qwen3-vl-router:4b"
# The consolidation invariant: dispatch runs on the classifier's tag.
assert loaded.local_dispatch_models[0].model_id == loaded.classifier.model