582 lines
21 KiB
Python
582 lines
21 KiB
Python
"""The classifier and the local verifier are separate endpoints.
|
|
|
|
They used to be one: the verifier derived its URL by stripping ``/v1`` off
|
|
``classifier.base_url``. That silently coupled two unrelated decisions, and
|
|
the coupling only becomes visible when the classifier is moved off-host —
|
|
pointing classification at a cloud provider would have sent every local
|
|
verification to ``<provider>/api/chat``, which does not exist.
|
|
|
|
The verifier speaks Ollama's NATIVE API (``/api/chat`` with ``think=False``)
|
|
because that is the only way to disable the reasoning trace, so it cannot
|
|
follow the classifier anywhere the classifier can go.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import copy
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
import yaml
|
|
|
|
from config import RouterConfig
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
|
|
|
|
@pytest.fixture
|
|
def raw() -> dict:
|
|
with open(ROOT / "config" / "config.yaml") as fh:
|
|
return yaml.safe_load(fh)
|
|
|
|
|
|
def test_moving_the_classifier_to_a_cloud_provider_leaves_the_verifier_local(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1"
|
|
cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY"
|
|
cfg["classifier"]["model"] = "deepseek-v4-flash"
|
|
|
|
loaded = RouterConfig(**cfg)
|
|
|
|
assert loaded.classifier.base_url == "https://api.neuralwatt.com/v1"
|
|
# The whole point: this did NOT follow the line above.
|
|
assert "localhost" in loaded.verification.base_url
|
|
|
|
|
|
def test_the_verifier_defaults_to_a_local_ollama(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["verification"].pop("base_url", None)
|
|
assert RouterConfig(**cfg).verification.base_url == "http://localhost:11434"
|
|
|
|
|
|
def test_an_absent_api_key_env_means_unauthenticated(raw):
|
|
# The local Ollama case. It ignores the key entirely, but the SDK requires
|
|
# one to be set, so the dispatcher substitutes a placeholder. Asserted on
|
|
# an explicit config rather than on whatever config.yaml currently says,
|
|
# which is a deployment choice and not a property of the code.
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["classifier"]["base_url"] = "http://localhost:11434/v1"
|
|
cfg["classifier"].pop("api_key_env", None)
|
|
assert RouterConfig(**cfg).classifier.api_key_env is None
|
|
|
|
cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY"
|
|
assert RouterConfig(**cfg).classifier.api_key_env == "NEURALWATT_API_KEY"
|
|
|
|
|
|
def test_verification_model_may_be_unset_while_both_run_on_one_host(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["classifier"]["base_url"] = "http://localhost:11434/v1"
|
|
cfg["verification"]["model"] = None
|
|
loaded = RouterConfig(**cfg)
|
|
# Resolved by the dispatcher as `verification.model or classifier.model`,
|
|
# which is right only because the hosts match.
|
|
assert loaded.verification.model is None
|
|
|
|
|
|
def test_an_unset_verifier_model_is_refused_once_the_hosts_differ(raw):
|
|
# The failure this prevents is SILENT, which is why it is a load-time
|
|
# error rather than a documented caveat. Observed directly: with the
|
|
# classifier on NeuralWatt and this left null, the verifier POSTed
|
|
# `deepseek-v4-flash` to localhost:11434, 404d, caught it, logged "local
|
|
# verification unavailable" and recorded no sample. Verification looked
|
|
# enabled while producing nothing.
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1"
|
|
cfg["classifier"]["model"] = "deepseek-v4-flash"
|
|
cfg["verification"]["model"] = None
|
|
cfg["verification"]["local_llm_enabled"] = True
|
|
|
|
with pytest.raises(ValueError, match="verification.model must be set"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_a_split_host_setup_loads_once_the_verifier_model_is_stated(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1"
|
|
cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY"
|
|
cfg["classifier"]["model"] = "deepseek-v4-flash"
|
|
cfg["verification"]["model"] = "qwen3.5:latest"
|
|
|
|
loaded = RouterConfig(**cfg)
|
|
assert loaded.classifier.model == "deepseek-v4-flash"
|
|
assert loaded.verification.model == "qwen3.5:latest"
|
|
|
|
|
|
def test_disabling_the_local_check_lifts_the_requirement(raw):
|
|
# A host with no local inference at all: nothing to name, nothing to guard.
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["classifier"]["base_url"] = "https://api.neuralwatt.com/v1"
|
|
cfg["verification"]["model"] = None
|
|
cfg["verification"]["local_llm_enabled"] = False
|
|
|
|
loaded = RouterConfig(**cfg)
|
|
assert loaded.verification.local_llm_enabled is False
|
|
|
|
|
|
def test_a_host_with_no_local_inference_can_turn_the_local_check_off(raw):
|
|
# An RPi has no usable local model. Structural verification is pure Python
|
|
# and keeps running; only the LLM check goes away.
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["verification"]["local_llm_enabled"] = False
|
|
loaded = RouterConfig(**cfg)
|
|
assert loaded.verification.local_llm_enabled is False
|
|
|
|
|
|
# --- an unknown key is an error, not a no-op ------------------------------
|
|
|
|
def test_a_key_in_the_wrong_section_is_rejected(raw):
|
|
# Exactly the mistake that shipped: max_input_chars belongs to the
|
|
# classifier and was written into verification, where pydantic's default
|
|
# extra="ignore" accepted it, dropped it, and left the code default in
|
|
# force. It carried the same value, so nothing looked wrong -- but editing
|
|
# it would have done nothing at all.
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["verification"]["max_input_chars"] = 4000
|
|
with pytest.raises(ValueError, match="max_input_chars"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_a_misspelled_key_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["routing"]["min_tool_proficency"] = 0.5 # sic
|
|
with pytest.raises(ValueError, match="min_tool_proficency"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_the_shipped_config_has_no_unknown_keys(raw):
|
|
# Guards the whole file, not just the sections a test happens to name.
|
|
RouterConfig(**raw)
|
|
|
|
|
|
def test_the_tool_filter_can_be_turned_off_from_config(raw):
|
|
# It is a knob to experiment with, so null must be a legal value rather
|
|
# than something requiring a code change.
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["routing"]["min_tool_proficiency"] = None
|
|
assert RouterConfig(**cfg).routing.min_tool_proficiency is None
|
|
|
|
|
|
def test_disabling_the_filter_lifts_the_category_name_check(raw):
|
|
# With no filter there is nothing to join, so an unused category name
|
|
# must not block startup.
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["routing"]["min_tool_proficiency"] = None
|
|
cfg["routing"]["tool_use_category"] = "not_a_real_category"
|
|
assert RouterConfig(**cfg).routing.min_tool_proficiency is None
|
|
|
|
|
|
# --- the CLI sanity check -------------------------------------------------
|
|
|
|
def test_the_config_summary_names_fields_that_exist(raw):
|
|
"""`python config.py` is the documented setup step, and it crashed.
|
|
|
|
It printed cfg.weights, which had been replaced by cfg.objective, so the
|
|
one command whose job is to prove the config is fine reported "Config
|
|
loaded OK" and then died with AttributeError. A function plus this test
|
|
means the names cannot rot silently again.
|
|
"""
|
|
from config import summary_lines
|
|
|
|
lines = summary_lines(RouterConfig(**raw))
|
|
|
|
assert lines[0] == "Config loaded OK"
|
|
body = "\n".join(lines[1:])
|
|
assert "quality_tolerance=" in body
|
|
assert "classifier:" in body
|
|
assert "dispatch providers:" in body
|
|
assert "billing_reset_day=" in body
|
|
|
|
|
|
|
|
|
|
# --- local_energy_call_sites log-once ---------------------------------------
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("site_key","attr","non_loopback","expected_sites"),
|
|
[
|
|
(
|
|
"classify","classifier",
|
|
"https://api.neuralwatt.com/v1",
|
|
{"classify": False, "verify": True, "local_vision": True},
|
|
),
|
|
(
|
|
"verify","verification",
|
|
"https://api.neuralwatt.com/v1",
|
|
{"classify": True, "verify": False, "local_vision": True},
|
|
),
|
|
(
|
|
"local_vision","local_vision",
|
|
"https://edge-ollama.example.com/v1",
|
|
{"classify": True, "verify": True, "local_vision": False},
|
|
),
|
|
],
|
|
ids=["classifier","verification","local_vision"],
|
|
)
|
|
def test_local_energy_call_sites_warns_once_per_branch(
|
|
raw, caplog, site_key, attr, non_loopback, expected_sites,
|
|
):
|
|
"""Each non-loopback call-site warns exactly once at config load."""
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["local_energy"]["enabled"] = True
|
|
cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0
|
|
cfg["local_energy"]["grid_intensity_g_per_kwh"] = 475.0
|
|
|
|
# Patch the specific section's base_url
|
|
section = cfg[attr]
|
|
section["base_url"] = non_loopback
|
|
|
|
# For the verification branch the hosts-differ validator fires unless
|
|
# we also set an explicit model name.
|
|
if attr == "verification":
|
|
cfg["classifier"]["model"] = "deepseek-v4-flash"
|
|
cfg["classifier"]["api_key_env"] = "NEURALWATT_API_KEY"
|
|
|
|
with caplog.at_level("WARNING", logger="router.config"):
|
|
loaded = RouterConfig(**cfg)
|
|
# Touch the property multiple times after load to verify idempotent
|
|
# logging (warning fires once, property is memoised).
|
|
_ = loaded.local_energy_call_sites
|
|
_ = loaded.local_energy_call_sites
|
|
|
|
warning_records = [
|
|
r for r in caplog.records
|
|
if r.levelname == "WARNING"
|
|
and r.name == "router.config"
|
|
and f"{site_key}.base_url points to a non-loopback host" in r.message
|
|
]
|
|
assert len(warning_records) == 1
|
|
|
|
sites = loaded.local_energy_call_sites
|
|
assert sites == expected_sites
|
|
|
|
|
|
def test_an_in_process_encoder_is_metered_whatever_base_url_says(raw, caplog):
|
|
"""A local_encoder classifier places no HTTP call at all.
|
|
|
|
The model runs IN-PROCESS on this host's own GPU, so classifier.base_url
|
|
describes nothing about where that work happens -- and gating on it
|
|
silently stopped metering a GPU whose electricity is on this machine's
|
|
bill, while warning about a URL nothing calls.
|
|
|
|
Not hypothetical: docs/local-models.md recommends pointing
|
|
classifier.base_url at a VPN address rather than 0.0.0.0, which is exactly
|
|
the configuration that used to switch encoder metering off.
|
|
"""
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["local_energy"]["enabled"] = True
|
|
cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0
|
|
cfg["classifier"]["mode"] = "local_encoder"
|
|
cfg["classifier"]["encoder"] = {}
|
|
cfg["classifier"]["base_url"] = "http://ollama.vpn.example:11434/v1"
|
|
|
|
with caplog.at_level("WARNING", logger="router.config"):
|
|
loaded = RouterConfig(**cfg)
|
|
|
|
assert loaded.local_energy_call_sites["classify"] is True
|
|
assert not [
|
|
r for r in caplog.records
|
|
if "classify.base_url points to a non-loopback host" in r.message
|
|
], "warning names a URL the encoder never calls"
|
|
|
|
|
|
def test_a_remote_ollama_classifier_is_still_not_metered(raw, caplog):
|
|
"""The narrowing must stay narrow: in local_llm mode base_url IS the
|
|
machine doing the work, and a remote one is genuinely unmeterable."""
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["local_energy"]["enabled"] = True
|
|
cfg["local_energy"]["tariff_usd_per_kwh"] = 8.0
|
|
cfg["classifier"]["mode"] = "local_llm"
|
|
cfg["classifier"]["base_url"] = "http://ollama.vpn.example:11434/v1"
|
|
|
|
loaded = RouterConfig(**cfg)
|
|
|
|
assert loaded.local_energy_call_sites["classify"] is False
|
|
|
|
|
|
def test_local_energy_call_sites_no_warning_when_disabled(raw, caplog):
|
|
"""When local_energy.enabled is false, no warning is emitted at all."""
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["local_energy"]["enabled"] = False
|
|
|
|
with caplog.at_level("WARNING", logger="router.config"):
|
|
loaded = RouterConfig(**cfg)
|
|
_ = loaded.local_energy_call_sites
|
|
_ = loaded.local_energy_call_sites
|
|
|
|
warning_records = [r for r in caplog.records if r.levelname == "WARNING"
|
|
and "local_energy" in r.message and r.name == "router.config"]
|
|
assert len(warning_records) == 0
|
|
|
|
# Values are all False when disabled.
|
|
assert loaded.local_energy_call_sites == {"classify": False, "verify": False, "local_vision": False}
|
|
|
|
|
|
|
|
def test_billing_reset_day_accepts_1_through_28():
|
|
"""Valid ranges 1..28 must be accepted."""
|
|
from config import Objective
|
|
|
|
for day in (1, 6, 15, 28):
|
|
obj = Objective(billing_reset_day=day)
|
|
assert obj.billing_reset_day == day
|
|
|
|
|
|
def test_billing_reset_day_accepts_none():
|
|
"""null disables the feature."""
|
|
from config import Objective
|
|
|
|
obj = Objective(billing_reset_day=None)
|
|
assert obj.billing_reset_day is None
|
|
|
|
|
|
def test_billing_reset_day_rejects_out_of_range(raw):
|
|
"""Values outside 1..28 are rejected."""
|
|
import copy
|
|
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["objective"]["billing_reset_day"] = 0
|
|
with pytest.raises(ValueError, match="1\.\.28"):
|
|
RouterConfig(**cfg)
|
|
|
|
cfg["objective"]["billing_reset_day"] = 29
|
|
with pytest.raises(ValueError, match="1\.\.28"):
|
|
RouterConfig(**cfg)
|
|
|
|
cfg["objective"]["billing_reset_day"] = 32
|
|
with pytest.raises(ValueError, match="1\.\.28"):
|
|
RouterConfig(**cfg)
|
|
|
|
cfg["objective"]["billing_reset_day"] = -1
|
|
with pytest.raises(ValueError, match="1\.\.28"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_adoption_window_seconds_rejects_zero():
|
|
"""0 is rejected by adoption_window_positive; null or absent means all time."""
|
|
from config import Objective
|
|
|
|
with pytest.raises(ValueError, match="adoption_window_seconds"):
|
|
Objective(adoption_window_seconds=0)
|
|
|
|
|
|
def test_adoption_window_seconds_null_is_accepted():
|
|
"""None is the documented all-time value."""
|
|
from config import Objective
|
|
|
|
obj = Objective(adoption_window_seconds=None)
|
|
assert obj.adoption_window_seconds is None
|
|
|
|
|
|
def test_a_removed_key_is_rejected_rather_than_ignored(raw):
|
|
"""log_path named a file nothing ever wrote; leaving it valid would lie."""
|
|
import copy
|
|
|
|
import pytest
|
|
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["logging"]["log_path"] = "router.log"
|
|
|
|
with pytest.raises(Exception, match="log_path|Extra inputs"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
# --- capability gates and the local vision fallback ------------------------
|
|
|
|
def test_the_shipped_config_loads_with_the_new_keys(raw):
|
|
loaded = RouterConfig(**raw)
|
|
assert loaded.routing.require_vision is True
|
|
assert loaded.routing.require_json_mode is True
|
|
assert loaded.local_vision.model == "qwen3-vl-router:4b"
|
|
assert loaded.local_vision.enabled is True
|
|
|
|
|
|
def test_a_typo_in_require_vision_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["routing"]["require_visoin"] = True # sic
|
|
with pytest.raises(ValueError, match="require_visoin"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_local_vision_can_be_disabled_from_config(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["local_vision"]["enabled"] = False
|
|
assert RouterConfig(**cfg).local_vision.enabled is False
|
|
|
|
|
|
def test_local_vision_defaults_when_absent(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg.pop("local_vision")
|
|
assert RouterConfig(**cfg).local_vision.enabled is True
|
|
|
|
|
|
def test_nonpositive_local_timeout_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["local_vision"]["timeout_seconds"] = 0
|
|
with pytest.raises(ValueError, match="timeout_seconds"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_the_shipped_config_loads_the_pinch_section(raw):
|
|
loaded = RouterConfig(**raw)
|
|
assert loaded.pinch.enabled is True
|
|
assert loaded.pinch.budget_tokens > 0
|
|
assert loaded.pinch.keep_last_turns > 0
|
|
|
|
|
|
def test_pinch_defaults_when_absent(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg.pop("pinch")
|
|
loaded = RouterConfig(**cfg)
|
|
assert loaded.pinch.enabled is True
|
|
assert loaded.pinch.budget_tokens == 50000
|
|
assert loaded.pinch.keep_last_turns == 4
|
|
|
|
|
|
def test_nonpositive_pinch_budget_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["budget_tokens"] = 0
|
|
with pytest.raises(ValueError, match="budget_tokens"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_nonpositive_pinch_keep_last_turns_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["keep_last_turns"] = 0
|
|
with pytest.raises(ValueError, match="keep_last_turns"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_classifier_context_framing_loads(raw):
|
|
# Defaults on (see config.yaml); the legacy layout is opt-out.
|
|
assert RouterConfig(**raw).classifier.context_framing is True
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["classifier"]["context_framing"] = False
|
|
assert RouterConfig(**cfg).classifier.context_framing is False
|
|
|
|
|
|
def test_nonminimum_pinch_max_summarize_chars_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["max_summarize_chars"] = 100
|
|
with pytest.raises(ValueError, match="max_summarize_chars.*>=.*3000"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_pinch_max_summarize_chars_at_valid_min_loads(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["max_summarize_chars"] = 3000
|
|
loaded = RouterConfig(**cfg)
|
|
assert loaded.pinch.max_summarize_chars == 3000
|
|
|
|
|
|
def test_pinch_max_summarize_chars_accepts_default(raw):
|
|
loaded = RouterConfig(**raw)
|
|
assert loaded.pinch.max_summarize_chars == 4000
|
|
|
|
|
|
def test_nonminimum_pinch_protected_max_chars_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["protected_max_chars"] = 100
|
|
with pytest.raises(ValueError, match="protected_max_chars.*>=.*3000"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_pinch_protected_max_chars_at_valid_min_loads(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["protected_max_chars"] = 3000
|
|
loaded = RouterConfig(**cfg)
|
|
assert loaded.pinch.protected_max_chars == 3000
|
|
|
|
|
|
def test_pinch_protected_max_chars_accepts_default(raw):
|
|
loaded = RouterConfig(**raw)
|
|
assert loaded.pinch.protected_max_chars == 20000
|
|
|
|
|
|
def test_pinch_protected_max_chars_null_disables(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["protected_max_chars"] = None
|
|
loaded = RouterConfig(**cfg)
|
|
assert loaded.pinch.protected_max_chars is None
|
|
|
|
|
|
def test_pinch_relevance_defaults_load(raw):
|
|
loaded = RouterConfig(**raw)
|
|
assert loaded.pinch.relevance.enabled is True
|
|
assert loaded.pinch.relevance.model == "nomic-embed-text"
|
|
assert loaded.pinch.relevance.min_candidates == 2
|
|
|
|
|
|
def test_pinch_relevance_defaults_when_pinch_absent(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg.pop("pinch")
|
|
loaded = RouterConfig(**cfg)
|
|
assert loaded.pinch.relevance.enabled is True
|
|
assert loaded.pinch.relevance.timeout_seconds == 10
|
|
|
|
|
|
def test_nonpositive_pinch_relevance_timeout_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["relevance"]["timeout_seconds"] = 0
|
|
with pytest.raises(ValueError, match="timeout_seconds"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_nonpositive_pinch_relevance_min_candidates_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["pinch"]["relevance"]["min_candidates"] = 0
|
|
with pytest.raises(ValueError, match="min_candidates"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_circuit_breaker_defaults_load(raw):
|
|
loaded = RouterConfig(**raw)
|
|
assert loaded.circuit_breaker.enabled is True
|
|
assert loaded.circuit_breaker.initial_cooldown_seconds == 30
|
|
assert loaded.circuit_breaker.max_cooldown_seconds == 600
|
|
assert loaded.circuit_breaker.backoff_multiplier == 2.0
|
|
|
|
|
|
def test_nonpositive_circuit_breaker_initial_cooldown_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["circuit_breaker"]["initial_cooldown_seconds"] = 0
|
|
with pytest.raises(ValueError, match="initial_cooldown_seconds"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_circuit_breaker_max_below_initial_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["circuit_breaker"]["initial_cooldown_seconds"] = 300
|
|
cfg["circuit_breaker"]["max_cooldown_seconds"] = 200
|
|
with pytest.raises(ValueError, match="max_cooldown_seconds"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_circuit_breaker_multiplier_at_or_below_one_is_rejected(raw):
|
|
cfg = copy.deepcopy(raw)
|
|
cfg["circuit_breaker"]["backoff_multiplier"] = 1.0
|
|
with pytest.raises(ValueError, match="backoff_multiplier"):
|
|
RouterConfig(**cfg)
|
|
|
|
|
|
def test_the_shipped_config_points_classifier_and_verifier_at_the_router_tag(raw):
|
|
# Context size is not a config-level knob: Ollama's OpenAI-compatible
|
|
# endpoint (0.22.0, verified live) silently ignores num_ctx as a
|
|
# per-request field, so it has to be baked into the Ollama model tag via
|
|
# a Modelfile instead (see classifier.model's comment in config.yaml).
|
|
# What config CAN still assert is that classifier and verification point
|
|
# at the same tagged model, so they share one resident instance rather
|
|
# than two differently-sized copies of the same base model.
|
|
#
|
|
# As of 2026-09-03 the SAME tag also serves local dispatch, so one model
|
|
# covers classification, verification and dispatch and only one instance
|
|
# stays resident. That co-residency is what lets the vision fallback stay
|
|
# loaded too (23.2GB of 24GB together) -- see docs/local-models.md.
|
|
loaded = RouterConfig(**raw)
|
|
assert loaded.classifier.model == "qwen2.5-coder-router:14b"
|
|
assert loaded.verification.model == loaded.classifier.model
|
|
assert loaded.local_vision.model == "qwen3-vl-router:4b"
|
|
# The consolidation invariant: dispatch runs on the classifier's tag.
|
|
assert loaded.local_dispatch_models[0].model_id == loaded.classifier.model
|
|
|