"""classify()'s dispatch on cfg.classifier.mode. Every test here asserts on WHICH client/function was constructed or called, not just the returned Classification -- the same style test_classifier_backoff.py established, because the whole point of a mode branch is which implementation actually ran. cloud_llm success is asserted to record source="classifier", specifically NOT "classifier_cloud" -- that string means "the cascade's backup step fired" and feeds the /metrics degradation-share warning as a degraded signal. An intentionally configured cloud primary is not degraded. """ from __future__ import annotations import contextlib import time from types import SimpleNamespace from unittest.mock import MagicMock import pytest import requests import dispatcher import local_decision import local_energy import local_encoder import session_cache @pytest.fixture(autouse=True) def _clean_state(monkeypatch): session_cache.clear() monkeypatch.setattr(dispatcher, "_last_classifier_failure", 0.0) dispatcher._provider_refusal_since.clear() monkeypatch.setattr(dispatcher, "_cached_auto_classifier", None) monkeypatch.setattr(dispatcher, "_auto_classifier_resolved_at", 0.0) token = dispatcher._current_session_key.set(None) yield dispatcher._current_session_key.reset(token) dispatcher._provider_refusal_since.clear() session_cache.clear() def _answer(): return dispatcher.Classification( task_category="coding_general", task_tier=2, required_context_tokens=100, confidence=0.9, source="classifier", ) # --- mode: local_llm (default) -------------------------------------------- def test_default_mode_is_local_llm_and_behaves_as_before(monkeypatch): assert dispatcher.cfg.classifier.mode == "local_llm" built = [] monkeypatch.setattr( dispatcher, "_classifier_client", lambda: built.append(1) or object() ) monkeypatch.setattr(dispatcher, "_classify_once", lambda *a, **k: _answer()) got = dispatcher.classify("do a thing", None) assert built == [1] assert got.source == "classifier" # --- mode: cloud_llm, pinned ----------------------------------------------- def test_cloud_llm_pinned_success_records_source_classifier_not_cloud(monkeypatch): """The critical distinction this feature introduces: intentional cloud primary is NOT the same signal as the cascade's degraded cloud step.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm") monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", False) monkeypatch.setattr( dispatcher.cfg.classifier, "cloud_primary", SimpleNamespace( base_url="https://cloud.example/v1", model="cloud-classifier", timeout_seconds=2, api_key_env=None, max_output_tokens=1024, ), ) monkeypatch.setattr( dispatcher, "_classifier_client", lambda: pytest.fail("local classifier dialled") ) monkeypatch.setattr(dispatcher, "_cloud_classifier_client", lambda cf: object()) monkeypatch.setattr( session_cache, "classify_one", lambda client, model, *a, **k: { "task_category": "reasoning_math", "task_tier": 3, "required_context_tokens": 1234, "confidence": 0.8, } if model == "cloud-classifier" else None, ) got = dispatcher.classify("do a thing", None) assert got.source == "classifier", "primary cloud success must not read as degraded" assert got.task_category == "reasoning_math" def test_cloud_llm_pinned_missing_api_key_falls_through_to_cascade(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm") monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", False) monkeypatch.setattr( dispatcher.cfg.classifier, "cloud_primary", SimpleNamespace( base_url="https://cloud.example/v1", model="cloud-classifier", timeout_seconds=2, api_key_env="DEFINITELY_NOT_SET", max_output_tokens=1024, ), ) monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_fallback", None) got = dispatcher.classify("do a thing", None) assert got.source == "fallback" # --- mode: cloud_llm, auto_classifier -------------------------------------- def test_cloud_llm_auto_resolves_and_dials_the_cheapest_candidate(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm") monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", True) monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary", None) resolved_row = {"model_id": "deepseek-v4-flash", "provider": "neuralwatt"} monkeypatch.setattr(dispatcher, "_resolve_auto_classifier", lambda: resolved_row) monkeypatch.setattr( dispatcher.cfg, "dispatch_providers", { "neuralwatt": SimpleNamespace( base_url="https://api.neuralwatt.com/v1", api_key_env="NEURALWATT_API_KEY" ) }, ) dialled = [] def fake_cloud_client(cf): dialled.append(cf.model) return object() monkeypatch.setattr(dispatcher, "_cloud_classifier_client", fake_cloud_client) monkeypatch.setattr( session_cache, "classify_one", lambda *a, **k: {"task_category": "coding_general", "task_tier": 1}, ) got = dispatcher.classify("do a thing", None) assert dialled == ["deepseek-v4-flash"] assert got.source == "classifier" def test_cloud_llm_auto_with_no_candidate_falls_through_to_cascade(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm") monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", True) monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary", None) monkeypatch.setattr(dispatcher, "_resolve_auto_classifier", lambda: None) monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_fallback", None) got = dispatcher.classify("do a thing", None) assert got.source == "fallback" def test_auto_classifier_resolution_is_cached_across_calls(monkeypatch): """A DB scan per request would sit on the latency floor for a value that only moves as fast as the catalog's own poll cadence.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm") monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", True) monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary", None) monkeypatch.setattr(dispatcher.cfg.classifier, "cooldown_seconds", 3600) resolve_calls = [] def fake_resolve(): resolve_calls.append(1) return {"model_id": "m", "provider": "neuralwatt"} monkeypatch.setattr(dispatcher, "_resolve_auto_classifier", fake_resolve) # Bypass the real caching internals by calling the cached wrapper twice # through the public entry point instead of the raw resolver: monkeypatch.setattr( dispatcher.cfg, "dispatch_providers", {"neuralwatt": SimpleNamespace(base_url="https://x/v1", api_key_env=None)}, ) monkeypatch.setattr(dispatcher, "_cloud_classifier_client", lambda cf: object()) monkeypatch.setattr( session_cache, "classify_one", lambda *a, **k: {"task_category": "coding_general", "task_tier": 1}, ) dispatcher.classify("a", None) dispatcher.classify("b", None) # _resolve_auto_classifier itself was monkeypatched above, so this only # proves classify() calls the resolver each time -- the resolver's OWN # caching is covered by test_resolve_auto_classifier_caches_within_the_cooldown_window. assert resolve_calls == [1, 1] def test_resolve_auto_classifier_caches_within_the_cooldown_window(monkeypatch): monkeypatch.setattr(dispatcher, "_cached_auto_classifier", None) monkeypatch.setattr(dispatcher, "_auto_classifier_resolved_at", 0.0) monkeypatch.setattr(dispatcher.cfg.classifier, "cooldown_seconds", 3600) monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_category", "general_chat") monkeypatch.setattr(dispatcher.cfg.routing, "allowed_access_levels", ["public"]) load_calls = [] def fake_load_candidates(conn, category): load_calls.append(category) return [{"model_id": "m", "provider": "neuralwatt"}] monkeypatch.setattr(dispatcher, "load_candidates", fake_load_candidates) monkeypatch.setattr(dispatcher, "cheapest_classifier_candidate", lambda rows, **k: rows[0]) monkeypatch.setattr(dispatcher, "_db", lambda: MagicMock()) first = dispatcher._resolve_auto_classifier() second = dispatcher._resolve_auto_classifier() assert first == second == {"model_id": "m", "provider": "neuralwatt"} assert len(load_calls) == 1, "second call within the cooldown window re-queried the DB" def test_resolve_auto_classifier_refreshes_after_the_cooldown_window(monkeypatch): monkeypatch.setattr(dispatcher, "_cached_auto_classifier", {"model_id": "stale"}) monkeypatch.setattr( dispatcher, "_auto_classifier_resolved_at", dispatcher.time.time() - 9999 ) monkeypatch.setattr(dispatcher.cfg.classifier, "cooldown_seconds", 30) monkeypatch.setattr(dispatcher, "load_candidates", lambda conn, category: []) monkeypatch.setattr( dispatcher, "cheapest_classifier_candidate", lambda rows, **k: {"model_id": "fresh"} ) monkeypatch.setattr(dispatcher, "_db", lambda: MagicMock()) got = dispatcher._resolve_auto_classifier() assert got == {"model_id": "fresh"} # --- mode: local_encoder ---------------------------------------------------- def test_local_encoder_success_records_source_classifier(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder") monkeypatch.setattr( dispatcher.cfg.classifier, "encoder", SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5), ) monkeypatch.setattr( dispatcher, "_classifier_client", lambda: pytest.fail("local LLM dialled") ) monkeypatch.setattr( local_encoder, "classify_zero_shot", lambda task, categories, **k: ("coding_refactor", 0.83), ) got = dispatcher.classify("refactor this function", None) assert got.source == "classifier" assert got.task_category == "coding_refactor" assert got.confidence == pytest.approx(0.83) def test_local_encoder_uses_fallback_tier_not_a_second_heuristic(monkeypatch): """A documented limitation, not a bug: the encoder produces a category, not a tier.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "encoder", SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5), ) monkeypatch.setattr( local_encoder, "classify_zero_shot", lambda task, categories, **k: ("general_chat", 0.9) ) got = dispatcher.classify("hello", None) assert got.task_tier == 2 def test_local_encoder_below_threshold_confidence_cascades(monkeypatch): """Treated identically to a local-LLM parse failure -- the cascade reuses 100% of its existing machinery, unmodified.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder") monkeypatch.setattr( dispatcher.cfg.classifier, "encoder", SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.6), ) monkeypatch.setattr( local_encoder, "classify_zero_shot", lambda task, categories, **k: ("general_chat", 0.2), ) monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_fallback", None) got = dispatcher.classify("do a thing", None) assert got.source == "fallback", "a low-confidence guess must not be treated as real" def test_local_encoder_below_threshold_uses_the_real_cascade(monkeypatch): """Not just 'degrades somehow' -- specifically walks _classify_cascade, same as every other mode's failure.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder") monkeypatch.setattr( dispatcher.cfg.classifier, "encoder", SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.6), ) monkeypatch.setattr( local_encoder, "classify_zero_shot", lambda task, categories, **k: ("x", 0.1) ) session_cache.put("sess-enc", task_category="coding_refactor", task_tier=3) token = dispatcher._current_session_key.set("sess-enc") try: got = dispatcher.classify("do a thing", None) finally: dispatcher._current_session_key.reset(token) assert got.source == "session_stale" assert got.task_category == "coding_refactor" # --- mode: local_decision --------------------------------------------------- def _decision_conf(confidence_min=0.5, coverage_min=0.3, tier_enabled=False): return SimpleNamespace( base_url="http://localhost:11434", model="stub-model", num_ctx=8192, timeout_s=10, confidence_min=confidence_min, coverage_min=coverage_min, tier_enabled=tier_enabled, ) def _logprob_response(label="coding_refactor"): return { "logprobs": [ { "token": label, "logprob": -0.01, "top_logprobs": [{"token": label, "logprob": -0.01}], } ] } def test_local_decision_success_records_source_classifier(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5) ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) calls = [] monkeypatch.setattr( local_decision, "classify_category", lambda *a, **k: calls.append((a, k)) or ("coding_refactor", 0.83, 0.9), ) got = dispatcher._classify_via_local_decision("sys", "refactor this function") assert got.source == "classifier" assert got.task_category == "coding_refactor" assert got.confidence == pytest.approx(0.83) assert calls[0][0][0] == "refactor this function" assert calls[0][1]["base_url"] == "http://localhost:11434" def test_local_decision_uses_fallback_tier(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 3) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5) ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) monkeypatch.setattr( local_decision, "classify_category", lambda *a, **k: ("general_chat", 0.9, 0.9), ) got = dispatcher._classify_via_local_decision("sys", "hello") assert got.task_tier == 3 def test_local_decision_below_confidence_cascades(monkeypatch): """A below-confidence local_decision verdict is a failure, not a guess.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.6) ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) monkeypatch.setattr( local_decision, "classify_category", lambda *a, **k: ("general_chat", 0.2, 0.9), ) with pytest.raises(RuntimeError, match="confidence"): dispatcher._classify_via_local_decision("sys", "do a thing") def test_local_decision_skip_reason_respected(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5) ) monkeypatch.setattr( dispatcher, "_local_classifier_skip_reason", lambda: "gaming_mode" ) with pytest.raises(dispatcher._ClassifierSkipped, match="gaming_mode"): dispatcher._classify_via_local_decision("sys", "do a thing") # --- startup: local_encoder mode must fail loudly, not on the first request - def test_startup_check_is_a_noop_for_other_modes(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_llm") monkeypatch.setattr( local_encoder, "ensure_available", lambda *a, **k: pytest.fail("ensure_available called for a non-encoder mode"), ) dispatcher._ensure_classifier_mode_ready() # must not raise, must not call ensure_available def test_startup_check_calls_ensure_available_for_local_encoder_mode(monkeypatch): monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder") monkeypatch.setattr( dispatcher.cfg.classifier, "encoder", SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5), ) calls = [] monkeypatch.setattr( local_encoder, "ensure_available", lambda model, device: calls.append((model, device)) ) dispatcher._ensure_classifier_mode_ready() assert calls == [("stub-model", "cpu")] def test_startup_check_surfaces_a_missing_dependency_loudly(monkeypatch): """The whole point: fail at boot, not as an opaque error on the first live classification request.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder") monkeypatch.setattr( dispatcher.cfg.classifier, "encoder", SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5), ) def fake_ensure_available(model, device): raise ImportError( "classifier.mode is 'local_encoder' but the 'transformers'/'torch' " "packages are not installed. Install them with: " "pip install -r requirements-encoder.txt" ) monkeypatch.setattr(local_encoder, "ensure_available", fake_ensure_available) with pytest.raises(ImportError, match="pip install -r requirements-encoder.txt"): dispatcher._ensure_classifier_mode_ready() # --- startup: local_decision mode must warn, not fail, if Ollama is down ---- def _decision_startup_conf(): return _decision_conf(confidence_min=0.5, coverage_min=0.3) def test_startup_check_validates_decision_mode(monkeypatch): """A healthy local_decision mode makes one classify_category call and passes.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "decision", _decision_startup_conf()) calls = [] monkeypatch.setattr( local_decision, "classify_category", lambda *a, **k: calls.append((a, k)) or ("coding_general", 0.9, 0.9), ) dispatcher._ensure_classifier_mode_ready() # must not raise assert len(calls) == 1 assert calls[0][0][0] == "test task" assert calls[0][1]["base_url"] == "http://localhost:11434" assert calls[0][1]["model"] == "stub-model" def test_startup_check_warns_on_decision_failure(monkeypatch, caplog): """Ollama down at boot: warn and continue — the cascade handles failures.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "decision", _decision_startup_conf()) def fake_classify_category(*a, **k): raise ConnectionError("Ollama is not running") monkeypatch.setattr(local_decision, "classify_category", fake_classify_category) caplog.set_level("WARNING", logger="llm_router") dispatcher._ensure_classifier_mode_ready() # must NOT raise assert any( "classifier_mode_unavailable" in r.getMessage() and "mode=local_decision" in r.getMessage() and "Ollama is not running" in r.getMessage() for r in caplog.records ), "a local_decision startup failure must be logged as classifier_mode_unavailable" def test_startup_check_is_noop_for_non_decision_modes(monkeypatch): """Encoder mode still fires ensure_available; decision path is untouched.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder") monkeypatch.setattr( dispatcher.cfg.classifier, "encoder", SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5), ) calls = [] monkeypatch.setattr( local_encoder, "ensure_available", lambda model, device: calls.append((model, device)) ) monkeypatch.setattr( local_decision, "classify_choice", lambda *a, **k: pytest.fail("classify_choice called for a non-decision mode"), ) dispatcher._ensure_classifier_mode_ready() assert calls == [("stub-model", "cpu")] def test_startup_check_skipped_when_local_compute_off(monkeypatch): """When local_compute.enabled is False (gaming mode), the local_decision startup probe must not call Ollama — it would load the 4b model onto the GPU. """ monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "decision", _decision_startup_conf()) monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", False) def probe_should_not_be_called(*a, **k): pytest.fail("classify_category must NOT be called when local_compute.enabled is False") monkeypatch.setattr(local_decision, "classify_category", probe_should_not_be_called) dispatcher._ensure_classifier_mode_ready() # --- tier classification in local_decision mode ------------------------------ def test_local_decision_tier_enabled_false_no_tier_calls(monkeypatch): """When tier_enabled is False (the default) classify_category fires only once — for category — and tier equals fallback_tier.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5) ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) calls = [] monkeypatch.setattr( local_decision, "classify_category", lambda *a, **k: calls.append(1) or ("coding_refactor", 0.83, 0.9), ) got = dispatcher._classify_via_local_decision("sys", "refactor this") assert got.task_tier == 2 assert len(calls) == 1, "classify_category should fire only once for category" def test_local_decision_tier_enabled_true_makes_tier_call(monkeypatch): """When tier_enabled is True classify_category fires once for category, and classify_choice fires once for tier via the tier function, and the returned tier matches the tier prompt's option letter mapping.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 1) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=True), ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) call_count = 0 def fake_classify_category(*a, **k): nonlocal call_count call_count += 1 return ("coding_refactor", 0.83, 0.9) def fake_classify_choice(*a, **k): nonlocal call_count call_count += 1 # Return B → tier 2 return ("B", 0.75, 0.85) monkeypatch.setattr(local_decision, "classify_category", fake_classify_category) monkeypatch.setattr(local_decision, "classify_choice", fake_classify_choice) got = dispatcher._classify_via_local_decision("sys", "refactor this") assert got.task_tier == 2 assert got.task_category == "coding_refactor" assert call_count == 2, ( "classify_category once for category + classify_choice once for tier" ) def test_local_decision_tier_failure_falls_to_fallback(monkeypatch): """If the tier call raises (network/model error) tier falls to fallback_tier with zero disruption to the category result.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 3) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=True), ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) call_count = 0 def fake_classify_category(*a, **k): nonlocal call_count call_count += 1 return ("coding_refactor", 0.83, 0.9) def fake_classify_choice(*a, **k): nonlocal call_count call_count += 1 # Second call: simulate network/model failure raise ConnectionError("Ollama is not running") monkeypatch.setattr(local_decision, "classify_category", fake_classify_category) monkeypatch.setattr(local_decision, "classify_choice", fake_classify_choice) got = dispatcher._classify_via_local_decision("sys", "refactor this") assert got.task_tier == 3, "tier should fall to fallback on failure" assert got.task_category == "coding_refactor" assert call_count == 2 def test_local_decision_tier_runs_in_parallel_with_category(monkeypatch): """Category and tier calls execute concurrently when tier_enabled is True. Both must hit a barrier before either completes — proving they run in separate threads inside a shared ThreadPoolExecutor.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=True), ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) import threading import time barrier = threading.Barrier(2, timeout=2) def fake_classify_category(*a, **k): time.sleep(0.05) barrier.wait() return ("coding_refactor", 0.83, 0.9) def fake_classify_choice(*a, **k): time.sleep(0.05) barrier.wait() return ("B", 0.75, 0.85) monkeypatch.setattr(local_decision, "classify_category", fake_classify_category) monkeypatch.setattr(local_decision, "classify_choice", fake_classify_choice) start = time.time() got = dispatcher._classify_via_local_decision("sys", "refactor this") elapsed = time.time() - start assert got.task_category == "coding_refactor" assert got.task_tier == 2 # With two 50 ms sleeps and a barrier, sequential code takes ≥0.2 s, # parallel code ≈0.05 s. Give generous slack but fail on sequential. assert elapsed < 0.25, f"took {elapsed:.3f}s — calls ran sequentially, not in parallel" def test_local_decision_metered_logs_correct_duration_power(monkeypatch): def fake_nvidia(*a, **k): return 100.0 def fake_post(*a, **k): time.sleep(0.2) return SimpleNamespace( json=lambda: { "model": "qwen3.5:4b", "message": {"role": "assistant", "content": "B"}, "logprobs": [{ "top_logprobs": [ {"token": "B", "logprob": -0.1}, {"token": "A", "logprob": -2.0}, {"token": "C", "logprob": -3.0}, ] }], }, raise_for_status=lambda: None, ) def capture_log(*a, measurement=None, **k): measurement_sent.append(SimpleNamespace( duration_seconds=measurement.duration_seconds, avg_power_watts=measurement.avg_power_watts, )) measurement_sent = [] monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=False), ) monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", True) monkeypatch.setattr( dispatcher, "cfg", SimpleNamespace( classifier=dispatcher.cfg.classifier, local_energy=SimpleNamespace( enabled=True, sample_interval_seconds=0.01, call_sites={"classify": True}, tariff=0.08, ), local_energy_call_sites={"classify": True}, ), ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) monkeypatch.setattr(dispatcher.local_energy, "sample_nvidia_smi", fake_nvidia) monkeypatch.setattr(local_decision.requests, "post", fake_post) monkeypatch.setattr(dispatcher, "_log_local_energy", capture_log) got = dispatcher._classify_via_local_decision("sys", "refactor this") assert got.task_category == "coding_refactor" assert len(measurement_sent) == 1, "expected one _log_local_energy call" sent = measurement_sent[0] assert sent.duration_seconds >= 0.15, \ f"expected >= 0.15, got {sent.duration_seconds}" assert sent.avg_power_watts == 100.0, \ f"expected 100.0, got {sent.avg_power_watts}" # --- HTTP-level tests: letter→category mapping via mocked requests.post ------ # These tests fake local_decision.requests.post (the lowest HTTP boundary) # so they exercise classify_choice → parse_logprobs → letter→category mapping # end-to-end, NOT just the return value of a stubbed classify_category. def make_response(letter="B", top_logprobs=None): """Create a mock response object like requests.post() returns. ``top_logprobs`` defaults to three options with B as clear winner. """ if top_logprobs is None: top_logprobs = [ {"token": letter, "logprob": -0.1}, {"token": "A", "logprob": -2.0}, {"token": "C", "logprob": -3.0}, ] return SimpleNamespace( json=lambda: { "model": "qwen3.5:4b", "message": {"role": "assistant", "content": letter}, "logprobs": [{"top_logprobs": top_logprobs}], }, raise_for_status=lambda: None, ) def test_classify_category_maps_letter_to_category(monkeypatch): """When requests.post returns top token 'B', classify_category maps it to the second key of _DECISION_DESCRIPTIONS ('coding_refactor').""" from src.local_decision import _DECISION_DESCRIPTIONS, classify_category second_key = list(_DECISION_DESCRIPTIONS.keys())[1] captured_calls = [] def fake_post(*a, **k): captured_calls.append((a, k)) return make_response(letter="B") monkeypatch.setattr(local_decision.requests, "post", fake_post) result = classify_category( "test task", base_url="http://localhost:11434", model="qwen3.5:4b", num_ctx=8192, timeout_s=10, coverage_min=0.3, ) label, confidence, coverage = result assert label == second_key, f"Expected {second_key!r}, got {label!r}" assert len(captured_calls) == 1 assert 0.0 < confidence <= 1.0 assert coverage > 0.3 def test_dispatcher_returns_category_from_letter(monkeypatch): """When the model returns letter 'B', _classify_via_local_decision returns the correct category ('coding_refactor') with source='classifier', proving the full requests.post → classify_choice → letter→category mapping works through the dispatcher.""" from src.local_decision import _DECISION_DESCRIPTIONS second_key = list(_DECISION_DESCRIPTIONS.keys())[1] captured_calls = [] def fake_post(*a, **k): captured_calls.append((a, k)) return make_response(letter="B") monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=False), ) monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) monkeypatch.setattr(local_decision.requests, "post", fake_post) result = dispatcher._classify_via_local_decision("sys", "test task") assert result.task_category == second_key, \ f"Expected {second_key!r}, got {result.task_category!r}" assert result.source == "classifier" assert len(captured_calls) == 1 def test_tier_enabled_false_is_default(monkeypatch): """tier_enabled defaults to False so the normal local_decision path never fires extra HTTP calls without explicit config.""" from config import LocalDecisionConfig assert LocalDecisionConfig().tier_enabled is False def test_unreachable_ollama_walks_the_cascade(monkeypatch): """When local_decision Ollama is unreachable, classify() returns a degraded fallback Classification instead of raising (c06bc36).""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf() ) dispatcher.cfg.classifier.decision.base_url = "http://127.0.0.1:1" monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None) monkeypatch.setattr( local_decision.requests, "post", lambda *a, **k: (_ for _ in ()).throw( requests.ConnectionError("Connection refused") ) ) got = dispatcher.classify("refactor this function", None) assert got.source == "fallback" assert got.confidence == 0.0 assert got.required_context_tokens == 0 assert got.task_tier == dispatcher.cfg.classifier.fallback_tier assert got.task_category == dispatcher.cfg.classifier.fallback_category def test_local_decision_connection_error_opens_circuit_unmetered(monkeypatch): """Transport failure in unmetered local_decision opens the classifier backoff and blocks subsequent calls until the circuit closes.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=False), ) # Make metering off so we hit the unmetered path monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", False) monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True) call_count = [0] def fake_post(*a, **k): call_count[0] += 1 raise requests.ConnectionError("Connection refused") monkeypatch.setattr(local_decision.requests, "post", fake_post) # First call — should record failure and return fallback got1 = dispatcher.classify("refactor this function", None) assert got1.source == "fallback" assert dispatcher._last_classifier_failure > 0.0 assert dispatcher._classifier_backoff_active() is True first_failure_at = dispatcher._last_classifier_failure # Second call — should be SKIPPED (post called exactly once total) got2 = dispatcher.classify("another task", None) assert call_count[0] == 1, f"post called {call_count[0]} times, expected 1" assert got2.source == "fallback" # A skip must never re-stamp the clock, or the circuit stays open forever # while traffic flows. Equality, not just "still set". assert dispatcher._last_classifier_failure == first_failure_at def test_local_decision_connection_error_opens_circuit_metered(monkeypatch): """Transport failure in metered local_decision opens the classifier backoff and blocks subsequent calls until the circuit closes.""" call_count = [0] def fake_nvidia(*a, **k): return 100.0 def fake_post(*a, **k): call_count[0] += 1 raise requests.ConnectionError("Connection refused") def capture_log(*a, measurement=None, **k): pass # just swallow — we don't care about the measurement monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=False), ) monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", True) monkeypatch.setattr(dispatcher.cfg.local_energy, "sample_interval_seconds", 0.01) monkeypatch.setattr(dispatcher.cfg, "_sites_cache", {"classify": True}) monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True) monkeypatch.setattr(dispatcher.local_energy, "sample_nvidia_smi", fake_nvidia) monkeypatch.setattr(local_decision.requests, "post", fake_post) monkeypatch.setattr(dispatcher, "_log_local_energy", capture_log) # First call — should record failure and return fallback got1 = dispatcher.classify("refactor this function", None) assert got1.source == "fallback" assert dispatcher._last_classifier_failure > 0.0 assert dispatcher._classifier_backoff_active() is True first_failure_at = dispatcher._last_classifier_failure # Second call — should be SKIPPED (post called exactly once total) got2 = dispatcher.classify("another task", None) assert call_count[0] == 1, f"post called {call_count[0]} times, expected 1" assert got2.source == "fallback" assert dispatcher._last_classifier_failure == first_failure_at def test_local_decision_below_confidence_min_does_not_open_circuit(monkeypatch): """A verdict below classifier.decision.confidence_min is the dispatcher's own RuntimeError, raised after a healthy answer. The endpoint is fine, so the circuit must stay closed. (The coverage-floor path is the next test.)""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=False), ) monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True) def fake_post(*a, **k): # Four letters with equal mass: coverage is about 0.99 (well above # coverage_min), but the winner holds only a quarter of it, so # confidence is about 0.25, below confidence_min (0.5). return SimpleNamespace( json=lambda: { "logprobs": [{ "top_logprobs": [ {"token": letter, "logprob": -1.4} for letter in ("A", "B", "C", "D") ] }], }, raise_for_status=lambda: None, ) monkeypatch.setattr(local_decision.requests, "post", fake_post) got = dispatcher.classify("refactor this function", None) assert got.source == "fallback" # cascaded on the confidence failure assert dispatcher._last_classifier_failure == 0.0 assert dispatcher._classifier_backoff_active() is False def test_local_decision_coverage_floor_does_not_open_circuit(monkeypatch): """parse_logprobs raising because the model put almost no mass on any option letter (coverage below coverage_min) is also not a transport failure — the circuit must stay closed.""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5, tier_enabled=False), ) monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True) def fake_post(*a, **k): # "Z" is not an option letter and A/B carry almost no mass, so total # option mass (about 0.012) is below coverage_min (0.3) and # parse_logprobs raises RuntimeError before any confidence is computed. return SimpleNamespace( json=lambda: { "model": "qwen3.5:4b", "message": {"role": "assistant", "content": "Z"}, "logprobs": [{ "top_logprobs": [ {"token": "Z", "logprob": -5.0}, # very low confidence {"token": "A", "logprob": -5.1}, {"token": "B", "logprob": -5.2}, ] }], }, raise_for_status=lambda: None, ) monkeypatch.setattr(local_decision.requests, "post", fake_post) # classify() should cascade to fallback because confidence < confidence_min got = dispatcher.classify("refactor this function", None) assert got.source == "fallback" # cascaded # But _last_classifier_failure must be 0.0 — this was NOT a transport error assert dispatcher._last_classifier_failure == 0.0 assert dispatcher._classifier_backoff_active() is False