The review of the backoff fix found two weak spots, both now closed: - The circuit tests asserted only that _last_classifier_failure was still set after the skipped second call. A skip that re-stamped the clock (holding the circuit open for as long as traffic flows) would have passed. Both tests now assert the timestamp is unchanged. - test_local_decision_low_confidence_does_not_open_circuit tripped the parse_logprobs coverage floor, not confidence_min, so the dispatcher's own confidence raise had no test. Split in two: a new test that actually lands below confidence_min (four equal letters, coverage 0.99, confidence 0.25), and the old one renamed and re-documented as the coverage-floor case. Mutation-checked in a scratch export: a skip that records, a confidence raise that records, and recording on any exception each fail the intended tests. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01KkCGRantZsSwmcFpet6FTa
1059 lines
40 KiB
Python
1059 lines
40 KiB
Python
"""classify()'s dispatch on cfg.classifier.mode.
|
|
|
|
Every test here asserts on WHICH client/function was constructed or called,
|
|
not just the returned Classification -- the same style
|
|
test_classifier_backoff.py established, because the whole point of a mode
|
|
branch is which implementation actually ran.
|
|
|
|
cloud_llm success is asserted to record source="classifier", specifically
|
|
NOT "classifier_cloud" -- that string means "the cascade's backup step
|
|
fired" and feeds the /metrics degradation-share warning as a degraded
|
|
signal. An intentionally configured cloud primary is not degraded.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import contextlib
|
|
import time
|
|
|
|
from types import SimpleNamespace
|
|
from unittest.mock import MagicMock
|
|
|
|
import pytest
|
|
import requests
|
|
|
|
import dispatcher
|
|
import local_decision
|
|
import local_energy
|
|
import local_encoder
|
|
import session_cache
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _clean_state(monkeypatch):
|
|
session_cache.clear()
|
|
monkeypatch.setattr(dispatcher, "_last_classifier_failure", 0.0)
|
|
dispatcher._provider_refusal_since.clear()
|
|
monkeypatch.setattr(dispatcher, "_cached_auto_classifier", None)
|
|
monkeypatch.setattr(dispatcher, "_auto_classifier_resolved_at", 0.0)
|
|
token = dispatcher._current_session_key.set(None)
|
|
yield
|
|
dispatcher._current_session_key.reset(token)
|
|
dispatcher._provider_refusal_since.clear()
|
|
session_cache.clear()
|
|
|
|
|
|
def _answer():
|
|
return dispatcher.Classification(
|
|
task_category="coding_general",
|
|
task_tier=2,
|
|
required_context_tokens=100,
|
|
confidence=0.9,
|
|
source="classifier",
|
|
)
|
|
|
|
|
|
# --- mode: local_llm (default) --------------------------------------------
|
|
|
|
|
|
def test_default_mode_is_local_llm_and_behaves_as_before(monkeypatch):
|
|
assert dispatcher.cfg.classifier.mode == "local_llm"
|
|
built = []
|
|
monkeypatch.setattr(
|
|
dispatcher, "_classifier_client", lambda: built.append(1) or object()
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_classify_once", lambda *a, **k: _answer())
|
|
got = dispatcher.classify("do a thing", None)
|
|
assert built == [1]
|
|
assert got.source == "classifier"
|
|
|
|
|
|
# --- mode: cloud_llm, pinned -----------------------------------------------
|
|
|
|
|
|
def test_cloud_llm_pinned_success_records_source_classifier_not_cloud(monkeypatch):
|
|
"""The critical distinction this feature introduces: intentional cloud
|
|
primary is NOT the same signal as the cascade's degraded cloud step."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", False)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"cloud_primary",
|
|
SimpleNamespace(
|
|
base_url="https://cloud.example/v1",
|
|
model="cloud-classifier",
|
|
timeout_seconds=2,
|
|
api_key_env=None,
|
|
max_output_tokens=1024,
|
|
),
|
|
)
|
|
monkeypatch.setattr(
|
|
dispatcher, "_classifier_client", lambda: pytest.fail("local classifier dialled")
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_cloud_classifier_client", lambda cf: object())
|
|
monkeypatch.setattr(
|
|
session_cache,
|
|
"classify_one",
|
|
lambda client, model, *a, **k: {
|
|
"task_category": "reasoning_math",
|
|
"task_tier": 3,
|
|
"required_context_tokens": 1234,
|
|
"confidence": 0.8,
|
|
}
|
|
if model == "cloud-classifier"
|
|
else None,
|
|
)
|
|
|
|
got = dispatcher.classify("do a thing", None)
|
|
|
|
assert got.source == "classifier", "primary cloud success must not read as degraded"
|
|
assert got.task_category == "reasoning_math"
|
|
|
|
|
|
def test_cloud_llm_pinned_missing_api_key_falls_through_to_cascade(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", False)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"cloud_primary",
|
|
SimpleNamespace(
|
|
base_url="https://cloud.example/v1",
|
|
model="cloud-classifier",
|
|
timeout_seconds=2,
|
|
api_key_env="DEFINITELY_NOT_SET",
|
|
max_output_tokens=1024,
|
|
),
|
|
)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_fallback", None)
|
|
got = dispatcher.classify("do a thing", None)
|
|
assert got.source == "fallback"
|
|
|
|
|
|
# --- mode: cloud_llm, auto_classifier --------------------------------------
|
|
|
|
|
|
def test_cloud_llm_auto_resolves_and_dials_the_cheapest_candidate(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", True)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary", None)
|
|
|
|
resolved_row = {"model_id": "deepseek-v4-flash", "provider": "neuralwatt"}
|
|
monkeypatch.setattr(dispatcher, "_resolve_auto_classifier", lambda: resolved_row)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg,
|
|
"dispatch_providers",
|
|
{
|
|
"neuralwatt": SimpleNamespace(
|
|
base_url="https://api.neuralwatt.com/v1", api_key_env="NEURALWATT_API_KEY"
|
|
)
|
|
},
|
|
)
|
|
dialled = []
|
|
|
|
def fake_cloud_client(cf):
|
|
dialled.append(cf.model)
|
|
return object()
|
|
|
|
monkeypatch.setattr(dispatcher, "_cloud_classifier_client", fake_cloud_client)
|
|
monkeypatch.setattr(
|
|
session_cache,
|
|
"classify_one",
|
|
lambda *a, **k: {"task_category": "coding_general", "task_tier": 1},
|
|
)
|
|
|
|
got = dispatcher.classify("do a thing", None)
|
|
|
|
assert dialled == ["deepseek-v4-flash"]
|
|
assert got.source == "classifier"
|
|
|
|
|
|
def test_cloud_llm_auto_with_no_candidate_falls_through_to_cascade(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", True)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary", None)
|
|
monkeypatch.setattr(dispatcher, "_resolve_auto_classifier", lambda: None)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_fallback", None)
|
|
|
|
got = dispatcher.classify("do a thing", None)
|
|
assert got.source == "fallback"
|
|
|
|
|
|
def test_auto_classifier_resolution_is_cached_across_calls(monkeypatch):
|
|
"""A DB scan per request would sit on the latency floor for a value
|
|
that only moves as fast as the catalog's own poll cadence."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "cloud_llm")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary_auto", True)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_primary", None)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cooldown_seconds", 3600)
|
|
|
|
resolve_calls = []
|
|
|
|
def fake_resolve():
|
|
resolve_calls.append(1)
|
|
return {"model_id": "m", "provider": "neuralwatt"}
|
|
|
|
monkeypatch.setattr(dispatcher, "_resolve_auto_classifier", fake_resolve)
|
|
# Bypass the real caching internals by calling the cached wrapper twice
|
|
# through the public entry point instead of the raw resolver:
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg,
|
|
"dispatch_providers",
|
|
{"neuralwatt": SimpleNamespace(base_url="https://x/v1", api_key_env=None)},
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_cloud_classifier_client", lambda cf: object())
|
|
monkeypatch.setattr(
|
|
session_cache,
|
|
"classify_one",
|
|
lambda *a, **k: {"task_category": "coding_general", "task_tier": 1},
|
|
)
|
|
|
|
dispatcher.classify("a", None)
|
|
dispatcher.classify("b", None)
|
|
# _resolve_auto_classifier itself was monkeypatched above, so this only
|
|
# proves classify() calls the resolver each time -- the resolver's OWN
|
|
# caching is covered by test_resolve_auto_classifier_caches_within_the_cooldown_window.
|
|
assert resolve_calls == [1, 1]
|
|
|
|
|
|
def test_resolve_auto_classifier_caches_within_the_cooldown_window(monkeypatch):
|
|
monkeypatch.setattr(dispatcher, "_cached_auto_classifier", None)
|
|
monkeypatch.setattr(dispatcher, "_auto_classifier_resolved_at", 0.0)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cooldown_seconds", 3600)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_category", "general_chat")
|
|
monkeypatch.setattr(dispatcher.cfg.routing, "allowed_access_levels", ["public"])
|
|
|
|
load_calls = []
|
|
|
|
def fake_load_candidates(conn, category):
|
|
load_calls.append(category)
|
|
return [{"model_id": "m", "provider": "neuralwatt"}]
|
|
|
|
monkeypatch.setattr(dispatcher, "load_candidates", fake_load_candidates)
|
|
monkeypatch.setattr(dispatcher, "cheapest_classifier_candidate", lambda rows, **k: rows[0])
|
|
monkeypatch.setattr(dispatcher, "_db", lambda: MagicMock())
|
|
|
|
first = dispatcher._resolve_auto_classifier()
|
|
second = dispatcher._resolve_auto_classifier()
|
|
|
|
assert first == second == {"model_id": "m", "provider": "neuralwatt"}
|
|
assert len(load_calls) == 1, "second call within the cooldown window re-queried the DB"
|
|
|
|
|
|
def test_resolve_auto_classifier_refreshes_after_the_cooldown_window(monkeypatch):
|
|
monkeypatch.setattr(dispatcher, "_cached_auto_classifier", {"model_id": "stale"})
|
|
monkeypatch.setattr(
|
|
dispatcher, "_auto_classifier_resolved_at", dispatcher.time.time() - 9999
|
|
)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cooldown_seconds", 30)
|
|
monkeypatch.setattr(dispatcher, "load_candidates", lambda conn, category: [])
|
|
monkeypatch.setattr(
|
|
dispatcher, "cheapest_classifier_candidate", lambda rows, **k: {"model_id": "fresh"}
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_db", lambda: MagicMock())
|
|
|
|
got = dispatcher._resolve_auto_classifier()
|
|
assert got == {"model_id": "fresh"}
|
|
|
|
|
|
# --- mode: local_encoder ----------------------------------------------------
|
|
|
|
|
|
def test_local_encoder_success_records_source_classifier(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"encoder",
|
|
SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5),
|
|
)
|
|
monkeypatch.setattr(
|
|
dispatcher, "_classifier_client", lambda: pytest.fail("local LLM dialled")
|
|
)
|
|
monkeypatch.setattr(
|
|
local_encoder,
|
|
"classify_zero_shot",
|
|
lambda task, categories, **k: ("coding_refactor", 0.83),
|
|
)
|
|
|
|
got = dispatcher.classify("refactor this function", None)
|
|
|
|
assert got.source == "classifier"
|
|
assert got.task_category == "coding_refactor"
|
|
assert got.confidence == pytest.approx(0.83)
|
|
|
|
|
|
def test_local_encoder_uses_fallback_tier_not_a_second_heuristic(monkeypatch):
|
|
"""A documented limitation, not a bug: the encoder produces a category,
|
|
not a tier."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"encoder",
|
|
SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5),
|
|
)
|
|
monkeypatch.setattr(
|
|
local_encoder, "classify_zero_shot", lambda task, categories, **k: ("general_chat", 0.9)
|
|
)
|
|
|
|
got = dispatcher.classify("hello", None)
|
|
assert got.task_tier == 2
|
|
|
|
|
|
def test_local_encoder_below_threshold_confidence_cascades(monkeypatch):
|
|
"""Treated identically to a local-LLM parse failure -- the cascade
|
|
reuses 100% of its existing machinery, unmodified."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"encoder",
|
|
SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.6),
|
|
)
|
|
monkeypatch.setattr(
|
|
local_encoder,
|
|
"classify_zero_shot",
|
|
lambda task, categories, **k: ("general_chat", 0.2),
|
|
)
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "cloud_fallback", None)
|
|
|
|
got = dispatcher.classify("do a thing", None)
|
|
|
|
assert got.source == "fallback", "a low-confidence guess must not be treated as real"
|
|
|
|
|
|
def test_local_encoder_below_threshold_uses_the_real_cascade(monkeypatch):
|
|
"""Not just 'degrades somehow' -- specifically walks _classify_cascade,
|
|
same as every other mode's failure."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"encoder",
|
|
SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.6),
|
|
)
|
|
monkeypatch.setattr(
|
|
local_encoder, "classify_zero_shot", lambda task, categories, **k: ("x", 0.1)
|
|
)
|
|
session_cache.put("sess-enc", task_category="coding_refactor", task_tier=3)
|
|
token = dispatcher._current_session_key.set("sess-enc")
|
|
try:
|
|
got = dispatcher.classify("do a thing", None)
|
|
finally:
|
|
dispatcher._current_session_key.reset(token)
|
|
|
|
assert got.source == "session_stale"
|
|
assert got.task_category == "coding_refactor"
|
|
|
|
|
|
# --- mode: local_decision ---------------------------------------------------
|
|
|
|
|
|
def _decision_conf(confidence_min=0.5, coverage_min=0.3, tier_enabled=False):
|
|
return SimpleNamespace(
|
|
base_url="http://localhost:11434",
|
|
model="stub-model",
|
|
num_ctx=8192,
|
|
timeout_s=10,
|
|
confidence_min=confidence_min,
|
|
coverage_min=coverage_min,
|
|
tier_enabled=tier_enabled,
|
|
)
|
|
|
|
|
|
def _logprob_response(label="coding_refactor"):
|
|
return {
|
|
"logprobs": [
|
|
{
|
|
"token": label,
|
|
"logprob": -0.01,
|
|
"top_logprobs": [{"token": label, "logprob": -0.01}],
|
|
}
|
|
]
|
|
}
|
|
|
|
|
|
def test_local_decision_success_records_source_classifier(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5)
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
|
|
calls = []
|
|
monkeypatch.setattr(
|
|
local_decision,
|
|
"classify_category",
|
|
lambda *a, **k: calls.append((a, k))
|
|
or ("coding_refactor", 0.83, 0.9),
|
|
)
|
|
|
|
got = dispatcher._classify_via_local_decision("sys", "refactor this function")
|
|
|
|
assert got.source == "classifier"
|
|
assert got.task_category == "coding_refactor"
|
|
assert got.confidence == pytest.approx(0.83)
|
|
assert calls[0][0][0] == "refactor this function"
|
|
assert calls[0][1]["base_url"] == "http://localhost:11434"
|
|
|
|
|
|
def test_local_decision_uses_fallback_tier(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 3)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5)
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
monkeypatch.setattr(
|
|
local_decision,
|
|
"classify_category",
|
|
lambda *a, **k: ("general_chat", 0.9, 0.9),
|
|
)
|
|
|
|
got = dispatcher._classify_via_local_decision("sys", "hello")
|
|
assert got.task_tier == 3
|
|
|
|
|
|
def test_local_decision_below_confidence_cascades(monkeypatch):
|
|
"""A below-confidence local_decision verdict is a failure, not a guess."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.6)
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
monkeypatch.setattr(
|
|
local_decision,
|
|
"classify_category",
|
|
lambda *a, **k: ("general_chat", 0.2, 0.9),
|
|
)
|
|
|
|
with pytest.raises(RuntimeError, match="confidence"):
|
|
dispatcher._classify_via_local_decision("sys", "do a thing")
|
|
|
|
|
|
def test_local_decision_skip_reason_respected(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5)
|
|
)
|
|
monkeypatch.setattr(
|
|
dispatcher, "_local_classifier_skip_reason", lambda: "gaming_mode"
|
|
)
|
|
|
|
with pytest.raises(dispatcher._ClassifierSkipped, match="gaming_mode"):
|
|
dispatcher._classify_via_local_decision("sys", "do a thing")
|
|
|
|
|
|
# --- startup: local_encoder mode must fail loudly, not on the first request -
|
|
|
|
|
|
def test_startup_check_is_a_noop_for_other_modes(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_llm")
|
|
monkeypatch.setattr(
|
|
local_encoder,
|
|
"ensure_available",
|
|
lambda *a, **k: pytest.fail("ensure_available called for a non-encoder mode"),
|
|
)
|
|
dispatcher._ensure_classifier_mode_ready() # must not raise, must not call ensure_available
|
|
|
|
|
|
def test_startup_check_calls_ensure_available_for_local_encoder_mode(monkeypatch):
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"encoder",
|
|
SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5),
|
|
)
|
|
calls = []
|
|
monkeypatch.setattr(
|
|
local_encoder, "ensure_available", lambda model, device: calls.append((model, device))
|
|
)
|
|
dispatcher._ensure_classifier_mode_ready()
|
|
assert calls == [("stub-model", "cpu")]
|
|
|
|
|
|
def test_startup_check_surfaces_a_missing_dependency_loudly(monkeypatch):
|
|
"""The whole point: fail at boot, not as an opaque error on the first
|
|
live classification request."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"encoder",
|
|
SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5),
|
|
)
|
|
|
|
def fake_ensure_available(model, device):
|
|
raise ImportError(
|
|
"classifier.mode is 'local_encoder' but the 'transformers'/'torch' "
|
|
"packages are not installed. Install them with: "
|
|
"pip install -r requirements-encoder.txt"
|
|
)
|
|
|
|
monkeypatch.setattr(local_encoder, "ensure_available", fake_ensure_available)
|
|
|
|
with pytest.raises(ImportError, match="pip install -r requirements-encoder.txt"):
|
|
dispatcher._ensure_classifier_mode_ready()
|
|
|
|
|
|
# --- startup: local_decision mode must warn, not fail, if Ollama is down ----
|
|
|
|
def _decision_startup_conf():
|
|
return _decision_conf(confidence_min=0.5, coverage_min=0.3)
|
|
|
|
|
|
def test_startup_check_validates_decision_mode(monkeypatch):
|
|
"""A healthy local_decision mode makes one classify_category call
|
|
and passes."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "decision", _decision_startup_conf())
|
|
calls = []
|
|
monkeypatch.setattr(
|
|
local_decision,
|
|
"classify_category",
|
|
lambda *a, **k: calls.append((a, k)) or ("coding_general", 0.9, 0.9),
|
|
)
|
|
dispatcher._ensure_classifier_mode_ready() # must not raise
|
|
assert len(calls) == 1
|
|
assert calls[0][0][0] == "test task"
|
|
assert calls[0][1]["base_url"] == "http://localhost:11434"
|
|
assert calls[0][1]["model"] == "stub-model"
|
|
|
|
|
|
def test_startup_check_warns_on_decision_failure(monkeypatch, caplog):
|
|
"""Ollama down at boot: warn and continue — the cascade handles failures."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "decision", _decision_startup_conf())
|
|
|
|
def fake_classify_category(*a, **k):
|
|
raise ConnectionError("Ollama is not running")
|
|
|
|
monkeypatch.setattr(local_decision, "classify_category", fake_classify_category)
|
|
|
|
caplog.set_level("WARNING", logger="llm_router")
|
|
dispatcher._ensure_classifier_mode_ready() # must NOT raise
|
|
|
|
assert any(
|
|
"classifier_mode_unavailable" in r.getMessage()
|
|
and "mode=local_decision" in r.getMessage()
|
|
and "Ollama is not running" in r.getMessage()
|
|
for r in caplog.records
|
|
), "a local_decision startup failure must be logged as classifier_mode_unavailable"
|
|
|
|
|
|
def test_startup_check_is_noop_for_non_decision_modes(monkeypatch):
|
|
"""Encoder mode still fires ensure_available; decision path is untouched."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_encoder")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier,
|
|
"encoder",
|
|
SimpleNamespace(model="stub-model", device="cpu", confidence_min=0.5),
|
|
)
|
|
calls = []
|
|
monkeypatch.setattr(
|
|
local_encoder, "ensure_available", lambda model, device: calls.append((model, device))
|
|
)
|
|
monkeypatch.setattr(
|
|
local_decision,
|
|
"classify_choice",
|
|
lambda *a, **k: pytest.fail("classify_choice called for a non-decision mode"),
|
|
)
|
|
dispatcher._ensure_classifier_mode_ready()
|
|
assert calls == [("stub-model", "cpu")]
|
|
|
|
|
|
def test_startup_check_skipped_when_local_compute_off(monkeypatch):
|
|
"""When local_compute.enabled is False (gaming mode), the local_decision
|
|
startup probe must not call Ollama — it would load the 4b model onto the GPU.
|
|
"""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "decision", _decision_startup_conf())
|
|
monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", False)
|
|
|
|
def probe_should_not_be_called(*a, **k):
|
|
pytest.fail("classify_category must NOT be called when local_compute.enabled is False")
|
|
|
|
monkeypatch.setattr(local_decision, "classify_category", probe_should_not_be_called)
|
|
|
|
dispatcher._ensure_classifier_mode_ready()
|
|
|
|
|
|
# --- tier classification in local_decision mode ------------------------------
|
|
|
|
|
|
def test_local_decision_tier_enabled_false_no_tier_calls(monkeypatch):
|
|
"""When tier_enabled is False (the default) classify_category fires
|
|
only once — for category — and tier equals fallback_tier."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision", _decision_conf(confidence_min=0.5)
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
|
|
calls = []
|
|
monkeypatch.setattr(
|
|
local_decision,
|
|
"classify_category",
|
|
lambda *a, **k: calls.append(1) or ("coding_refactor", 0.83, 0.9),
|
|
)
|
|
|
|
got = dispatcher._classify_via_local_decision("sys", "refactor this")
|
|
|
|
assert got.task_tier == 2
|
|
assert len(calls) == 1, "classify_category should fire only once for category"
|
|
|
|
|
|
def test_local_decision_tier_enabled_true_makes_tier_call(monkeypatch):
|
|
"""When tier_enabled is True classify_category fires once for
|
|
category, and classify_choice fires once for tier via the tier
|
|
function, and the returned tier matches the tier prompt's
|
|
option letter mapping."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 1)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=True),
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
|
|
call_count = 0
|
|
|
|
def fake_classify_category(*a, **k):
|
|
nonlocal call_count
|
|
call_count += 1
|
|
return ("coding_refactor", 0.83, 0.9)
|
|
|
|
def fake_classify_choice(*a, **k):
|
|
nonlocal call_count
|
|
call_count += 1
|
|
# Return B → tier 2
|
|
return ("B", 0.75, 0.85)
|
|
|
|
monkeypatch.setattr(local_decision, "classify_category", fake_classify_category)
|
|
monkeypatch.setattr(local_decision, "classify_choice", fake_classify_choice)
|
|
|
|
got = dispatcher._classify_via_local_decision("sys", "refactor this")
|
|
|
|
assert got.task_tier == 2
|
|
assert got.task_category == "coding_refactor"
|
|
assert call_count == 2, (
|
|
"classify_category once for category + classify_choice once for tier"
|
|
)
|
|
|
|
|
|
def test_local_decision_tier_failure_falls_to_fallback(monkeypatch):
|
|
"""If the tier call raises (network/model error) tier falls to
|
|
fallback_tier with zero disruption to the category result."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 3)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=True),
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
|
|
call_count = 0
|
|
|
|
def fake_classify_category(*a, **k):
|
|
nonlocal call_count
|
|
call_count += 1
|
|
return ("coding_refactor", 0.83, 0.9)
|
|
|
|
def fake_classify_choice(*a, **k):
|
|
nonlocal call_count
|
|
call_count += 1
|
|
# Second call: simulate network/model failure
|
|
raise ConnectionError("Ollama is not running")
|
|
|
|
monkeypatch.setattr(local_decision, "classify_category", fake_classify_category)
|
|
monkeypatch.setattr(local_decision, "classify_choice", fake_classify_choice)
|
|
|
|
got = dispatcher._classify_via_local_decision("sys", "refactor this")
|
|
|
|
assert got.task_tier == 3, "tier should fall to fallback on failure"
|
|
assert got.task_category == "coding_refactor"
|
|
assert call_count == 2
|
|
|
|
|
|
def test_local_decision_tier_runs_in_parallel_with_category(monkeypatch):
|
|
"""Category and tier calls execute concurrently when tier_enabled is
|
|
True. Both must hit a barrier before either completes — proving they
|
|
run in separate threads inside a shared ThreadPoolExecutor."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=True),
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
|
|
import threading
|
|
import time
|
|
|
|
barrier = threading.Barrier(2, timeout=2)
|
|
|
|
def fake_classify_category(*a, **k):
|
|
time.sleep(0.05)
|
|
barrier.wait()
|
|
return ("coding_refactor", 0.83, 0.9)
|
|
|
|
def fake_classify_choice(*a, **k):
|
|
time.sleep(0.05)
|
|
barrier.wait()
|
|
return ("B", 0.75, 0.85)
|
|
|
|
monkeypatch.setattr(local_decision, "classify_category", fake_classify_category)
|
|
monkeypatch.setattr(local_decision, "classify_choice", fake_classify_choice)
|
|
|
|
start = time.time()
|
|
got = dispatcher._classify_via_local_decision("sys", "refactor this")
|
|
elapsed = time.time() - start
|
|
|
|
assert got.task_category == "coding_refactor"
|
|
assert got.task_tier == 2
|
|
# With two 50 ms sleeps and a barrier, sequential code takes ≥0.2 s,
|
|
# parallel code ≈0.05 s. Give generous slack but fail on sequential.
|
|
assert elapsed < 0.25, f"took {elapsed:.3f}s — calls ran sequentially, not in parallel"
|
|
|
|
|
|
def test_local_decision_metered_logs_correct_duration_power(monkeypatch):
|
|
def fake_nvidia(*a, **k):
|
|
return 100.0
|
|
|
|
def fake_post(*a, **k):
|
|
time.sleep(0.2)
|
|
return SimpleNamespace(
|
|
json=lambda: {
|
|
"model": "qwen3.5:4b",
|
|
"message": {"role": "assistant", "content": "B"},
|
|
"logprobs": [{
|
|
"top_logprobs": [
|
|
{"token": "B", "logprob": -0.1},
|
|
{"token": "A", "logprob": -2.0},
|
|
{"token": "C", "logprob": -3.0},
|
|
]
|
|
}],
|
|
},
|
|
raise_for_status=lambda: None,
|
|
)
|
|
|
|
def capture_log(*a, measurement=None, **k):
|
|
measurement_sent.append(SimpleNamespace(
|
|
duration_seconds=measurement.duration_seconds,
|
|
avg_power_watts=measurement.avg_power_watts,
|
|
))
|
|
|
|
measurement_sent = []
|
|
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=False),
|
|
)
|
|
monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", True)
|
|
monkeypatch.setattr(
|
|
dispatcher, "cfg",
|
|
SimpleNamespace(
|
|
classifier=dispatcher.cfg.classifier,
|
|
local_energy=SimpleNamespace(
|
|
enabled=True,
|
|
sample_interval_seconds=0.01,
|
|
call_sites={"classify": True},
|
|
tariff=0.08,
|
|
),
|
|
local_energy_call_sites={"classify": True},
|
|
),
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
monkeypatch.setattr(dispatcher.local_energy, "sample_nvidia_smi", fake_nvidia)
|
|
monkeypatch.setattr(local_decision.requests, "post", fake_post)
|
|
monkeypatch.setattr(dispatcher, "_log_local_energy", capture_log)
|
|
|
|
got = dispatcher._classify_via_local_decision("sys", "refactor this")
|
|
assert got.task_category == "coding_refactor"
|
|
|
|
assert len(measurement_sent) == 1, "expected one _log_local_energy call"
|
|
sent = measurement_sent[0]
|
|
assert sent.duration_seconds >= 0.15, \
|
|
f"expected >= 0.15, got {sent.duration_seconds}"
|
|
assert sent.avg_power_watts == 100.0, \
|
|
f"expected 100.0, got {sent.avg_power_watts}"
|
|
|
|
|
|
# --- HTTP-level tests: letter→category mapping via mocked requests.post ------
|
|
# These tests fake local_decision.requests.post (the lowest HTTP boundary)
|
|
# so they exercise classify_choice → parse_logprobs → letter→category mapping
|
|
# end-to-end, NOT just the return value of a stubbed classify_category.
|
|
|
|
|
|
def make_response(letter="B", top_logprobs=None):
|
|
"""Create a mock response object like requests.post() returns.
|
|
|
|
``top_logprobs`` defaults to three options with B as clear winner.
|
|
"""
|
|
if top_logprobs is None:
|
|
top_logprobs = [
|
|
{"token": letter, "logprob": -0.1},
|
|
{"token": "A", "logprob": -2.0},
|
|
{"token": "C", "logprob": -3.0},
|
|
]
|
|
return SimpleNamespace(
|
|
json=lambda: {
|
|
"model": "qwen3.5:4b",
|
|
"message": {"role": "assistant", "content": letter},
|
|
"logprobs": [{"top_logprobs": top_logprobs}],
|
|
},
|
|
raise_for_status=lambda: None,
|
|
)
|
|
|
|
|
|
def test_classify_category_maps_letter_to_category(monkeypatch):
|
|
"""When requests.post returns top token 'B', classify_category maps it
|
|
to the second key of _DECISION_DESCRIPTIONS ('coding_refactor')."""
|
|
from src.local_decision import _DECISION_DESCRIPTIONS, classify_category
|
|
|
|
second_key = list(_DECISION_DESCRIPTIONS.keys())[1]
|
|
|
|
captured_calls = []
|
|
|
|
def fake_post(*a, **k):
|
|
captured_calls.append((a, k))
|
|
return make_response(letter="B")
|
|
|
|
monkeypatch.setattr(local_decision.requests, "post", fake_post)
|
|
|
|
result = classify_category(
|
|
"test task",
|
|
base_url="http://localhost:11434",
|
|
model="qwen3.5:4b",
|
|
num_ctx=8192,
|
|
timeout_s=10,
|
|
coverage_min=0.3,
|
|
)
|
|
|
|
label, confidence, coverage = result
|
|
assert label == second_key, f"Expected {second_key!r}, got {label!r}"
|
|
assert len(captured_calls) == 1
|
|
assert 0.0 < confidence <= 1.0
|
|
assert coverage > 0.3
|
|
|
|
|
|
def test_dispatcher_returns_category_from_letter(monkeypatch):
|
|
"""When the model returns letter 'B', _classify_via_local_decision
|
|
returns the correct category ('coding_refactor') with source='classifier',
|
|
proving the full requests.post → classify_choice → letter→category mapping
|
|
works through the dispatcher."""
|
|
from src.local_decision import _DECISION_DESCRIPTIONS
|
|
|
|
second_key = list(_DECISION_DESCRIPTIONS.keys())[1]
|
|
|
|
captured_calls = []
|
|
|
|
def fake_post(*a, **k):
|
|
captured_calls.append((a, k))
|
|
return make_response(letter="B")
|
|
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=False),
|
|
)
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
monkeypatch.setattr(local_decision.requests, "post", fake_post)
|
|
|
|
result = dispatcher._classify_via_local_decision("sys", "test task")
|
|
|
|
assert result.task_category == second_key, \
|
|
f"Expected {second_key!r}, got {result.task_category!r}"
|
|
assert result.source == "classifier"
|
|
assert len(captured_calls) == 1
|
|
|
|
|
|
def test_tier_enabled_false_is_default(monkeypatch):
|
|
"""tier_enabled defaults to False so the normal local_decision path
|
|
never fires extra HTTP calls without explicit config."""
|
|
from config import LocalDecisionConfig
|
|
|
|
assert LocalDecisionConfig().tier_enabled is False
|
|
|
|
|
|
def test_unreachable_ollama_walks_the_cascade(monkeypatch):
|
|
"""When local_decision Ollama is unreachable, classify() returns a
|
|
degraded fallback Classification instead of raising (c06bc36)."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision", _decision_conf()
|
|
)
|
|
dispatcher.cfg.classifier.decision.base_url = "http://127.0.0.1:1"
|
|
monkeypatch.setattr(dispatcher, "_local_classifier_skip_reason", lambda: None)
|
|
monkeypatch.setattr(
|
|
local_decision.requests, "post", lambda *a, **k: (_ for _ in ()).throw(
|
|
requests.ConnectionError("Connection refused")
|
|
)
|
|
)
|
|
|
|
got = dispatcher.classify("refactor this function", None)
|
|
|
|
assert got.source == "fallback"
|
|
assert got.confidence == 0.0
|
|
assert got.required_context_tokens == 0
|
|
assert got.task_tier == dispatcher.cfg.classifier.fallback_tier
|
|
assert got.task_category == dispatcher.cfg.classifier.fallback_category
|
|
|
|
|
|
def test_local_decision_connection_error_opens_circuit_unmetered(monkeypatch):
|
|
"""Transport failure in unmetered local_decision opens the classifier
|
|
backoff and blocks subsequent calls until the circuit closes."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=False),
|
|
)
|
|
# Make metering off so we hit the unmetered path
|
|
monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", False)
|
|
monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True)
|
|
|
|
call_count = [0]
|
|
def fake_post(*a, **k):
|
|
call_count[0] += 1
|
|
raise requests.ConnectionError("Connection refused")
|
|
|
|
monkeypatch.setattr(local_decision.requests, "post", fake_post)
|
|
|
|
# First call — should record failure and return fallback
|
|
got1 = dispatcher.classify("refactor this function", None)
|
|
assert got1.source == "fallback"
|
|
assert dispatcher._last_classifier_failure > 0.0
|
|
assert dispatcher._classifier_backoff_active() is True
|
|
first_failure_at = dispatcher._last_classifier_failure
|
|
|
|
# Second call — should be SKIPPED (post called exactly once total)
|
|
got2 = dispatcher.classify("another task", None)
|
|
assert call_count[0] == 1, f"post called {call_count[0]} times, expected 1"
|
|
assert got2.source == "fallback"
|
|
|
|
# A skip must never re-stamp the clock, or the circuit stays open forever
|
|
# while traffic flows. Equality, not just "still set".
|
|
assert dispatcher._last_classifier_failure == first_failure_at
|
|
|
|
|
|
def test_local_decision_connection_error_opens_circuit_metered(monkeypatch):
|
|
"""Transport failure in metered local_decision opens the classifier
|
|
backoff and blocks subsequent calls until the circuit closes."""
|
|
call_count = [0]
|
|
|
|
def fake_nvidia(*a, **k):
|
|
return 100.0
|
|
|
|
def fake_post(*a, **k):
|
|
call_count[0] += 1
|
|
raise requests.ConnectionError("Connection refused")
|
|
|
|
def capture_log(*a, measurement=None, **k):
|
|
pass # just swallow — we don't care about the measurement
|
|
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=False),
|
|
)
|
|
monkeypatch.setattr(dispatcher.cfg.local_energy, "enabled", True)
|
|
monkeypatch.setattr(dispatcher.cfg.local_energy, "sample_interval_seconds", 0.01)
|
|
monkeypatch.setattr(dispatcher.cfg, "_sites_cache", {"classify": True})
|
|
monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True)
|
|
monkeypatch.setattr(dispatcher.local_energy, "sample_nvidia_smi", fake_nvidia)
|
|
monkeypatch.setattr(local_decision.requests, "post", fake_post)
|
|
monkeypatch.setattr(dispatcher, "_log_local_energy", capture_log)
|
|
|
|
# First call — should record failure and return fallback
|
|
got1 = dispatcher.classify("refactor this function", None)
|
|
assert got1.source == "fallback"
|
|
assert dispatcher._last_classifier_failure > 0.0
|
|
assert dispatcher._classifier_backoff_active() is True
|
|
first_failure_at = dispatcher._last_classifier_failure
|
|
|
|
# Second call — should be SKIPPED (post called exactly once total)
|
|
got2 = dispatcher.classify("another task", None)
|
|
assert call_count[0] == 1, f"post called {call_count[0]} times, expected 1"
|
|
assert got2.source == "fallback"
|
|
assert dispatcher._last_classifier_failure == first_failure_at
|
|
|
|
|
|
def test_local_decision_below_confidence_min_does_not_open_circuit(monkeypatch):
|
|
"""A verdict below classifier.decision.confidence_min is the dispatcher's
|
|
own RuntimeError, raised after a healthy answer. The endpoint is fine, so
|
|
the circuit must stay closed. (The coverage-floor path is the next test.)"""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=False),
|
|
)
|
|
monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True)
|
|
|
|
def fake_post(*a, **k):
|
|
# Four letters with equal mass: coverage is about 0.99 (well above
|
|
# coverage_min), but the winner holds only a quarter of it, so
|
|
# confidence is about 0.25, below confidence_min (0.5).
|
|
return SimpleNamespace(
|
|
json=lambda: {
|
|
"logprobs": [{
|
|
"top_logprobs": [
|
|
{"token": letter, "logprob": -1.4}
|
|
for letter in ("A", "B", "C", "D")
|
|
]
|
|
}],
|
|
},
|
|
raise_for_status=lambda: None,
|
|
)
|
|
|
|
monkeypatch.setattr(local_decision.requests, "post", fake_post)
|
|
|
|
got = dispatcher.classify("refactor this function", None)
|
|
assert got.source == "fallback" # cascaded on the confidence failure
|
|
|
|
assert dispatcher._last_classifier_failure == 0.0
|
|
assert dispatcher._classifier_backoff_active() is False
|
|
|
|
|
|
def test_local_decision_coverage_floor_does_not_open_circuit(monkeypatch):
|
|
"""parse_logprobs raising because the model put almost no mass on any
|
|
option letter (coverage below coverage_min) is also not a transport
|
|
failure — the circuit must stay closed."""
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision")
|
|
monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2)
|
|
monkeypatch.setattr(
|
|
dispatcher.cfg.classifier, "decision",
|
|
_decision_conf(confidence_min=0.5, tier_enabled=False),
|
|
)
|
|
monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True)
|
|
|
|
def fake_post(*a, **k):
|
|
# "Z" is not an option letter and A/B carry almost no mass, so total
|
|
# option mass (about 0.012) is below coverage_min (0.3) and
|
|
# parse_logprobs raises RuntimeError before any confidence is computed.
|
|
return SimpleNamespace(
|
|
json=lambda: {
|
|
"model": "qwen3.5:4b",
|
|
"message": {"role": "assistant", "content": "Z"},
|
|
"logprobs": [{
|
|
"top_logprobs": [
|
|
{"token": "Z", "logprob": -5.0}, # very low confidence
|
|
{"token": "A", "logprob": -5.1},
|
|
{"token": "B", "logprob": -5.2},
|
|
]
|
|
}],
|
|
},
|
|
raise_for_status=lambda: None,
|
|
)
|
|
|
|
monkeypatch.setattr(local_decision.requests, "post", fake_post)
|
|
|
|
# classify() should cascade to fallback because confidence < confidence_min
|
|
got = dispatcher.classify("refactor this function", None)
|
|
assert got.source == "fallback" # cascaded
|
|
|
|
# But _last_classifier_failure must be 0.0 — this was NOT a transport error
|
|
assert dispatcher._last_classifier_failure == 0.0
|
|
assert dispatcher._classifier_backoff_active() is False
|