From a6fc80c96ada6ab93c3a63edabcf97cabc0bdc44 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Fri, 2 Oct 2026 23:04:09 -0400 Subject: [PATCH] test(dispatcher): tighten the local_decision backoff tests The review of the backoff fix found two weak spots, both now closed: - The circuit tests asserted only that _last_classifier_failure was still set after the skipped second call. A skip that re-stamped the clock (holding the circuit open for as long as traffic flows) would have passed. Both tests now assert the timestamp is unchanged. - test_local_decision_low_confidence_does_not_open_circuit tripped the parse_logprobs coverage floor, not confidence_min, so the dispatcher's own confidence raise had no test. Split in two: a new test that actually lands below confidence_min (four equal letters, coverage 0.99, confidence 0.25), and the old one renamed and re-documented as the coverage-floor case. Mutation-checked in a scratch export: a skip that records, a confidence raise that records, and recording on any exception each fail the intended tests. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_01KkCGRantZsSwmcFpet6FTa --- tests/test_classifier_modes_dispatch.py | 57 ++++++++++++++++++++++--- 1 file changed, 50 insertions(+), 7 deletions(-) diff --git a/tests/test_classifier_modes_dispatch.py b/tests/test_classifier_modes_dispatch.py index 5ea04ff..4d8a552 100644 --- a/tests/test_classifier_modes_dispatch.py +++ b/tests/test_classifier_modes_dispatch.py @@ -924,14 +924,16 @@ def test_local_decision_connection_error_opens_circuit_unmetered(monkeypatch): assert got1.source == "fallback" assert dispatcher._last_classifier_failure > 0.0 assert dispatcher._classifier_backoff_active() is True + first_failure_at = dispatcher._last_classifier_failure # Second call — should be SKIPPED (post called exactly once total) got2 = dispatcher.classify("another task", None) assert call_count[0] == 1, f"post called {call_count[0]} times, expected 1" assert got2.source == "fallback" - # _last_classifier_failure unchanged by second call - assert dispatcher._last_classifier_failure > 0.0 # still set from first call + # A skip must never re-stamp the clock, or the circuit stays open forever + # while traffic flows. Equality, not just "still set". + assert dispatcher._last_classifier_failure == first_failure_at def test_local_decision_connection_error_opens_circuit_metered(monkeypatch): @@ -968,16 +970,19 @@ def test_local_decision_connection_error_opens_circuit_metered(monkeypatch): assert got1.source == "fallback" assert dispatcher._last_classifier_failure > 0.0 assert dispatcher._classifier_backoff_active() is True + first_failure_at = dispatcher._last_classifier_failure # Second call — should be SKIPPED (post called exactly once total) got2 = dispatcher.classify("another task", None) assert call_count[0] == 1, f"post called {call_count[0]} times, expected 1" assert got2.source == "fallback" + assert dispatcher._last_classifier_failure == first_failure_at -def test_local_decision_low_confidence_does_not_open_circuit(monkeypatch): - """A successful classify with low confidence is a real answer, not a - transport failure — the circuit must stay closed.""" +def test_local_decision_below_confidence_min_does_not_open_circuit(monkeypatch): + """A verdict below classifier.decision.confidence_min is the dispatcher's + own RuntimeError, raised after a healthy answer. The endpoint is fine, so + the circuit must stay closed. (The coverage-floor path is the next test.)""" monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) monkeypatch.setattr( @@ -987,8 +992,46 @@ def test_local_decision_low_confidence_does_not_open_circuit(monkeypatch): monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True) def fake_post(*a, **k): - # Return a response where the winning token has very low logprob - # → confidence below 0.5 → below confidence_min → RuntimeError + # Four letters with equal mass: coverage is about 0.99 (well above + # coverage_min), but the winner holds only a quarter of it, so + # confidence is about 0.25, below confidence_min (0.5). + return SimpleNamespace( + json=lambda: { + "logprobs": [{ + "top_logprobs": [ + {"token": letter, "logprob": -1.4} + for letter in ("A", "B", "C", "D") + ] + }], + }, + raise_for_status=lambda: None, + ) + + monkeypatch.setattr(local_decision.requests, "post", fake_post) + + got = dispatcher.classify("refactor this function", None) + assert got.source == "fallback" # cascaded on the confidence failure + + assert dispatcher._last_classifier_failure == 0.0 + assert dispatcher._classifier_backoff_active() is False + + +def test_local_decision_coverage_floor_does_not_open_circuit(monkeypatch): + """parse_logprobs raising because the model put almost no mass on any + option letter (coverage below coverage_min) is also not a transport + failure — the circuit must stay closed.""" + monkeypatch.setattr(dispatcher.cfg.classifier, "mode", "local_decision") + monkeypatch.setattr(dispatcher.cfg.classifier, "fallback_tier", 2) + monkeypatch.setattr( + dispatcher.cfg.classifier, "decision", + _decision_conf(confidence_min=0.5, tier_enabled=False), + ) + monkeypatch.setattr(dispatcher.cfg.local_compute, "enabled", True) + + def fake_post(*a, **k): + # "Z" is not an option letter and A/B carry almost no mass, so total + # option mass (about 0.012) is below coverage_min (0.3) and + # parse_logprobs raises RuntimeError before any confidence is computed. return SimpleNamespace( json=lambda: { "model": "qwen3.5:4b", -- 2.49.1