"""Tests for iteration.py — spending a tier's retry budget. Two ideas under test. First, that a retry is matched to the failure: a truncated answer needs a bigger budget, not a different model, and a malformed one first retries the same model (cache-preserving) before escalating. Second, that declining to retry is a real outcome — every wasted attempt burns energy against a fixed kWh quota. """ import pytest from iteration import attempts_allowed, plan_retry, RetryPlan BY_TIER = {1: 0, 2: 1, 3: 2} # --- how much iteration a tier buys --------------------------------------- def test_tier_one_buys_no_retries(): # Cheap/simple work: one shot. Iterating on it costs more than it is worth. assert attempts_allowed(1, "batch", BY_TIER, 1) == 0 def test_higher_tiers_buy_more_attempts(): assert attempts_allowed(2, "batch", BY_TIER, 5) == 1 assert attempts_allowed(3, "batch", BY_TIER, 5) == 2 def test_interactive_work_is_capped_below_its_tier_budget(): # Every retry doubles time-to-answer, and in interactive use latency IS a # quality loss — a slow correct answer can be worth less than a fast one # the user can judge themselves. assert attempts_allowed(3, "interactive", BY_TIER, 1) == 1 assert attempts_allowed(3, "batch", BY_TIER, 1) == 2 def test_an_unknown_tier_buys_nothing(): assert attempts_allowed(9, "batch", BY_TIER, 5) == 0 # --- truncation: more budget, same model ---------------------------------- def test_truncation_retries_the_same_model_with_more_tokens(): # A different model would also run out. The budget is the problem. plan = plan_retry("truncated", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False) assert plan.model_id == "qwen3.6-35b" assert plan.max_tokens == 2000 assert plan.same_model is True def test_truncation_retry_respects_the_model_ceiling(): plan = plan_retry("truncated", "gemma-4-31b", 10000, 16384, [], False) assert plan.max_tokens == 16384 assert plan.same_model is True def test_no_retry_when_already_at_the_model_ceiling(): # Doubling would change nothing, and the attempt costs quota assert plan_retry("truncated", "gemma-4-31b", 16384, 16384, [], False) is None def test_a_client_chosen_cap_is_not_overridden(): # The caller explicitly asked for a short answer. Ignoring that would # override an instruction, and they may well want it short. assert plan_retry("truncated", "qwen3.6-35b", 40, 16384, [("kimi-k3", 65536)], True) is None def test_no_cap_set_escalates_to_a_model_that_can_emit_more(): # The model's own output ceiling is the wall, so retrying IT changes # nothing — but a roomier candidate might finish the answer. Without this # the truncation branch was unreachable in practice: via /v1 the cap is # either the client's (not ours to override) or absent. plan = plan_retry("truncated", "gemma-4-31b", None, 16384, [("kimi-k3", 65536)], False) assert plan.model_id == "kimi-k3" assert plan.same_model is False def test_no_roomier_candidate_means_no_retry(): plan = plan_retry("truncated", "deepseek-v4-flash", None, 65536, [("gemma-4-31b", 16384)], False) assert plan is None def test_candidates_with_unknown_ceilings_are_not_assumed_roomier(): # 11 of 19 catalog rows report no max_output_tokens; guessing they are # bigger would waste the attempt assert plan_retry("truncated", "gemma-4-31b", None, 16384, [("qwen3.6-35b", None)], False) is None # --- malformed: same-model first (cache-preserving), then escalate --------- def test_malformed_prefers_same_model_retry(): # More tokens will not help, but the SAME model may produce valid output # on a second try, and retrying the same model keeps the provider's prompt # cache hot. Same-model retry comes before escalation. plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536), ("gemma-4-31b", 16384)], False) assert plan.model_id == "qwen3.6-35b" # same-model, not kimi-k3 assert plan.max_tokens == 1000 # same budget assert plan.same_model is True # cache-preserving flag assert "retrying" in plan.detail assert "cache-preserving" in plan.detail def test_malformed_same_model_ignores_client_cap(): # The cap is irrelevant — the answer did not parse, it was not cut off. # Same-model retry still happens regardless of client cap. plan = plan_retry("malformed", "qwen3.6-35b", 40, 16384, [("kimi-k3", 65536)], True) assert plan.model_id == "qwen3.6-35b" # same-model, not kimi-k3 assert plan.same_model is True def test_malformed_retries_same_model_even_with_no_alternatives(): # No runners-up doesn't mean no retry: same-model retry is always worth # trying because it preserves the cache. plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [], False) assert plan.model_id == "qwen3.6-35b" assert plan.same_model is True def test_malformed_escalates_after_same_model_exhausted(): # Same-model retry exhausted; escalate to next candidate. plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536), ("gemma-4-31b", 16384)], False, retried_same_model=True) assert plan.model_id == "kimi-k3" assert plan.same_model is False assert "escalating" in plan.detail def test_malformed_exhausted_with_no_alternatives_declines(): # Same-model already tried, no candidates to escalate to. assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [], False, retried_same_model=True) is None def test_malformed_exhausted_escalation_blocked_by_large_prompt(): # Prompt is too large for the re-bill to be worth it, escalation gated. assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False, prompt_tokens=100000, max_rebill_prompt_tokens=65536, retried_same_model=True) is None def test_malformed_exhausted_escalation_allowed_at_boundary(): # At the exact boundary: prompt_tokens <= max_rebill_prompt_tokens. plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False, prompt_tokens=65536, max_rebill_prompt_tokens=65536, retried_same_model=True) assert plan.model_id == "kimi-k3" assert plan.same_model is False def test_malformed_exhausted_escalation_blocked_at_one_over(): # One token over the boundary: escalation blocked. assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False, prompt_tokens=65537, max_rebill_prompt_tokens=65536, retried_same_model=True) is None # --- truncation escalation also gated on prompt size ----------------------- def test_truncation_escalation_blocked_by_large_prompt(): # A roomier candidate exists, but the prompt is too large to justify a # cold re-bill. assert plan_retry("truncated", "gemma-4-31b", None, 16384, [("kimi-k3", 65536)], False, prompt_tokens=100000, max_rebill_prompt_tokens=65536) is None def test_truncation_escalation_allowed_when_prompt_small(): plan = plan_retry("truncated", "gemma-4-31b", None, 16384, [("kimi-k3", 65536)], False, prompt_tokens=64000, max_rebill_prompt_tokens=65536) assert plan.model_id == "kimi-k3" assert plan.same_model is False # --- prompt-size gate: zero means unlimited (backward compatible) ---------- def test_zero_threshold_allows_escalation(): # Default config: max_rebill_prompt_tokens=0 means no limit. # Escalation works regardless of prompt size. plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False, prompt_tokens=999999, retried_same_model=True) assert plan.model_id == "kimi-k3" assert plan.same_model is False def test_zero_threshold_allows_truncation_escalation(): plan = plan_retry("truncated", "gemma-4-31b", None, 16384, [("kimi-k3", 65536)], False, prompt_tokens=999999) assert plan.model_id == "kimi-k3" # --- what must NOT buy a retry -------------------------------------------- @pytest.mark.parametrize("verdict", ["ok", "unverifiable"]) def test_non_failures_buy_no_retry(verdict): # 'unverifiable' is most prose traffic. Retrying it would burn quota # across the majority of requests for no signal at all. assert plan_retry(verdict, "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False) is None