Adds prompt-size awareness to the retry budget (item 4.1 of
plans/token-waste-waves.md). Two coupled changes:
(A) Malformed same-model-first (cache-preserving)
plan_retry() now returns a same-model retry for malformed verdicts
(with same_model=True), keeping the provider prompt cache hot
(~0.92 hit rate vs ~0.35 on switch). Only when retried_same_model
is passed as True (second consecutive malformed) does it escalate
to runners_up[0]. The dispatcher loop tracks retried_same_model
and only drops alternatives[1:] on escalation, not on same-model
retry.
(B) Prompt-size gate on escalation
New parameter max_rebill_prompt_tokens: when prompt_tokens exceeds
this threshold, escalation (cache-destroying cold re-bill) is
suppressed. Same-model retries are always allowed because they
preserve the cache. 0 = unlimited (backward-compatible default).
Config surface:
- iteration.max_rebill_prompt_tokens (int, default 0) in config.yaml
and IterationConfig Pydantic model.
Dispatcher: threads decision.classification.required_context_tokens
as prompt_tokens and cfg.iteration.max_rebill_prompt_tokens as the
threshold into plan_retry(); tracks retried_same_model in the retry
loop.
Tests: 13 existing + 12 new covering:
- same-model malformed retry preferred over escalation
- same-model malformed works with no runners_up
- escalation after same_model exhausted
- escalation blocked/gated on prompt-size threshold (including
boundary and one-over)
- truncation escalation also gated (no change to same-model
truncation)
- zero threshold = unlimited (backward compat)
215 lines
8.8 KiB
Python
215 lines
8.8 KiB
Python
"""Tests for iteration.py — spending a tier's retry budget.
|
|
|
|
Two ideas under test. First, that a retry is matched to the failure: a
|
|
truncated answer needs a bigger budget, not a different model, and a malformed
|
|
one first retries the same model (cache-preserving) before escalating.
|
|
Second, that declining to retry is a real outcome — every wasted attempt
|
|
burns energy against a fixed kWh quota.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from iteration import attempts_allowed, plan_retry, RetryPlan
|
|
|
|
BY_TIER = {1: 0, 2: 1, 3: 2}
|
|
|
|
|
|
# --- how much iteration a tier buys ---------------------------------------
|
|
|
|
def test_tier_one_buys_no_retries():
|
|
# Cheap/simple work: one shot. Iterating on it costs more than it is worth.
|
|
assert attempts_allowed(1, "batch", BY_TIER, 1) == 0
|
|
|
|
|
|
def test_higher_tiers_buy_more_attempts():
|
|
assert attempts_allowed(2, "batch", BY_TIER, 5) == 1
|
|
assert attempts_allowed(3, "batch", BY_TIER, 5) == 2
|
|
|
|
|
|
def test_interactive_work_is_capped_below_its_tier_budget():
|
|
# Every retry doubles time-to-answer, and in interactive use latency IS a
|
|
# quality loss — a slow correct answer can be worth less than a fast one
|
|
# the user can judge themselves.
|
|
assert attempts_allowed(3, "interactive", BY_TIER, 1) == 1
|
|
assert attempts_allowed(3, "batch", BY_TIER, 1) == 2
|
|
|
|
|
|
def test_an_unknown_tier_buys_nothing():
|
|
assert attempts_allowed(9, "batch", BY_TIER, 5) == 0
|
|
|
|
|
|
# --- truncation: more budget, same model ----------------------------------
|
|
|
|
def test_truncation_retries_the_same_model_with_more_tokens():
|
|
# A different model would also run out. The budget is the problem.
|
|
plan = plan_retry("truncated", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False)
|
|
assert plan.model_id == "qwen3.6-35b"
|
|
assert plan.max_tokens == 2000
|
|
assert plan.same_model is True
|
|
|
|
|
|
def test_truncation_retry_respects_the_model_ceiling():
|
|
plan = plan_retry("truncated", "gemma-4-31b", 10000, 16384, [], False)
|
|
assert plan.max_tokens == 16384
|
|
assert plan.same_model is True
|
|
|
|
|
|
def test_no_retry_when_already_at_the_model_ceiling():
|
|
# Doubling would change nothing, and the attempt costs quota
|
|
assert plan_retry("truncated", "gemma-4-31b", 16384, 16384, [], False) is None
|
|
|
|
|
|
def test_a_client_chosen_cap_is_not_overridden():
|
|
# The caller explicitly asked for a short answer. Ignoring that would
|
|
# override an instruction, and they may well want it short.
|
|
assert plan_retry("truncated", "qwen3.6-35b", 40, 16384, [("kimi-k3", 65536)], True) is None
|
|
|
|
|
|
def test_no_cap_set_escalates_to_a_model_that_can_emit_more():
|
|
# The model's own output ceiling is the wall, so retrying IT changes
|
|
# nothing — but a roomier candidate might finish the answer. Without this
|
|
# the truncation branch was unreachable in practice: via /v1 the cap is
|
|
# either the client's (not ours to override) or absent.
|
|
plan = plan_retry("truncated", "gemma-4-31b", None, 16384, [("kimi-k3", 65536)], False)
|
|
assert plan.model_id == "kimi-k3"
|
|
assert plan.same_model is False
|
|
|
|
|
|
def test_no_roomier_candidate_means_no_retry():
|
|
plan = plan_retry("truncated", "deepseek-v4-flash", None, 65536,
|
|
[("gemma-4-31b", 16384)], False)
|
|
assert plan is None
|
|
|
|
|
|
def test_candidates_with_unknown_ceilings_are_not_assumed_roomier():
|
|
# 11 of 19 catalog rows report no max_output_tokens; guessing they are
|
|
# bigger would waste the attempt
|
|
assert plan_retry("truncated", "gemma-4-31b", None, 16384,
|
|
[("qwen3.6-35b", None)], False) is None
|
|
|
|
|
|
# --- malformed: same-model first (cache-preserving), then escalate ---------
|
|
|
|
def test_malformed_prefers_same_model_retry():
|
|
# More tokens will not help, but the SAME model may produce valid output
|
|
# on a second try, and retrying the same model keeps the provider's prompt
|
|
# cache hot. Same-model retry comes before escalation.
|
|
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
|
|
[("kimi-k3", 65536), ("gemma-4-31b", 16384)], False)
|
|
assert plan.model_id == "qwen3.6-35b" # same-model, not kimi-k3
|
|
assert plan.max_tokens == 1000 # same budget
|
|
assert plan.same_model is True # cache-preserving flag
|
|
assert "retrying" in plan.detail
|
|
assert "cache-preserving" in plan.detail
|
|
|
|
|
|
def test_malformed_same_model_ignores_client_cap():
|
|
# The cap is irrelevant — the answer did not parse, it was not cut off.
|
|
# Same-model retry still happens regardless of client cap.
|
|
plan = plan_retry("malformed", "qwen3.6-35b", 40, 16384,
|
|
[("kimi-k3", 65536)], True)
|
|
assert plan.model_id == "qwen3.6-35b" # same-model, not kimi-k3
|
|
assert plan.same_model is True
|
|
|
|
|
|
def test_malformed_retries_same_model_even_with_no_alternatives():
|
|
# No runners-up doesn't mean no retry: same-model retry is always worth
|
|
# trying because it preserves the cache.
|
|
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [], False)
|
|
assert plan.model_id == "qwen3.6-35b"
|
|
assert plan.same_model is True
|
|
|
|
|
|
def test_malformed_escalates_after_same_model_exhausted():
|
|
# Same-model retry exhausted; escalate to next candidate.
|
|
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
|
|
[("kimi-k3", 65536), ("gemma-4-31b", 16384)], False,
|
|
retried_same_model=True)
|
|
assert plan.model_id == "kimi-k3"
|
|
assert plan.same_model is False
|
|
assert "escalating" in plan.detail
|
|
|
|
|
|
def test_malformed_exhausted_with_no_alternatives_declines():
|
|
# Same-model already tried, no candidates to escalate to.
|
|
assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [], False,
|
|
retried_same_model=True) is None
|
|
|
|
|
|
def test_malformed_exhausted_escalation_blocked_by_large_prompt():
|
|
# Prompt is too large for the re-bill to be worth it, escalation gated.
|
|
assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
|
|
[("kimi-k3", 65536)], False,
|
|
prompt_tokens=100000,
|
|
max_rebill_prompt_tokens=65536,
|
|
retried_same_model=True) is None
|
|
|
|
|
|
def test_malformed_exhausted_escalation_allowed_at_boundary():
|
|
# At the exact boundary: prompt_tokens <= max_rebill_prompt_tokens.
|
|
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
|
|
[("kimi-k3", 65536)], False,
|
|
prompt_tokens=65536,
|
|
max_rebill_prompt_tokens=65536,
|
|
retried_same_model=True)
|
|
assert plan.model_id == "kimi-k3"
|
|
assert plan.same_model is False
|
|
|
|
|
|
def test_malformed_exhausted_escalation_blocked_at_one_over():
|
|
# One token over the boundary: escalation blocked.
|
|
assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
|
|
[("kimi-k3", 65536)], False,
|
|
prompt_tokens=65537,
|
|
max_rebill_prompt_tokens=65536,
|
|
retried_same_model=True) is None
|
|
|
|
|
|
# --- truncation escalation also gated on prompt size -----------------------
|
|
|
|
def test_truncation_escalation_blocked_by_large_prompt():
|
|
# A roomier candidate exists, but the prompt is too large to justify a
|
|
# cold re-bill.
|
|
assert plan_retry("truncated", "gemma-4-31b", None, 16384,
|
|
[("kimi-k3", 65536)], False,
|
|
prompt_tokens=100000,
|
|
max_rebill_prompt_tokens=65536) is None
|
|
|
|
|
|
def test_truncation_escalation_allowed_when_prompt_small():
|
|
plan = plan_retry("truncated", "gemma-4-31b", None, 16384,
|
|
[("kimi-k3", 65536)], False,
|
|
prompt_tokens=64000,
|
|
max_rebill_prompt_tokens=65536)
|
|
assert plan.model_id == "kimi-k3"
|
|
assert plan.same_model is False
|
|
|
|
|
|
# --- prompt-size gate: zero means unlimited (backward compatible) ----------
|
|
|
|
def test_zero_threshold_allows_escalation():
|
|
# Default config: max_rebill_prompt_tokens=0 means no limit.
|
|
# Escalation works regardless of prompt size.
|
|
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
|
|
[("kimi-k3", 65536)], False,
|
|
prompt_tokens=999999,
|
|
retried_same_model=True)
|
|
assert plan.model_id == "kimi-k3"
|
|
assert plan.same_model is False
|
|
|
|
|
|
def test_zero_threshold_allows_truncation_escalation():
|
|
plan = plan_retry("truncated", "gemma-4-31b", None, 16384,
|
|
[("kimi-k3", 65536)], False,
|
|
prompt_tokens=999999)
|
|
assert plan.model_id == "kimi-k3"
|
|
|
|
|
|
# --- what must NOT buy a retry --------------------------------------------
|
|
|
|
@pytest.mark.parametrize("verdict", ["ok", "unverifiable"])
|
|
def test_non_failures_buy_no_retry(verdict):
|
|
# 'unverifiable' is most prose traffic. Retrying it would burn quota
|
|
# across the majority of requests for no signal at all.
|
|
assert plan_retry(verdict, "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False) is None
|