Files
6krrt/tests/test_iteration.py
adlee-was-taken c0b3175730 feat(iteration): budget retries in prompt re-bills; prefer cache-preserving same-model retry on malformed
Adds prompt-size awareness to the retry budget (item 4.1 of
plans/token-waste-waves.md). Two coupled changes:

(A) Malformed same-model-first (cache-preserving)
    plan_retry() now returns a same-model retry for malformed verdicts
    (with same_model=True), keeping the provider prompt cache hot
    (~0.92 hit rate vs ~0.35 on switch). Only when retried_same_model
    is passed as True (second consecutive malformed) does it escalate
    to runners_up[0]. The dispatcher loop tracks retried_same_model
    and only drops alternatives[1:] on escalation, not on same-model
    retry.

(B) Prompt-size gate on escalation
    New parameter max_rebill_prompt_tokens: when prompt_tokens exceeds
    this threshold, escalation (cache-destroying cold re-bill) is
    suppressed. Same-model retries are always allowed because they
    preserve the cache. 0 = unlimited (backward-compatible default).

Config surface:
  - iteration.max_rebill_prompt_tokens (int, default 0) in config.yaml
    and IterationConfig Pydantic model.

Dispatcher: threads decision.classification.required_context_tokens
as prompt_tokens and cfg.iteration.max_rebill_prompt_tokens as the
threshold into plan_retry(); tracks retried_same_model in the retry
loop.

Tests: 13 existing + 12 new covering:
  - same-model malformed retry preferred over escalation
  - same-model malformed works with no runners_up
  - escalation after same_model exhausted
  - escalation blocked/gated on prompt-size threshold (including
    boundary and one-over)
  - truncation escalation also gated (no change to same-model
    truncation)
  - zero threshold = unlimited (backward compat)
2026-09-18 00:32:55 -04:00

215 lines
8.8 KiB
Python

"""Tests for iteration.py — spending a tier's retry budget.
Two ideas under test. First, that a retry is matched to the failure: a
truncated answer needs a bigger budget, not a different model, and a malformed
one first retries the same model (cache-preserving) before escalating.
Second, that declining to retry is a real outcome — every wasted attempt
burns energy against a fixed kWh quota.
"""
import pytest
from iteration import attempts_allowed, plan_retry, RetryPlan
BY_TIER = {1: 0, 2: 1, 3: 2}
# --- how much iteration a tier buys ---------------------------------------
def test_tier_one_buys_no_retries():
# Cheap/simple work: one shot. Iterating on it costs more than it is worth.
assert attempts_allowed(1, "batch", BY_TIER, 1) == 0
def test_higher_tiers_buy_more_attempts():
assert attempts_allowed(2, "batch", BY_TIER, 5) == 1
assert attempts_allowed(3, "batch", BY_TIER, 5) == 2
def test_interactive_work_is_capped_below_its_tier_budget():
# Every retry doubles time-to-answer, and in interactive use latency IS a
# quality loss — a slow correct answer can be worth less than a fast one
# the user can judge themselves.
assert attempts_allowed(3, "interactive", BY_TIER, 1) == 1
assert attempts_allowed(3, "batch", BY_TIER, 1) == 2
def test_an_unknown_tier_buys_nothing():
assert attempts_allowed(9, "batch", BY_TIER, 5) == 0
# --- truncation: more budget, same model ----------------------------------
def test_truncation_retries_the_same_model_with_more_tokens():
# A different model would also run out. The budget is the problem.
plan = plan_retry("truncated", "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False)
assert plan.model_id == "qwen3.6-35b"
assert plan.max_tokens == 2000
assert plan.same_model is True
def test_truncation_retry_respects_the_model_ceiling():
plan = plan_retry("truncated", "gemma-4-31b", 10000, 16384, [], False)
assert plan.max_tokens == 16384
assert plan.same_model is True
def test_no_retry_when_already_at_the_model_ceiling():
# Doubling would change nothing, and the attempt costs quota
assert plan_retry("truncated", "gemma-4-31b", 16384, 16384, [], False) is None
def test_a_client_chosen_cap_is_not_overridden():
# The caller explicitly asked for a short answer. Ignoring that would
# override an instruction, and they may well want it short.
assert plan_retry("truncated", "qwen3.6-35b", 40, 16384, [("kimi-k3", 65536)], True) is None
def test_no_cap_set_escalates_to_a_model_that_can_emit_more():
# The model's own output ceiling is the wall, so retrying IT changes
# nothing — but a roomier candidate might finish the answer. Without this
# the truncation branch was unreachable in practice: via /v1 the cap is
# either the client's (not ours to override) or absent.
plan = plan_retry("truncated", "gemma-4-31b", None, 16384, [("kimi-k3", 65536)], False)
assert plan.model_id == "kimi-k3"
assert plan.same_model is False
def test_no_roomier_candidate_means_no_retry():
plan = plan_retry("truncated", "deepseek-v4-flash", None, 65536,
[("gemma-4-31b", 16384)], False)
assert plan is None
def test_candidates_with_unknown_ceilings_are_not_assumed_roomier():
# 11 of 19 catalog rows report no max_output_tokens; guessing they are
# bigger would waste the attempt
assert plan_retry("truncated", "gemma-4-31b", None, 16384,
[("qwen3.6-35b", None)], False) is None
# --- malformed: same-model first (cache-preserving), then escalate ---------
def test_malformed_prefers_same_model_retry():
# More tokens will not help, but the SAME model may produce valid output
# on a second try, and retrying the same model keeps the provider's prompt
# cache hot. Same-model retry comes before escalation.
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
[("kimi-k3", 65536), ("gemma-4-31b", 16384)], False)
assert plan.model_id == "qwen3.6-35b" # same-model, not kimi-k3
assert plan.max_tokens == 1000 # same budget
assert plan.same_model is True # cache-preserving flag
assert "retrying" in plan.detail
assert "cache-preserving" in plan.detail
def test_malformed_same_model_ignores_client_cap():
# The cap is irrelevant — the answer did not parse, it was not cut off.
# Same-model retry still happens regardless of client cap.
plan = plan_retry("malformed", "qwen3.6-35b", 40, 16384,
[("kimi-k3", 65536)], True)
assert plan.model_id == "qwen3.6-35b" # same-model, not kimi-k3
assert plan.same_model is True
def test_malformed_retries_same_model_even_with_no_alternatives():
# No runners-up doesn't mean no retry: same-model retry is always worth
# trying because it preserves the cache.
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [], False)
assert plan.model_id == "qwen3.6-35b"
assert plan.same_model is True
def test_malformed_escalates_after_same_model_exhausted():
# Same-model retry exhausted; escalate to next candidate.
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
[("kimi-k3", 65536), ("gemma-4-31b", 16384)], False,
retried_same_model=True)
assert plan.model_id == "kimi-k3"
assert plan.same_model is False
assert "escalating" in plan.detail
def test_malformed_exhausted_with_no_alternatives_declines():
# Same-model already tried, no candidates to escalate to.
assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384, [], False,
retried_same_model=True) is None
def test_malformed_exhausted_escalation_blocked_by_large_prompt():
# Prompt is too large for the re-bill to be worth it, escalation gated.
assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
[("kimi-k3", 65536)], False,
prompt_tokens=100000,
max_rebill_prompt_tokens=65536,
retried_same_model=True) is None
def test_malformed_exhausted_escalation_allowed_at_boundary():
# At the exact boundary: prompt_tokens <= max_rebill_prompt_tokens.
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
[("kimi-k3", 65536)], False,
prompt_tokens=65536,
max_rebill_prompt_tokens=65536,
retried_same_model=True)
assert plan.model_id == "kimi-k3"
assert plan.same_model is False
def test_malformed_exhausted_escalation_blocked_at_one_over():
# One token over the boundary: escalation blocked.
assert plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
[("kimi-k3", 65536)], False,
prompt_tokens=65537,
max_rebill_prompt_tokens=65536,
retried_same_model=True) is None
# --- truncation escalation also gated on prompt size -----------------------
def test_truncation_escalation_blocked_by_large_prompt():
# A roomier candidate exists, but the prompt is too large to justify a
# cold re-bill.
assert plan_retry("truncated", "gemma-4-31b", None, 16384,
[("kimi-k3", 65536)], False,
prompt_tokens=100000,
max_rebill_prompt_tokens=65536) is None
def test_truncation_escalation_allowed_when_prompt_small():
plan = plan_retry("truncated", "gemma-4-31b", None, 16384,
[("kimi-k3", 65536)], False,
prompt_tokens=64000,
max_rebill_prompt_tokens=65536)
assert plan.model_id == "kimi-k3"
assert plan.same_model is False
# --- prompt-size gate: zero means unlimited (backward compatible) ----------
def test_zero_threshold_allows_escalation():
# Default config: max_rebill_prompt_tokens=0 means no limit.
# Escalation works regardless of prompt size.
plan = plan_retry("malformed", "qwen3.6-35b", 1000, 16384,
[("kimi-k3", 65536)], False,
prompt_tokens=999999,
retried_same_model=True)
assert plan.model_id == "kimi-k3"
assert plan.same_model is False
def test_zero_threshold_allows_truncation_escalation():
plan = plan_retry("truncated", "gemma-4-31b", None, 16384,
[("kimi-k3", 65536)], False,
prompt_tokens=999999)
assert plan.model_id == "kimi-k3"
# --- what must NOT buy a retry --------------------------------------------
@pytest.mark.parametrize("verdict", ["ok", "unverifiable"])
def test_non_failures_buy_no_retry(verdict):
# 'unverifiable' is most prose traffic. Retrying it would burn quota
# across the majority of requests for no signal at all.
assert plan_retry(verdict, "qwen3.6-35b", 1000, 16384, [("kimi-k3", 65536)], False) is None