12 of 30 active OpenRouter rows had an effective context window of 0 and
had never been selected once across 23,000+ decisions -- including
moonshotai/kimi-k3 and z-ai/glm-5.3 at 1M advertised context.
The two providers do not mean the same thing by max_output_tokens.
NeuralWatt reports a genuine per-request output cap: at most 0.16 of
context across all 19 rows. OpenRouter reports
top_provider.max_completion_tokens, which is a ceiling on what you may
ASK for -- 0.8-0.9 of context on a dozen rows.
effective_context_window subtracted it whole:
int(1048576 * 0.75) - 943718 == -157286
and max(_, 0) turned that into 0, which routing.py's context filter
reads as "fits nothing". Nothing warned, because a row that is never
eligible never rejects anything either -- neither the capability-ceiling
detector nor the rejection detector can see a candidate that silently
fails a hard filter.
context.max_output_reserve_fraction (0.5) caps the catalog-derived
reserve at a fraction of the usable window. A per_model_overrides
reserve is a measurement rather than a parsed field, so it is still used
as written.
Simulated over the live catalog: exactly those 12 rows recover, plus a
7% nudge on deepseek/deepseek-v3.2 (57344 -> 61440). No NeuralWatt row
moves.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
249 lines
9.1 KiB
Python
249 lines
9.1 KiB
Python
"""Tests for poller.py's pure catalog-parsing helpers.
|
|
|
|
The NeuralWatt catalog encodes serving class in the model id and access
|
|
gating only in prose, so both are parsed rather than read from a field.
|
|
Cases below are taken from the live catalog.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from pathlib import Path
|
|
|
|
from poller import parse_access_level, parse_base_model_id, parse_serving_class
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
|
|
|
|
# --- serving class --------------------------------------------------------
|
|
|
|
@pytest.mark.parametrize(
|
|
"model_id,expected",
|
|
[
|
|
# base rows take the schema defaults
|
|
("gemma-4-31b", ("standard", "default", "full")),
|
|
("kimi-k3", ("standard", "default", "full")),
|
|
# one dimension at a time
|
|
("kimi-k3-fast", ("standard", "reduced", "full")),
|
|
("kimi-k3-flex", ("flex", "default", "full")),
|
|
("glm-5.2-short", ("standard", "default", "short")),
|
|
# the dimensions are orthogonal and combine freely
|
|
("glm-5.2-short-fast", ("standard", "reduced", "short")),
|
|
("glm-5.2-short-flex", ("flex", "default", "short")),
|
|
("glm-5.2-short-fast-flex", ("flex", "reduced", "short")),
|
|
],
|
|
)
|
|
def test_parse_serving_class(model_id, expected):
|
|
assert parse_serving_class(model_id) == expected
|
|
|
|
|
|
def test_flash_is_not_read_as_fast():
|
|
# Given: a base model whose name ends in 'flash' — a near-miss for the
|
|
# '-fast' suffix that a substring match would misclassify, wrongly
|
|
# demoting the cheapest model in the catalog out of tier 1
|
|
# Then: whole-segment matching leaves it a plain standard row
|
|
assert parse_serving_class("deepseek-v4-flash") == ("standard", "default", "full")
|
|
assert parse_serving_class("deepseek-v4-flash-flex") == ("flex", "default", "full")
|
|
|
|
|
|
def test_namespaced_and_mixed_case_ids():
|
|
# Given: the HF-style duplicate row, which carries a namespace and caps
|
|
assert parse_serving_class("deepseek-ai/DeepSeek-V4-Flash") == (
|
|
"standard",
|
|
"default",
|
|
"full",
|
|
)
|
|
|
|
|
|
def test_suffix_only_id_is_not_stripped_to_nothing():
|
|
# Given: a degenerate id consisting solely of a suffix token, the loop
|
|
# must leave at least one segment rather than consuming the whole id
|
|
assert parse_serving_class("fast") == ("standard", "default", "full")
|
|
|
|
|
|
# --- access level ---------------------------------------------------------
|
|
|
|
@pytest.mark.parametrize(
|
|
"display_name,description,expected",
|
|
[
|
|
("GLM-5.2 (short)", "... Private preview (grant-gated).", "preview"),
|
|
("GLM-5.2", "Private GLM-5.2 test canary", "canary"),
|
|
# gating can appear in the display name rather than the description
|
|
("DeepSeek V4 Flash 0731 (Canary)", "1M context window.", "canary"),
|
|
("Gemma 4 31B", "Google Gemma 4 31B — multimodal with tool calling.", "public"),
|
|
(None, None, "public"),
|
|
],
|
|
)
|
|
def test_parse_access_level(display_name, description, expected):
|
|
assert parse_access_level(display_name, description) == expected
|
|
|
|
|
|
def test_preview_takes_precedence_over_canary():
|
|
# Given: prose mentioning both, the more restrictive label wins
|
|
assert parse_access_level("x", "Private preview (grant-gated) canary") == "preview"
|
|
|
|
|
|
# --- model family ---------------------------------------------------------
|
|
|
|
@pytest.mark.parametrize(
|
|
"model_id,expected",
|
|
[
|
|
("gemma-4-31b", "gemma-4-31b"),
|
|
("kimi-k3", "kimi-k3"),
|
|
("kimi-k3-fast", "kimi-k3"),
|
|
("kimi-k3-flex", "kimi-k3"),
|
|
# every serving dimension collapses, in any combination
|
|
("glm-5.2-short-fast-flex", "glm-5.2"),
|
|
("glm-5.2-short", "glm-5.2"),
|
|
# the HF-style duplicate row folds into the same family
|
|
("deepseek-ai/DeepSeek-V4-Flash", "deepseek-v4-flash"),
|
|
# ...without 'flash' being mistaken for the '-fast' suffix
|
|
("deepseek-v4-flash", "deepseek-v4-flash"),
|
|
("deepseek-v4-flash-flex", "deepseek-v4-flash"),
|
|
],
|
|
)
|
|
def test_parse_base_model_id(model_id, expected):
|
|
assert parse_base_model_id(model_id) == expected
|
|
|
|
|
|
def test_family_never_collapses_to_empty():
|
|
# A degenerate id made only of suffix tokens must keep a segment, or every
|
|
# such row would share one meaningless family
|
|
assert parse_base_model_id("flex") == "flex"
|
|
|
|
|
|
# --- per-model context overrides ------------------------------------------
|
|
#
|
|
# The knob validated and was never read, while config.yaml shipped a worked
|
|
# example for it. Anyone who followed that example got silence.
|
|
|
|
def _row(**overrides):
|
|
from poller import ModelRow
|
|
|
|
fields = dict(
|
|
model_id="qwen3.6-35b", provider="neuralwatt", base_model_id="qwen3.6-35b",
|
|
display_name=None, cost_per_1m_prompt=None, cost_per_1m_completion=None,
|
|
cost_per_1m_prompt_cached=None, context_window=200_000,
|
|
max_output_tokens=None, supports_tools=True, supports_json_mode=True,
|
|
supports_vision=False, supports_reasoning=True,
|
|
reasoning_default_enabled=True, latency_class="standard",
|
|
reasoning_mode="default", context_variant="full", access_level="public",
|
|
pricing_tbd=False, deprecated=False,
|
|
)
|
|
fields.update(overrides)
|
|
return ModelRow(**fields)
|
|
|
|
|
|
def _cfg(overrides=None):
|
|
import copy
|
|
|
|
import yaml
|
|
|
|
from config import RouterConfig
|
|
|
|
with open(ROOT / "config" / "config.yaml") as fh:
|
|
raw = copy.deepcopy(yaml.safe_load(fh))
|
|
raw["context"]["per_model_overrides"] = overrides or {}
|
|
return RouterConfig(**raw)
|
|
|
|
|
|
def test_the_global_safety_factor_applies_without_an_override():
|
|
# 200000 * 0.75 - 4096
|
|
assert _row().effective_context_window(_cfg()) == 145_904
|
|
|
|
|
|
def test_a_per_model_safety_factor_is_actually_applied():
|
|
cfg = _cfg({"qwen3.6-35b": {"safety_factor": 0.85}})
|
|
|
|
# 200000 * 0.85 - 4096
|
|
assert _row().effective_context_window(cfg) == 165_904
|
|
|
|
|
|
def test_an_override_for_another_model_does_not_leak():
|
|
cfg = _cfg({"kimi-k3": {"safety_factor": 0.85}})
|
|
|
|
assert _row().effective_context_window(cfg) == 145_904
|
|
|
|
|
|
def test_a_zero_output_reserve_means_zero_not_unset():
|
|
"""Present-but-falsy: `or` would silently fall through to 4096."""
|
|
cfg = _cfg({"qwen3.6-35b": {"output_reserve_tokens": 0}})
|
|
|
|
assert _row().effective_context_window(cfg) == 150_000
|
|
|
|
|
|
def test_an_override_reserve_beats_the_models_advertised_ceiling():
|
|
cfg = _cfg({"qwen3.6-35b": {"output_reserve_tokens": 8192}})
|
|
|
|
assert _row(max_output_tokens=16384).effective_context_window(cfg) == 141_808
|
|
|
|
|
|
def test_each_key_falls_back_independently():
|
|
cfg = _cfg({"qwen3.6-35b": {"safety_factor": 0.85}})
|
|
|
|
# factor overridden, reserve still the global default
|
|
assert _row().effective_context_window(cfg) == 165_904
|
|
|
|
|
|
def test_a_typo_inside_an_override_is_an_error():
|
|
"""The point of typing it: StrictModel has to reach inside the block."""
|
|
with pytest.raises(Exception, match="safety_factr|Extra inputs"):
|
|
_cfg({"qwen3.6-35b": {"safety_factr": 0.85}})
|
|
|
|
|
|
def test_an_out_of_range_override_is_an_error():
|
|
with pytest.raises(Exception, match="safety_factor"):
|
|
_cfg({"qwen3.6-35b": {"safety_factor": 1.5}})
|
|
|
|
|
|
# --- the reserve ceiling ---------------------------------------------------
|
|
#
|
|
# Live catalog, 2026-09-08: 12 of 30 active OpenRouter rows had an effective
|
|
# context of 0 and had never been selected across 23,000+ decisions. The two
|
|
# providers do not mean the same thing by max_output_tokens -- NeuralWatt's
|
|
# is a real per-request cap (<=0.16 of context on all 19 rows), OpenRouter's
|
|
# is top_provider.max_completion_tokens, "the most you may ask for", which
|
|
# on a dozen rows is 0.8-0.9 of context.
|
|
|
|
def test_a_catalog_reserve_cannot_consume_the_whole_window():
|
|
"""The bug verbatim: kimi-k3 at 1,048,576 context reporting a 943,718
|
|
max_completion_tokens. 0.75 * 1048576 == 786432, minus 943718 is
|
|
negative, and max(_, 0) turned that into a silently unroutable row."""
|
|
row = _row(context_window=1_048_576, max_output_tokens=943_718)
|
|
|
|
# 786432, halved by max_output_reserve_fraction: 0.5
|
|
assert row.effective_context_window(_cfg()) == 393_216
|
|
|
|
|
|
def test_a_reserve_below_the_ceiling_is_still_subtracted_whole():
|
|
"""Every NeuralWatt row sits here, so the fix must not move any of them.
|
|
glm-5.2-short: 199,984 context, 32,000 max output, 117,988 before."""
|
|
row = _row(context_window=199_984, max_output_tokens=32_000)
|
|
|
|
assert row.effective_context_window(_cfg()) == 117_988
|
|
|
|
|
|
def test_the_ceiling_does_not_override_a_measured_reserve():
|
|
"""A per_model_overrides reserve is someone's measurement, not a field
|
|
parsed out of a provider catalog, so it is used as written even when it
|
|
exceeds the fraction."""
|
|
cfg = _cfg({"qwen3.6-35b": {"output_reserve_tokens": 140_000}})
|
|
|
|
# 150000 - 140000, not 150000 - 75000
|
|
assert _row().effective_context_window(cfg) == 10_000
|
|
|
|
|
|
def test_a_reserve_fraction_of_one_is_refused():
|
|
"""1.0 permits exactly the reserve that produced the incident."""
|
|
import copy
|
|
|
|
import yaml
|
|
|
|
from config import RouterConfig
|
|
|
|
with open(ROOT / "config" / "config.yaml") as fh:
|
|
raw = copy.deepcopy(yaml.safe_load(fh))
|
|
raw["context"]["max_output_reserve_fraction"] = 1.0
|
|
|
|
with pytest.raises(Exception, match="max_output_reserve_fraction"):
|
|
RouterConfig(**raw)
|