"""Pure candidate selection and ranking for the local LLM model router. Like ``scoring.py`` and ``tiering.py``, this module is free of I/O: rows come in as dicts, thresholds come in as arguments. ``dispatcher.py`` owns the DB reads and the provider call. Two stages, in order: 1. **Hard filters** (``select_candidates``) — disqualify outright. A model that cannot hold the context, is below the required tier, is stale, or is not actually reachable by this account is not a cost tradeoff to be weighed; it is not a candidate at all (design doc §3.2, §4). 2. **Ranking** (``rank_candidates``) — order what survives by quality, with cost breaking ties inside ``quality_tolerance``. NOT a weighted blend: that was measured to be nearly inert, since moving the cost weight from 0.4 to zero changed the winner in only 2 of 6 categories. The flex filter is the one that is easy to get wrong. NeuralWatt's ``-flex`` rows are the same weights at the same advertised price, so on every scored dimension they tie exactly with their standard sibling — but they are "held server-side during peak until a capacity gap opens". Left to the scorer, a coin flip decides whether an interactive request waits out a capacity gap. Latency tolerance is therefore a filter, not a weight. """ from __future__ import annotations from collections.abc import Sequence from scoring import cost_score, proficiency_score INTERACTIVE = "interactive" BATCH = "batch" def capability_gate_reason( row: dict, *, require_vision: bool = False, require_json_mode: bool = False, ) -> str | None: """Same fail-closed rule ``rejection_reason`` uses for vision/json-mode, pulled out so a pinned-model check can reuse it without re-implementing it — see dispatcher._check_pinned_capabilities. Returns a reason string when the gate fails, or None when the row passes. Reason order: vision(unknown), vision(unsupported), json_mode(unknown), json_mode(unsupported). """ if require_vision: vision = row.get("supports_vision") if vision is None: return "vision(unknown)" if not vision: return "vision(unsupported)" if require_json_mode: jm = row.get("supports_json_mode") if jm is None: return "json_mode(unknown)" if not jm: return "json_mode(unsupported)" return None def rejection_reason( row: dict, *, required_context_tokens: int, required_tier: int, latency_tolerance: str, allowed_access_levels: Sequence[str], exclude_stale: bool, exclude_deprecated: bool, min_tool_proficiency: float | None = None, require_vision: bool = False, require_json_mode: bool = False, ) -> str | None: """Why this row is not a candidate, or None if it is one. Phrased as the reason rather than a bool so the debug log can say WHICH filter dropped a model and with what numbers, without a second implementation of these rules drifting away from this one. Reasons are single tokens so a log line needs no quoting. A NULL ``effective_context_window`` fails the context filter: an unknown window cannot be shown to be large enough, and silently truncating mid-task is worse than routing elsewhere. ``min_tool_proficiency`` is set when the REQUEST carries tool definitions, and is a capability requirement rather than a preference — which is why it is a filter and not a term in the ranking. The distinction that matters: this does not ask whether the task is "agentic". It asks whether the model can be trusted with tools that are on the table. Those are different questions, and the measured failure is the second one — `deepseek-v4-flash` scores 0.33 here, and the recorded case is a *non*-agentic prompt ("it is 1:20pm, my meeting is at 3pm, how many minutes?") where it called two tools instead of subtracting. A model that over-reaches for tools is a hazard on every request where tools exist, not only on the ones a classifier would label agentic. That also sidesteps the classifier entirely, which is the point: asked to identify six unambiguous tool-use prompts, the local models managed 2/6 and 1/6. Whether tools are present is stated in the request body. Reading it is exact and free; inferring it is neither. """ eff_ctx = row.get("effective_context_window") if eff_ctx is None: return "context(unknown)" if eff_ctx < required_context_tokens: return f"context({eff_ctx}<{required_context_tokens})" tier = row.get("tier") if tier is None: return "tier(unknown)" if tier < required_tier: return f"tier({tier}<{required_tier})" availability = row.get("availability") if exclude_stale and availability == "stale": return "stale" if exclude_deprecated and (availability == "deprecated" or row.get("deprecated")): return "deprecated" access_level = row.get("access_level", "public") if access_level not in allowed_access_levels: return f"access_level({access_level})" # Flex rows may be queued behind a capacity gap; only batch work opts in. if latency_tolerance == INTERACTIVE and row.get("latency_class") == "flex": return "latency_class(flex)" # Capability flags fail CLOSED on unknown, deliberately unlike the tool # gate just below. A proficiency MEASUREMENT with no evidence is "unproven, # not bad" -- admitting it loses nothing. But a capability FLAG that is # absent means "cannot confirm the capability", and routing a request that # needs vision or JSON mode to a model that might lack it is a guaranteed # provider 400. Same precedent as the context(unknown) gate above. gate = capability_gate_reason( row, require_vision=require_vision, require_json_mode=require_json_mode, ) if gate is not None: return gate if min_tool_proficiency is not None: tool_score = row.get("tool_proficiency") # Absent evidence does not disqualify, consistently with the tier-1 # context gate: a model nobody has evaluated for tool use yet is # unproven, not proven bad. Only a measured score below the bar drops # a candidate. if tool_score is not None and tool_score < min_tool_proficiency: return f"tool_proficiency({tool_score:g}<{min_tool_proficiency:g})" return None def is_eligible(row: dict, **filters) -> bool: """Whether a model row survives every hard filter. A thin wrapper so there is exactly one copy of the rules; see ``rejection_reason`` for what they are and why. """ return rejection_reason(row, **filters) is None def select_candidates( rows: Sequence[dict], *, required_context_tokens: int, required_tier: int, latency_tolerance: str, allowed_access_levels: Sequence[str], exclude_stale: bool, exclude_deprecated: bool, min_tool_proficiency: float | None = None, require_vision: bool = False, require_json_mode: bool = False, ) -> list[dict]: """Apply every hard filter, preserving input order.""" return [ row for row in rows if is_eligible( row, required_context_tokens=required_context_tokens, required_tier=required_tier, latency_tolerance=latency_tolerance, allowed_access_levels=allowed_access_levels, exclude_stale=exclude_stale, exclude_deprecated=exclude_deprecated, min_tool_proficiency=min_tool_proficiency, require_vision=require_vision, require_json_mode=require_json_mode, ) ] def estimated_cost( row: dict, prompt_tokens: int, completion_tokens: int, cache_rate: float, ) -> float | None: """What this request should cost on this model, from catalog prices. Replaced a benchmark. The old cost signal came from a fixed 400-token reference sweep, and it was measured to be WRONG for real traffic: on a 400-token prompt glm-5.2-fast looked 3.2x cheaper than deepseek-v4-flash, but on a realistic 70k-token prompt deepseek is 5.0x cheaper. Attribution inverts with prompt size — glm batches beautifully on toy prompts and badly on real ones — so a fixed-shape benchmark cannot rank models for a workload of a different shape. Catalog prices, scaled to THIS request's shape, get the same answer as the live measurement (7.8x vs 5.0x, same direction). They are also free, need no sweep, and update whenever the poller runs. They are not what gets billed — NeuralWatt charges per kWh — but billing is capped at a multiple of token price, so this tracks the real ordering and bounds it. Cheap and directionally right beats precise about the wrong workload. """ prompt_price = row.get("cost_per_1m_prompt") completion_price = row.get("cost_per_1m_completion") if prompt_price is None or completion_price is None: return None # Agent traffic resends the conversation every turn, so most prompt tokens # hit the provider's prefix cache and are billed at the cached rate. cached_price = row.get("cost_per_1m_prompt_cached") if cached_price is None: cached_price = prompt_price fresh = prompt_tokens * (1.0 - cache_rate) * prompt_price cached = prompt_tokens * cache_rate * cached_price return (fresh + cached + completion_tokens * completion_price) / 1_000_000 def within_budget(row: dict, max_energy_kwh: float | None) -> bool: """Whether a candidate's measured energy is inside the per-request ceiling. Energy rather than dollars because the plan is a subscription with a fixed kWh quota. Dollars accrue and can be reasoned about after the fact; a quota is a wall you hit in the middle of a task. A model with no energy measurement is admitted. Excluding the unmeasured would mean a newly listed model could never be picked and so could never acquire a measurement — the same trap the neutral-0.5 default avoids elsewhere. """ if max_energy_kwh is None: return True energy = row.get("energy") return energy is None or energy <= max_energy_kwh def rank_candidates( rows: Sequence[dict], *, quality_tolerance: float = 0.1, max_energy_per_request: float | None = None, prompt_tokens: int = 0, completion_tokens: int = 500, cache_rate: float = 0.84, ) -> list[dict]: """Order candidates: best quality first, cheapest among equals. This replaced a weighted blend of cost, eco and proficiency, for reasons measurement forced: - **Eco is no longer an objective.** Carbon is still logged per request, but it is not something this router optimizes; that judgement is made outside it. Leaving it in meant 20% of every decision optimized an unstated goal. - **Cost is a constraint and a tiebreak, not a weight.** Under the blend, 60% of the decision adjudicated differences of fractions of a cent — all real traffic to date totals $0.07 — and min-max normalization made "expensive" relative to whoever else happened to be a candidate, so a model could lose for being 2x a very cheap model when both round to nothing. A ceiling states the cost mandate as a guarantee instead. - **Quality is the objective**, which is what the blend obscured: turning the cost weight from 0.4 all the way to zero changed the winner in only 2 of 6 categories, because the blend was never really steering on quality at all. ``quality_tolerance`` is the band inside which two proficiency scores are treated as equal. It is not a preference — it reflects measurement noise. Proficiency currently rests on 2-3 samples per category, so a gap of 0.05 is indistinguishable from sampling variation, and paying more for it would be buying noise. Widen it as confidence falls, narrow it as samples accumulate. Returns each row plus ``proficiency_score``, ``cost_score`` (reported for visibility only, no longer part of the decision) and ``composite``, which is now simply the effective quality after the tolerance band. """ affordable = [r for r in rows if within_budget(r, max_energy_per_request)] # Priced for THIS request rather than for a benchmark, which is the whole # point: the ordering depends on the workload's shape. A row the catalog # has no price for keeps whatever measured cost it arrived with rather # than losing the field — a missing list price is not free. estimates = [] for r in affordable: est = estimated_cost(r, prompt_tokens, completion_tokens, cache_rate) estimates.append(r.get("cost") if est is None else est) # cost_score is retained purely so callers can still see the spread; it # does not enter the ordering. cost_scores = cost_score(estimates) ranked = [] for row, c_s, est in zip(affordable, cost_scores, estimates): p_s = proficiency_score(row.get("proficiency")) ranked.append({**row, "cost": est, "cost_score": c_s, "proficiency_score": p_s, "composite": p_s}) if not ranked: return [] # Quality first, but only differences larger than the tolerance count. # Bucketing by band means a 0.01 edge cannot outrank a 10x cost saving, # while a real gap (tool_use_agentic spans 0.67) still decides outright. best = max(r["proficiency_score"] for r in ranked) def band(r: dict) -> float: # 0 means "no band": rank strictly by quality, with cost breaking only # exact ties. A legitimate setting -- "never trade quality for cost" -- # and one the config validator has always accepted, while this line # divided by it. if quality_tolerance <= 0: return -r["proficiency_score"] return float(int((best - r["proficiency_score"]) / quality_tolerance)) ranked.sort( key=lambda r: ( band(r), r["cost"] if r.get("cost") is not None else float("inf"), r["model_id"], ) ) return ranked