Add COOLDOWN_FLOOR, FALLBACK_TIER_MIN, DEGRADED_WARN_MIN_CEILING, etc. constants and field_validators for cooldown_seconds, fallback_tier, degraded_warn_min. Update test_admin_config.py for category/advanced fields. Ultraworked with Sisyphus Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
2127 lines
88 KiB
Python
2127 lines
88 KiB
Python
"""
|
|
Loads and validates config.yaml for the local LLM router.
|
|
|
|
Usage:
|
|
from config import load_config
|
|
cfg = load_config("config/config.yaml")
|
|
cfg.objective.quality_tolerance # etc.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from enum import Enum
|
|
import logging
|
|
import os
|
|
from pathlib import Path
|
|
from typing import Literal, Optional
|
|
from urllib.parse import urlparse
|
|
|
|
import yaml
|
|
from pydantic import BaseModel, ConfigDict, Field, PrivateAttr, field_validator, model_validator
|
|
|
|
|
|
class StrictModel(BaseModel):
|
|
"""Base for every config section: an unknown key is an error.
|
|
|
|
Pydantic ignores extra keys by default, which makes a typo or a
|
|
misplaced setting silently do nothing while the file still loads and
|
|
still looks configured. That is not hypothetical here — `max_input_chars`
|
|
was written into the `verification:` block instead of `classifier:`,
|
|
where it was accepted, ignored, and had no effect. It happened to carry
|
|
the same value as the code default, so nothing visibly broke; editing it
|
|
would simply have done nothing.
|
|
|
|
Anyone tuning this file needs a wrong key to say so.
|
|
"""
|
|
|
|
model_config = ConfigDict(extra="forbid")
|
|
|
|
|
|
class CreditAttenuationConfig(StrictModel):
|
|
"""Optional cost-inflation tiebreak for providers with a polled account balance.
|
|
|
|
When enabled, a provider whose prepaid balance is near depletion gets its
|
|
comparison cost multiplied inside the quality-first ranking. Quality bands
|
|
still win; the multiplier only breaks ties. Telemetry providers (those
|
|
billed per-completion via ``allowance_remaining_usd``) are always treated
|
|
as multiplier 1.0 — their "low" reading is normal overage-invoice noise,
|
|
not a depleting pool, so attenuation never applies to them.
|
|
"""
|
|
|
|
enabled: bool = False
|
|
soft_floor_usd: float = 5.0
|
|
zero_floor_usd: float = 0.0
|
|
max_multiplier: float = 5.0
|
|
refresh_seconds: int = 300
|
|
|
|
@field_validator("soft_floor_usd")
|
|
@classmethod
|
|
def soft_floor_positive(cls, v: float) -> float:
|
|
if v <= 0:
|
|
raise ValueError("objective.credit_attenuation.soft_floor_usd must be > 0")
|
|
return v
|
|
|
|
@field_validator("zero_floor_usd")
|
|
@classmethod
|
|
def zero_floor_non_negative(cls, v: float) -> float:
|
|
if v < 0:
|
|
raise ValueError(
|
|
"objective.credit_attenuation.zero_floor_usd must be >= 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("max_multiplier")
|
|
@classmethod
|
|
def max_multiplier_above_one(cls, v: float) -> float:
|
|
if v <= 1.0:
|
|
raise ValueError(
|
|
"objective.credit_attenuation.max_multiplier must be > 1.0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("refresh_seconds")
|
|
@classmethod
|
|
def refresh_seconds_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError(
|
|
"objective.credit_attenuation.refresh_seconds must be > 0"
|
|
)
|
|
return v
|
|
|
|
@model_validator(mode="after")
|
|
def zero_below_soft(self) -> "CreditAttenuationConfig":
|
|
if self.zero_floor_usd >= self.soft_floor_usd:
|
|
raise ValueError(
|
|
"objective.credit_attenuation.zero_floor_usd "
|
|
"must be < soft_floor_usd"
|
|
)
|
|
return self
|
|
|
|
|
|
class Objective(StrictModel):
|
|
"""What the router optimizes: quality, bounded by cost.
|
|
|
|
Replaced a three-way weighted blend. See config.yaml for why — briefly,
|
|
the cost weight was measured to be nearly inert while consuming 40% of
|
|
every decision.
|
|
"""
|
|
|
|
quality_tolerance: float = 0.10
|
|
plan_pace_warn_ratio: float = 1.25
|
|
assumed_cache_rate: float = 0.917
|
|
assumed_completion_tokens: int = 500
|
|
max_energy_per_request: Optional[float] = None
|
|
plan_kwh_per_period: Optional[float] = None
|
|
billing_reset_day: Optional[int] = None
|
|
quota_burn_window_hours: Optional[int] = None
|
|
quota_runway_warning_hours: Optional[int] = None
|
|
quota_burn_min_segment_samples: Optional[int] = None
|
|
quota_burn_min_segment_hours: Optional[float] = None
|
|
rejection_warning_window_hours: Optional[int] = None
|
|
rejection_warning_baseline_hours: Optional[int] = None
|
|
rejection_warning_min_count: Optional[int] = None
|
|
selection_coverage_window_hours: Optional[int] = None
|
|
cache_rate_window_hours: Optional[int] = None
|
|
cache_rate_warn_margin: Optional[float] = None
|
|
cache_rate_warn_min_observations: Optional[int] = None
|
|
|
|
# Premise-expiry checks for the quality_tolerance and cost-as-tiebreak
|
|
# justifications in config.yaml. These are /metrics warning thresholds,
|
|
# not dispatch controls — they fire once the comment's premise is clearly
|
|
# expired, so the operator knows to re-read that justification.
|
|
proficiency_depth_warn_min_samples: Optional[int] = None
|
|
proficiency_depth_warn_min_rows: Optional[int] = None
|
|
cumulative_spend_warn_usd: Optional[float] = None
|
|
cumulative_spend_warn_min_rows: Optional[int] = None
|
|
|
|
# Report-only measurement windows. Neither series is read by routing;
|
|
# see metrics.cost_estimate_calibration / metrics.latency_series.
|
|
cost_calibration_window_hours: Optional[int] = None
|
|
cost_calibration_min_observations: Optional[int] = None
|
|
latency_window_hours: Optional[int] = None
|
|
latency_min_observations: Optional[int] = None
|
|
|
|
# Gate: with this off, routing is byte-identical to today.
|
|
# Flip on only after the Wave 1 post-restart baseline day.
|
|
incumbent_cache_pricing: bool = False
|
|
|
|
# Challenger cache-rate dial.
|
|
# - None (default): neutral, follows assumed_cache_rate.
|
|
# - 0.0: challengers priced as fully cold prompts (maximum incumbent advantage).
|
|
incumbent_challenger_cache_rate: Optional[float] = None
|
|
|
|
# Seconds between refreshes of measured per-(provider, model) cache rates.
|
|
incumbent_rate_refresh_seconds: int = 300
|
|
|
|
# Minimum reported-cache observations before a per-(provider, model)
|
|
# cache rate is trusted for pricing decisions.
|
|
incumbent_rate_min_observations: int = 25
|
|
|
|
# Window (in seconds) for the conversation adoption counter in /metrics.
|
|
# Recent route_decisions rows within this window whose session_key starts
|
|
# with "c:" are counted. Null or absent means "all time" (no window);
|
|
# 0 is rejected by the adoption_window_positive validator below.
|
|
adoption_window_seconds: Optional[int] = None
|
|
|
|
credit_attenuation: CreditAttenuationConfig = CreditAttenuationConfig()
|
|
|
|
@field_validator("quality_tolerance")
|
|
@classmethod
|
|
def tolerance_in_range(cls, v: float) -> float:
|
|
if not (0.0 <= v < 1.0):
|
|
raise ValueError("objective.quality_tolerance must be in [0, 1)")
|
|
return v
|
|
|
|
@field_validator("plan_pace_warn_ratio")
|
|
@classmethod
|
|
def pace_warn_ratio_positive(cls, v: float) -> float:
|
|
if v <= 0:
|
|
raise ValueError("objective.plan_pace_warn_ratio must be > 0")
|
|
return v
|
|
|
|
@field_validator("billing_reset_day")
|
|
@classmethod
|
|
def billing_reset_day_range(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and not (1 <= v <= 28):
|
|
raise ValueError(
|
|
"objective.billing_reset_day must be in 1..28, or null to disable"
|
|
)
|
|
return v
|
|
|
|
@field_validator("proficiency_depth_warn_min_samples")
|
|
@classmethod
|
|
def depth_min_samples_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.proficiency_depth_warn_min_samples must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("proficiency_depth_warn_min_rows")
|
|
@classmethod
|
|
def depth_min_rows_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.proficiency_depth_warn_min_rows must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("cumulative_spend_warn_usd")
|
|
@classmethod
|
|
def spend_warn_usd_positive(cls, v: Optional[float]) -> Optional[float]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.cumulative_spend_warn_usd must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("cumulative_spend_warn_min_rows")
|
|
@classmethod
|
|
def spend_min_rows_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.cumulative_spend_warn_min_rows must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("max_energy_per_request")
|
|
@classmethod
|
|
def ceiling_positive(cls, v: Optional[float]) -> Optional[float]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.max_energy_per_request must be > 0 kWh, or null to disable"
|
|
)
|
|
return v
|
|
|
|
@field_validator("quota_burn_window_hours")
|
|
@classmethod
|
|
def burn_window_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.quota_burn_window_hours must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("quota_runway_warning_hours")
|
|
@classmethod
|
|
def runway_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.quota_runway_warning_hours must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("quota_burn_min_segment_samples")
|
|
@classmethod
|
|
def segment_samples_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.quota_burn_min_segment_samples must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("quota_burn_min_segment_hours")
|
|
@classmethod
|
|
def segment_hours_positive(cls, v: Optional[float]) -> Optional[float]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.quota_burn_min_segment_hours must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("rejection_warning_window_hours")
|
|
@classmethod
|
|
def rejection_window_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.rejection_warning_window_hours must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("rejection_warning_baseline_hours")
|
|
@classmethod
|
|
def rejection_baseline_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.rejection_warning_baseline_hours must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("rejection_warning_min_count")
|
|
@classmethod
|
|
def rejection_min_count_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.rejection_warning_min_count must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("selection_coverage_window_hours")
|
|
@classmethod
|
|
def selection_window_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.selection_coverage_window_hours must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("cache_rate_window_hours")
|
|
@classmethod
|
|
def cache_rate_window_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.cache_rate_window_hours must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator("cache_rate_warn_margin")
|
|
@classmethod
|
|
def cache_rate_margin_in_range(cls, v: Optional[float]) -> Optional[float]:
|
|
# A margin of 0 warns on any float noise at all; a margin of 1 can
|
|
# never be exceeded, because both the measured rate and the assumed
|
|
# one live in [0, 1] and their absolute difference cannot reach 1
|
|
# without one of them being an exact 0 against an exact 1.
|
|
if v is not None and not (0.0 < v < 1.0):
|
|
raise ValueError(
|
|
"objective.cache_rate_warn_margin must be in (0, 1)"
|
|
)
|
|
return v
|
|
|
|
@field_validator("cache_rate_warn_min_observations")
|
|
@classmethod
|
|
def cache_rate_min_obs_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.cache_rate_warn_min_observations must be > 0"
|
|
)
|
|
return v
|
|
|
|
@field_validator(
|
|
"cost_calibration_window_hours",
|
|
"cost_calibration_min_observations",
|
|
"latency_window_hours",
|
|
"latency_min_observations",
|
|
)
|
|
@classmethod
|
|
def report_window_positive(cls, v: Optional[int], info) -> Optional[int]:
|
|
# One validator for four knobs of the same shape. The field name comes
|
|
# from the ValidationInfo rather than being hardcoded, so a renamed
|
|
# knob cannot keep reporting the old name -- the failure mode that
|
|
# makes a strict-config error message actively misleading.
|
|
if v is not None and v <= 0:
|
|
raise ValueError(f"objective.{info.field_name} must be > 0")
|
|
return v
|
|
|
|
@field_validator("incumbent_rate_refresh_seconds")
|
|
@classmethod
|
|
def incumbent_refresh_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("objective.incumbent_rate_refresh_seconds must be > 0")
|
|
return v
|
|
|
|
@field_validator("incumbent_challenger_cache_rate")
|
|
@classmethod
|
|
def challenger_cache_rate_in_range(cls, v: Optional[float]) -> Optional[float]:
|
|
if v is not None and not (0.0 <= v <= 1.0):
|
|
raise ValueError("objective.incumbent_challenger_cache_rate must be in [0, 1]")
|
|
return v
|
|
|
|
@field_validator("incumbent_rate_min_observations")
|
|
@classmethod
|
|
def incumbent_min_obs_non_negative(cls, v: int) -> int:
|
|
if v < 0:
|
|
raise ValueError("objective.incumbent_rate_min_observations must be >= 0")
|
|
return v
|
|
|
|
@field_validator("adoption_window_seconds")
|
|
@classmethod
|
|
def adoption_window_positive(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"objective.adoption_window_seconds must be > 0, or null/absent for all time"
|
|
)
|
|
return v
|
|
|
|
@model_validator(mode="after")
|
|
def _resolve_challenger_cache_rate(self) -> "Objective":
|
|
if self.incumbent_challenger_cache_rate is None:
|
|
self.incumbent_challenger_cache_rate = self.assumed_cache_rate
|
|
return self
|
|
|
|
|
|
class ContextOverride(StrictModel):
|
|
"""Per-model context handling, for a row whose real limits are known.
|
|
|
|
Typed rather than a bare ``dict`` so a typo INSIDE an override is an error
|
|
too. That is the whole point of StrictModel, and it was not true here: the
|
|
block validated, nothing read it, and config.yaml shipped a worked example
|
|
for it -- so anyone who followed that example got silence.
|
|
|
|
Both fields are optional; whichever is absent falls back to the global.
|
|
"""
|
|
|
|
safety_factor: Optional[float] = None
|
|
output_reserve_tokens: Optional[int] = None
|
|
|
|
@field_validator("safety_factor")
|
|
@classmethod
|
|
def factor_in_range(cls, v: Optional[float]) -> Optional[float]:
|
|
if v is not None and not (0.0 < v <= 1.0):
|
|
raise ValueError(
|
|
"context.per_model_overrides[...].safety_factor must be in (0, 1]"
|
|
)
|
|
return v
|
|
|
|
@field_validator("output_reserve_tokens")
|
|
@classmethod
|
|
def reserve_not_negative(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v < 0:
|
|
raise ValueError(
|
|
"context.per_model_overrides[...].output_reserve_tokens must be >= 0"
|
|
)
|
|
return v
|
|
|
|
|
|
class ContextConfig(StrictModel):
|
|
safety_factor: float
|
|
default_output_reserve_tokens: int
|
|
# Ceiling on the output reserve, as a fraction of the usable window. See
|
|
# poller.ModelRow.effective_context_window for what this exists to stop.
|
|
max_output_reserve_fraction: float
|
|
# Read by poller.ModelRow.effective_context_window.
|
|
per_model_overrides: dict[str, ContextOverride] = {}
|
|
|
|
@field_validator("safety_factor")
|
|
@classmethod
|
|
def factor_in_range(cls, v: float) -> float:
|
|
if not (0.0 < v <= 1.0):
|
|
raise ValueError("context.safety_factor must be in (0, 1]")
|
|
return v
|
|
|
|
@field_validator("max_output_reserve_fraction")
|
|
@classmethod
|
|
def reserve_fraction_in_range(cls, v: float) -> float:
|
|
# Exclusive at both ends: 0 would reserve nothing at all for the
|
|
# answer, 1 would allow a reserve that consumes the whole window --
|
|
# which is the exact failure this ceiling exists to prevent.
|
|
if not (0.0 < v < 1.0):
|
|
raise ValueError(
|
|
"context.max_output_reserve_fraction must be in (0, 1)"
|
|
)
|
|
return v
|
|
|
|
|
|
class ProficiencyConfig(StrictModel):
|
|
self_eval_min_samples: int
|
|
leaderboard_weight: float
|
|
self_eval_weight: float
|
|
categories: list[str]
|
|
# Prior strength (k) for empirical-Bayes shrunken estimate: the number of
|
|
# pseudo-observations the peer prior contributes, controlling how far a
|
|
# noisy per-model score is pulled toward the global average. 20 was chosen
|
|
# as a conservative starting point — thin self-eval data on a single model
|
|
# (n < 100) would dominate without any shrinkage; 20 pseudo-observations
|
|
# dampens that without erasing the per-model signal.
|
|
outcome_prior_strength: int = 20
|
|
|
|
@model_validator(mode="after")
|
|
def blend_weights_sum_to_one(self) -> "ProficiencyConfig":
|
|
total = round(self.leaderboard_weight + self.self_eval_weight, 6)
|
|
if total != 1.0:
|
|
raise ValueError(
|
|
"proficiency.leaderboard_weight + self_eval_weight must sum to 1.0, "
|
|
f"got {total}"
|
|
)
|
|
return self
|
|
|
|
|
|
class TieringConfig(StrictModel):
|
|
cheap_completion_max: float
|
|
tier1_context_max: float = float("inf")
|
|
model_tiers: dict[str, int]
|
|
|
|
@field_validator("cheap_completion_max")
|
|
@classmethod
|
|
def max_must_be_positive(cls, v: float) -> float:
|
|
if v <= 0:
|
|
raise ValueError("tiering.cheap_completion_max must be > 0")
|
|
return v
|
|
|
|
@field_validator("tier1_context_max")
|
|
@classmethod
|
|
def context_max_must_be_positive(cls, v: float) -> float:
|
|
if v <= 0:
|
|
raise ValueError("tiering.tier1_context_max must be > 0")
|
|
return v
|
|
|
|
@field_validator("model_tiers")
|
|
@classmethod
|
|
def overrides_in_range(cls, v: dict[str, int]) -> dict[str, int]:
|
|
for model_id, tier in v.items():
|
|
if tier not in (1, 2, 3):
|
|
raise ValueError(
|
|
f"tiering.model_tiers[{model_id!r}] must be in {{1, 2, 3}}, "
|
|
f"got {tier}"
|
|
)
|
|
return v
|
|
|
|
|
|
class FlexPreference(Enum):
|
|
"""Operator's stance on routing to ``-flex`` serving-class rows.
|
|
|
|
A 4-position scale:
|
|
- ``no-flex``: never route to a flex row.
|
|
- ``auto``: decide per request (the default).
|
|
- ``prefer-flex``: flex first, standard as fallback.
|
|
- ``force-flex``: flex only.
|
|
"""
|
|
|
|
no_flex = "no-flex"
|
|
auto = "auto"
|
|
prefer_flex = "prefer-flex"
|
|
force_flex = "force-flex"
|
|
|
|
|
|
class RoutingProfile(StrictModel):
|
|
"""A named candidate-set filter requested through ``auto:<name>``.
|
|
|
|
Profiles restrict the candidate set; they do NOT change ranking. All
|
|
fields are optional, and any absent field falls back to the request's
|
|
own value or the global default. This keeps profiles small rewrite
|
|
rules rather than a parallel config layer.
|
|
"""
|
|
|
|
provider: Optional[str] = None
|
|
min_tier: Optional[int] = None
|
|
max_tier: Optional[int] = None
|
|
latency_tolerance: Optional[Literal["interactive", "batch"]] = None
|
|
max_cost_per_1m_completion: Optional[float] = None
|
|
allowed_model_ids: Optional[set[str]] = None
|
|
|
|
@field_validator("min_tier", "max_tier")
|
|
@classmethod
|
|
def tier_in_range(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v not in (1, 2, 3):
|
|
raise ValueError(
|
|
f"profiles[...].{cls.__name__}.tier must be in {{1, 2, 3}}, got {v}"
|
|
)
|
|
return v
|
|
|
|
@model_validator(mode="after")
|
|
def min_not_above_max(self) -> "RoutingProfile":
|
|
if self.min_tier is not None and self.max_tier is not None:
|
|
if self.min_tier > self.max_tier:
|
|
raise ValueError(
|
|
f"profiles[...].{self.__class__.__name__}.min_tier ({self.min_tier}) "
|
|
f"must be <= max_tier ({self.max_tier})"
|
|
)
|
|
return self
|
|
|
|
@field_validator("max_cost_per_1m_completion")
|
|
@classmethod
|
|
def cost_positive(cls, v: Optional[float]) -> Optional[float]:
|
|
if v is not None and v <= 0:
|
|
raise ValueError(
|
|
"profiles[...].max_cost_per_1m_completion must be > 0, or null"
|
|
)
|
|
return v
|
|
|
|
@field_validator("allowed_model_ids")
|
|
@classmethod
|
|
def allowed_model_ids_non_empty(cls, v: Optional[set[str]]) -> Optional[set[str]]:
|
|
if v is not None and not v:
|
|
raise ValueError("profiles[...].allowed_model_ids must be non-empty when set")
|
|
return v
|
|
|
|
|
|
# Built-in routing profiles. Profiles are selectable as ``auto:<name>``; they
|
|
# restrict the candidate set and may override the effective latency_tolerance.
|
|
# Defined in config.py so RouterConfig validators and modules that may not
|
|
# import dispatcher (e.g., admin.py, metrics.py) can enumerate them directly.
|
|
BUILTIN_PROFILES: dict[str, RoutingProfile] = {
|
|
"default": RoutingProfile(),
|
|
"batch": RoutingProfile(latency_tolerance="batch"),
|
|
"locality": RoutingProfile(provider="ollama-local"),
|
|
"bigboybritches": RoutingProfile(min_tier=3),
|
|
"onlycheaps": RoutingProfile(max_cost_per_1m_completion=0.5),
|
|
}
|
|
|
|
|
|
class RoutingConfig(StrictModel):
|
|
allowed_access_levels: list[str]
|
|
default_latency_tolerance: str
|
|
# Operator's default stance on flex serving-class rows for requests
|
|
# that do not state one explicitly.
|
|
default_flex_preference: FlexPreference = FlexPreference.auto
|
|
# Bare ``auto`` resolves to this profile. It names a built-in profile by
|
|
# default; operators may override it via config (or the admin Controls page
|
|
# in a later wave) to change the implicit behavior of unqualified ``auto``.
|
|
default_profile: str = "default"
|
|
# Applied only when the REQUEST carries tool definitions. None disables it.
|
|
min_tool_proficiency: Optional[float] = 0.5
|
|
tool_use_category: str = "tool_use_agentic"
|
|
|
|
# Whether a request carrying image parts is hard-restricted to vision-capable
|
|
# models. A wrong guess here is a guaranteed 400, so this gates by default
|
|
# and routing fails closed when the catalog flag is unknown.
|
|
require_vision: bool = True
|
|
# Whether a response_format requiring json_object/json_schema is hard-restricted
|
|
# to JSON-mode-capable models. Same guaranteed-failure argument.
|
|
require_json_mode: bool = True
|
|
|
|
@field_validator("allowed_access_levels")
|
|
@classmethod
|
|
def levels_known(cls, v: list[str]) -> list[str]:
|
|
known = {"public", "preview", "canary"}
|
|
unknown = set(v) - known
|
|
if unknown:
|
|
raise ValueError(
|
|
f"routing.allowed_access_levels contains unknown levels {sorted(unknown)}; "
|
|
f"must be a subset of {sorted(known)}"
|
|
)
|
|
if not v:
|
|
raise ValueError("routing.allowed_access_levels must not be empty")
|
|
return v
|
|
|
|
@field_validator("default_latency_tolerance")
|
|
@classmethod
|
|
def tolerance_known(cls, v: str) -> str:
|
|
if v not in ("interactive", "batch"):
|
|
raise ValueError(
|
|
f"routing.default_latency_tolerance must be 'interactive' or 'batch', got {v!r}"
|
|
)
|
|
return v
|
|
|
|
@field_validator("default_profile")
|
|
@classmethod
|
|
def default_profile_non_empty(cls, v: str) -> str:
|
|
if not v or not v.strip():
|
|
raise ValueError("routing.default_profile must be a non-empty string")
|
|
return v
|
|
|
|
|
|
class VerificationConfig(StrictModel):
|
|
local_llm_enabled: bool = True
|
|
min_completion_tokens: int = 600
|
|
timeout_seconds: int = 60
|
|
max_output_tokens: int = 1024
|
|
outcome_attribution_window_seconds: int = 120
|
|
# The local checker's OWN endpoint, no longer derived from the
|
|
# classifier's. It speaks Ollama's NATIVE API (/api/chat, think=False),
|
|
# which no cloud provider offers, so it must keep pointing at an Ollama
|
|
# instance even when classification has been moved off this machine.
|
|
base_url: str = "http://localhost:11434"
|
|
# None means "whatever the classifier uses", which is correct only while
|
|
# both run on the same local Ollama. Set it explicitly once they diverge.
|
|
model: Optional[str] = None
|
|
|
|
|
|
class LocalVisionConfig(StrictModel):
|
|
enabled: bool = True
|
|
base_url: str = "http://localhost:11434/v1" # OpenAI-compatible (classifier shape)
|
|
api_key_env: Optional[str] = None
|
|
model: str = "qwen3-vl:4b"
|
|
timeout_seconds: int = 60
|
|
max_images: int = 4
|
|
max_image_bytes: int = 9 * 1024 * 1024 # 9 MiB, Ollama default cap
|
|
|
|
@field_validator("timeout_seconds")
|
|
@classmethod
|
|
def timeout_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("local_vision.timeout_seconds must be > 0")
|
|
return v
|
|
|
|
@field_validator("max_images")
|
|
@classmethod
|
|
def images_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("local_vision.max_images must be > 0")
|
|
return v
|
|
|
|
@field_validator("max_image_bytes")
|
|
@classmethod
|
|
def image_bytes_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("local_vision.max_image_bytes must be > 0")
|
|
return v
|
|
|
|
|
|
class EscalationConfig(StrictModel):
|
|
enabled: bool
|
|
max_tier: int
|
|
min_confidence_before_bump: float
|
|
# Off by default: the iteration budget escalates on evidence instead.
|
|
preemptive_on_low_confidence: bool = False
|
|
|
|
|
|
class IterationConfig(StrictModel):
|
|
"""A tier's budget for corrective attempts after a verification failure."""
|
|
|
|
enabled: bool = True
|
|
attempts_by_tier: dict[int, int] = {1: 0, 2: 1, 3: 2}
|
|
max_attempts_interactive: int = 1
|
|
# Maximum conversation prompt-tokens for which escalation (cache-destroying
|
|
# re-bill on a different model) is allowed. Same-model retries preserve
|
|
# the cache and are not gated. 0 = no limit (backward-compatible default).
|
|
max_rebill_prompt_tokens: int = 0
|
|
|
|
@field_validator("attempts_by_tier")
|
|
@classmethod
|
|
def attempts_sane(cls, v: dict[int, int]) -> dict[int, int]:
|
|
for tier, attempts in v.items():
|
|
if attempts < 0:
|
|
raise ValueError(f"iteration.attempts_by_tier[{tier}] must be >= 0")
|
|
if attempts > 5:
|
|
raise ValueError(
|
|
f"iteration.attempts_by_tier[{tier}]={attempts} is implausibly "
|
|
"high; each attempt spends energy against a fixed quota"
|
|
)
|
|
return v
|
|
|
|
|
|
class FreshnessConfig(StrictModel):
|
|
stale_after_days: int
|
|
exclude_stale: bool
|
|
exclude_deprecated: bool
|
|
# Seconds to wait after an admin allowlist edit before re-polling the
|
|
# catalog. Debounce, not delay: each further edit restarts the clock, so
|
|
# adding five models one at a time costs one poll rather than five.
|
|
# 0 disables the trigger entirely and the scheduled timer stays the only
|
|
# path. See admin._schedule_catalog_poll.
|
|
repoll_after_allowlist_change_seconds: float = 0.0
|
|
|
|
@field_validator("repoll_after_allowlist_change_seconds")
|
|
@classmethod
|
|
def repoll_non_negative(cls, v: float) -> float:
|
|
if v < 0:
|
|
raise ValueError(
|
|
"freshness.repoll_after_allowlist_change_seconds must be >= 0"
|
|
)
|
|
return v
|
|
|
|
|
|
class PinchRelevanceConfig(StrictModel):
|
|
"""Embedding-model relevance scoring for pinch trimming.
|
|
|
|
When enabled (and ``pinch.enabled`` is also true), the dispatcher embeds
|
|
the current-turn query with the old tool-result candidates and trims the
|
|
least relevant first, so a relevant-but-old result survives. On by
|
|
default; any failure reverts to uniform trimming. This must point at an
|
|
EMBEDDING model, never ``classifier.model`` or ``verification.model``.
|
|
"""
|
|
|
|
enabled: bool = True
|
|
# OpenAI-compatible embeddings endpoint on the same local Ollama.
|
|
model: str = "nomic-embed-text"
|
|
base_url: str = "http://localhost:11434/v1"
|
|
timeout_seconds: int = 10
|
|
# Below this many trim-eligible candidates, skip the embedding round-trip.
|
|
min_candidates: int = 2
|
|
|
|
@field_validator("timeout_seconds")
|
|
@classmethod
|
|
def timeout_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("pinch.relevance.timeout_seconds must be > 0")
|
|
return v
|
|
|
|
@field_validator("min_candidates")
|
|
@classmethod
|
|
def min_candidates_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("pinch.relevance.min_candidates must be > 0")
|
|
return v
|
|
|
|
|
|
class PinchConfig(StrictModel):
|
|
"""Optional relevance-based context pruning (Port of llmrouter's pinch).
|
|
|
|
Prunes the provider-bound conversation — not the classifier input — when it
|
|
exceeds ``budget_tokens``, so a long agent session ships fewer prompt tokens
|
|
upstream. User/assistant/system messages are always kept; only tool results
|
|
are summarized or dropped (they carry the bulk of a long session's tokens).
|
|
"""
|
|
|
|
enabled: bool = True
|
|
budget_tokens: int = 50000
|
|
# How many recent user turns (plus their assistant replies and tool results)
|
|
# are protected from pruning.
|
|
keep_last_turns: int = 4
|
|
# Tool results longer than this many characters are summarized in place.
|
|
max_summarize_chars: int = 4000
|
|
# keep_last_turns has no size limit inside it -- an entire autonomous
|
|
# tool-call loop with no new user message can be one protected turn, and
|
|
# one outsized tool result inside it (a full verbose test run, a huge
|
|
# file read) ships verbatim regardless of size. Measured live: a
|
|
# 324k-token conversation shrank only ~8% because nearly all of it sat
|
|
# inside the protected window. This is deliberately a much higher bar
|
|
# than max_summarize_chars -- recent results are more likely to still
|
|
# matter -- so it only catches true outliers. None disables it.
|
|
protected_max_chars: Optional[int] = 20000
|
|
# Record, on each decision row, where this turn's pruned payload stopped
|
|
# matching the previous turn's in the same session -- the provider bills
|
|
# the longest byte-identical prefix at the cached rate, so a divergence
|
|
# early in the payload re-bills everything after it. HASHES ONLY, held in
|
|
# process memory for one turn; what reaches route_decisions is three
|
|
# integers. Measured on the live median payload (98k tokens in, 74k out):
|
|
# ~1.0 ms per turn, against a request path whose floor is a provider
|
|
# round-trip of 1.4-2.0 s. On by default because a probe that is off
|
|
# measures nothing, and Wave 3 of plans/token-waste-waves.md is waiting on
|
|
# what it says.
|
|
prefix_probe: bool = True
|
|
relevance: PinchRelevanceConfig = PinchRelevanceConfig()
|
|
|
|
@field_validator("budget_tokens")
|
|
@classmethod
|
|
def budget_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("pinch.budget_tokens must be > 0")
|
|
return v
|
|
|
|
@field_validator("keep_last_turns")
|
|
@classmethod
|
|
def turns_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("pinch.keep_last_turns must be > 0")
|
|
return v
|
|
|
|
@field_validator("max_summarize_chars")
|
|
@classmethod
|
|
def summarize_chars_valid(cls, v: int) -> int:
|
|
if v < 3000:
|
|
raise ValueError(
|
|
"pinch.max_summarize_chars must be >= 3000 (below this, "
|
|
"summarization grows the message)"
|
|
)
|
|
return v
|
|
|
|
@field_validator("protected_max_chars")
|
|
@classmethod
|
|
def protected_max_chars_valid(cls, v: Optional[int]) -> Optional[int]:
|
|
if v is not None and v < 3000:
|
|
raise ValueError(
|
|
"pinch.protected_max_chars must be >= 3000, or null to "
|
|
"disable (below this, elision grows the message, same "
|
|
"floor as max_summarize_chars)"
|
|
)
|
|
return v
|
|
|
|
|
|
# The legal range for ``session_cache.staleness_seconds``, kept as module
|
|
# constants because three surfaces have to agree on it: this validator (file
|
|
# and overlay edits), ``admin._INT_KNOBS`` (the runtime write, which bypasses
|
|
# every Pydantic validator), and the number field's min/max in
|
|
# ``admin/frontend/controls.html``. Two of the three import these; the third is
|
|
# pinned by a test.
|
|
#
|
|
# The floor is 5, not 0, and that is the pre-existing ``> 0`` rule rather than
|
|
# a new opinion. It matters more than it looks: 0 would make ``session_cache.get``
|
|
# miss on every turn, which LOOKS like "stop reusing labels" but is not —
|
|
# ``session_cache.put`` still writes, and the classifier-failure cascade's
|
|
# ``stale_read`` ignores staleness entirely, so a 0-second window still replays
|
|
# a session's label whenever the classifier is down. The knob that actually
|
|
# stops reuse is ``session_cache.enabled``, and it has its own control.
|
|
#
|
|
# The ceiling is a judgement, and the judgement is that an unbounded window is
|
|
# a cache that never expires. 7200 is 6x the shipped default and ~8.5x the
|
|
# longest single-classification run measured on live traffic (107 consecutive
|
|
# turns across 840 seconds), so it is far above any value there is
|
|
# a reason to try, while still guaranteeing a label cannot outlive the working
|
|
# session that produced it. See CLAUDE.md, north star rule 2.
|
|
STALENESS_SECONDS_MIN = 5
|
|
STALENESS_SECONDS_MAX = 7200
|
|
|
|
|
|
class SessionCacheConfig(StrictModel):
|
|
"""Per-session classification cache (in-memory, process lifetime).
|
|
|
|
Remembers the last task_category/task_tier decision for each session for
|
|
``staleness_seconds``, so a long agent session skips the classifier
|
|
round-trip on every turn. Capability flags (tools/images/json) are NEVER
|
|
cached — they are read fresh from each request body. Fallback
|
|
classifications are NEVER cached. No persistence: the cache lives only in
|
|
process memory, so a restart just reclassifies once per session.
|
|
"""
|
|
|
|
enabled: bool = False
|
|
# Seconds since a cached classification was written before it is treated
|
|
# as expired and the next turn reclassifies from scratch.
|
|
staleness_seconds: int = 1200
|
|
|
|
@field_validator("staleness_seconds")
|
|
@classmethod
|
|
def staleness_in_range(cls, v: int) -> int:
|
|
if not (STALENESS_SECONDS_MIN <= v <= STALENESS_SECONDS_MAX):
|
|
raise ValueError(
|
|
f"session_cache.staleness_seconds must be between "
|
|
f"{STALENESS_SECONDS_MIN} and {STALENESS_SECONDS_MAX} "
|
|
f"(to stop reusing classifications set session_cache.enabled "
|
|
f"to false; 0 is not that, it still caches and the "
|
|
f"classifier-failure cascade still replays the entry)"
|
|
)
|
|
return v
|
|
|
|
|
|
class ExplorationConfig(StrictModel):
|
|
"""Epsilon-greedy exploration over ranked candidates.
|
|
|
|
On the epsilon share of requests, the router picks the hard-filter-eligible
|
|
candidate with the fewest outcome samples (tie-break: lowest cost) instead
|
|
of the highest-score model. This injects exploration into a system that
|
|
would otherwise converge on a single winner, which is how exposure bias
|
|
creeps in — the model that happens to be sampled more looks better, and
|
|
gets sampled more. Off by default: ship it, watch route_decisions
|
|
with ``was_exploration=True`` on real traffic, then enable it.
|
|
"""
|
|
|
|
enabled: bool = True
|
|
# Probability of exploration per request. 0 = never explore, 1 = always
|
|
# explore. 0.03 means ~3% of requests take an exploratory path. Above 1.0
|
|
# is an error — epsilon-greedy with epsilon > 1 is silently self-defeating;
|
|
# it stops being "mostly-exploit" and becomes random selection.
|
|
epsilon: float = 0.03
|
|
# Maximum cost ratio for an exploration candidate. The explorer picks the
|
|
# least-sampled eligible row; this caps its cost at this multiple of the
|
|
# ranking winner so an unproven frontier row doesn't eat the explore budget.
|
|
# A ratio below 1.0 is an error — you cannot go below the winner's own cost.
|
|
max_cost_ratio: float = 4.0
|
|
# Tier ceiling for exploration: only models in tiers 1..max_tier are
|
|
# eligible for the explore path. Tier 2 is a safety gate — you do not
|
|
# want exploration spending frontier budgets on random, unproven calls.
|
|
max_tier: int = 2
|
|
|
|
@field_validator("epsilon")
|
|
@classmethod
|
|
def epsilon_in_range(cls, v: float) -> float:
|
|
if not (0.0 <= v <= 1.0):
|
|
raise ValueError(
|
|
"exploration.epsilon must be in [0, 1], got "
|
|
f"{v} — above 1, epsilon-greedy stops being "
|
|
"mostly-exploit and becomes random selection"
|
|
)
|
|
return v
|
|
|
|
@field_validator("max_cost_ratio")
|
|
@classmethod
|
|
def cost_ratio_positive(cls, v: float) -> float:
|
|
if v < 1.0:
|
|
raise ValueError(
|
|
"exploration.max_cost_ratio must be >= 1.0, got "
|
|
f"{v} — a ratio below the winner's own cost "
|
|
"would always reject the exploration candidate"
|
|
)
|
|
return v
|
|
|
|
@field_validator("max_tier")
|
|
@classmethod
|
|
def tier_in_range(cls, v: int) -> int:
|
|
if v not in (1, 2, 3):
|
|
raise ValueError(
|
|
"exploration.max_tier must be in {1, 2, 3}, "
|
|
f"got {v}"
|
|
)
|
|
return v
|
|
|
|
|
|
class CircuitBreakerConfig(StrictModel):
|
|
"""Passive circuit breaker for upstream model availability.
|
|
|
|
When enabled, a model that returns 5xx is temporarily skipped by routing
|
|
(with exponential backoff). Recovery is passive: a real request that would
|
|
have picked it becomes the probe once the cooldown passes. On by default.
|
|
"""
|
|
|
|
enabled: bool = True
|
|
initial_cooldown_seconds: int = 30
|
|
max_cooldown_seconds: int = 600
|
|
backoff_multiplier: float = 2.0
|
|
|
|
@field_validator("initial_cooldown_seconds")
|
|
@classmethod
|
|
def initial_positive(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("circuit_breaker.initial_cooldown_seconds must be > 0")
|
|
return v
|
|
|
|
@field_validator("max_cooldown_seconds")
|
|
@classmethod
|
|
def max_at_least_initial(cls, v: int) -> int:
|
|
if v <= 0:
|
|
raise ValueError("circuit_breaker.max_cooldown_seconds must be > 0")
|
|
return v
|
|
|
|
@model_validator(mode="after")
|
|
def max_gte_initial(self) -> "CircuitBreakerConfig":
|
|
if self.max_cooldown_seconds < self.initial_cooldown_seconds:
|
|
raise ValueError(
|
|
"circuit_breaker.max_cooldown_seconds must be >= "
|
|
"initial_cooldown_seconds"
|
|
)
|
|
return self
|
|
|
|
@field_validator("backoff_multiplier")
|
|
@classmethod
|
|
def multiplier_above_one(cls, v: float) -> float:
|
|
if v <= 1.0:
|
|
raise ValueError("circuit_breaker.backoff_multiplier must be > 1.0")
|
|
return v
|
|
|
|
|
|
class DatabaseConfig(StrictModel):
|
|
path: str
|
|
|
|
|
|
class CloudFallbackConfig(StrictModel):
|
|
"""A cloud classifier used ONLY when the local one is unreachable.
|
|
|
|
Deliberately its own block rather than reusing ``ClassifierConfig``: the
|
|
two differ in the ways that matter under failure. This one has a short
|
|
timeout because it sits on the latency floor of every request, and a
|
|
missing ``api_key_env`` degrades to the next cascade step instead of
|
|
raising — an optional fallback that fails the request would be worse than
|
|
the outage it exists to soften.
|
|
"""
|
|
|
|
base_url: str
|
|
model: str
|
|
# Short by design: this runs after the local attempt has already spent
|
|
# its own timeout, so it adds to a latency budget that is already over.
|
|
timeout_seconds: int = 2
|
|
api_key_env: Optional[str] = None
|
|
max_output_tokens: int = 1024
|
|
|
|
|
|
class LocalEncoderConfig(StrictModel):
|
|
"""A non-generative encoder model used for zero-shot category classification.
|
|
|
|
Embedding+centroid architecture, not NLI cross-encoding: the input runs
|
|
through the encoder once and is scored by cosine similarity against
|
|
precomputed embeddings of the description text, rather than cross-encoded
|
|
pairwise against each label.
|
|
|
|
This still structurally rules out the one failure mode that has cost this
|
|
project two classifier generations already (see docs/local-models.md):
|
|
a generative model spending its budget on an unbounded reasoning trace
|
|
and returning no parseable JSON. An encoder scored against a fixed label
|
|
set cannot produce that failure — it isn't generative, so there is no
|
|
trace to run away.
|
|
|
|
Zero-shot, not fine-tuned, and deliberately so for now: this router
|
|
never stores raw task text anywhere (see ``docs/operations.md`` and the
|
|
test that enforces it), so a supervised model has no training corpus to
|
|
learn from without a new, separate opt-in data-capture feature.
|
|
"""
|
|
|
|
# BAAI/bge-large-en-v1.5 -- a strong general-purpose English embedding
|
|
# model, CPU-viable at this size (~1.3 GB), and well-established in the
|
|
# embedding model space. This is Wave 5.1 of the token-waste plan
|
|
# (plans/token-waste-waves.md): replacing the previous NLI cross-encoding
|
|
# approach with an embedding+centroid one.
|
|
model: str = "BAAI/bge-large-en-v1.5"
|
|
device: Literal["cpu", "cuda"] = "cpu"
|
|
# minimum raw similarity score to accept the encoder's verdict
|
|
# (not a probability -- the softmax-amplified cosine similarity
|
|
# _SOFTMAX_TEMPERATURE produces is monotonic but has no probabilistic
|
|
# meaning, so the minimum is a heuristic threshold, not a confidence
|
|
# calibration). Below this threshold the classification is treated as
|
|
# a FAILURE rather than a low-confidence answer. Caught live 2026-09-06:
|
|
# the admin UI took a raw number with no conversion or bound, so typing
|
|
# the intuitive "80" broke classification on every request. The UI now
|
|
# converts 0-100 to 0.0-1.0 before saving; this validator is the
|
|
# fail-closed backstop for any other caller.
|
|
confidence_min: float = 0.5
|
|
|
|
@field_validator("confidence_min")
|
|
@classmethod
|
|
def confidence_min_in_range(cls, v: float) -> float:
|
|
if not (0.0 <= v <= 1.0):
|
|
raise ValueError(
|
|
f"classifier.encoder.confidence_min must be in [0.0, 1.0], "
|
|
f"got {v!r}"
|
|
)
|
|
return v
|
|
|
|
@model_validator(mode="before")
|
|
@classmethod
|
|
def deprecate_confidence_threshold(cls, data: dict) -> dict:
|
|
"""Accept old confidence_threshold key with a logged deprecation warning."""
|
|
if isinstance(data, dict) and "confidence_threshold" in data:
|
|
if "confidence_min" not in data:
|
|
data["confidence_min"] = data.pop("confidence_threshold")
|
|
import logs
|
|
logs.warning(
|
|
"confidence_threshold_deprecated",
|
|
key="confidence_threshold",
|
|
replacement="confidence_min",
|
|
suggestion="rename classifier.encoder.confidence_threshold "
|
|
"to confidence_min in config",
|
|
)
|
|
else:
|
|
del data["confidence_threshold"]
|
|
import logs
|
|
logs.warning(
|
|
"confidence_threshold_deprecated_both",
|
|
key="confidence_threshold",
|
|
message="both confidence_threshold and confidence_min "
|
|
"provided; using confidence_min",
|
|
)
|
|
return data
|
|
|
|
# Design sketch (see _TierFeatureClassifier in local_encoder.py): predict
|
|
# task_tier from non-textual request features (prompt token count, tools
|
|
# array length, fenced-block count, conversation depth, diff presence)
|
|
# instead of from task text. Not wired — config keys reserved for future
|
|
# implementation. No admin control because both fields are read nowhere
|
|
# outside config.py; an admin toggle would control nothing real until
|
|
# this feature (item E in the plan) is implemented.
|
|
tier_from_features: bool = False
|
|
tier_feature_fields: list[str] = Field(default_factory=list)
|
|
|
|
|
|
class LocalDecisionConfig(StrictModel):
|
|
"""A generative local model that picks a task category by choice.
|
|
|
|
Unlike the embedding+centroid ``LocalEncoderConfig``, this is a small
|
|
generative LLM asked to return one of a fixed set of category labels
|
|
(see ``classify_choice`` in local_decision.py). Confidence comes from the
|
|
model's logprobs on the chosen label, not from a similarity score.
|
|
"""
|
|
|
|
base_url: str = "http://localhost:11434"
|
|
model: str = "qwen3.5:4b"
|
|
num_ctx: int = 8192
|
|
timeout_s: int = 10
|
|
# minimum logprob-derived confidence to accept the model's verdict.
|
|
# Below this threshold the classification is treated as a FAILURE
|
|
# rather than a low-confidence answer, mirroring the encoder's
|
|
# confidence_min semantics.
|
|
confidence_min: float = 0.5
|
|
# minimum total probability mass on option letters for one call;
|
|
# below this threshold parse_logprobs raises RuntimeError and the
|
|
# classification cascades as a failure.
|
|
coverage_min: float = 0.3
|
|
# whether the classifier may also decide task_tier (vs. only category).
|
|
tier_enabled: bool = False
|
|
|
|
@field_validator("confidence_min")
|
|
@classmethod
|
|
def confidence_min_in_range(cls, v: float) -> float:
|
|
if not (0.0 <= v <= 1.0):
|
|
raise ValueError(
|
|
f"classifier.decision.confidence_min must be in [0.0, 1.0], "
|
|
f"got {v!r}"
|
|
)
|
|
return v
|
|
|
|
|
|
# Bounds for the degraded-share warning. Named, not inlined in the validators,
|
|
# because a runtime write goes straight past Pydantic and the admin registry's
|
|
# bounds are then the only check: it imports these so the runtime path and the
|
|
# load validator cannot disagree about what a legal value is. The threshold is
|
|
# exclusive of 0 (a share of 0 is always reached) and inclusive of 1.
|
|
DEGRADED_WARN_MIN_FLOOR = 1
|
|
DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE = 0.0
|
|
DEGRADED_WARN_THRESHOLD_MAX = 1.0
|
|
# Bounds for the classifier cooldown window. At least 1s, at most 1h —
|
|
# a cloud-classifier retry during a sustained local outage cannot burn
|
|
# more than one attempt per hour per process.
|
|
COOLDOWN_FLOOR = 1
|
|
COOLDOWN_CEILING = 3600
|
|
# Bounds for the fallback tier. Must correspond to a real tier in the
|
|
# routing table (tier 1 = cheap+small, tier 3 = frontier).
|
|
FALLBACK_TIER_MIN = 1
|
|
FALLBACK_TIER_MAX = 3
|
|
# Upper bound on the degradation-warning sample-size floor. A value past
|
|
# 10 000 classifications per window is a configuration mistake regardless
|
|
# of traffic volume.
|
|
DEGRADED_WARN_MIN_CEILING = 10000
|
|
# Minimum value for the degradation-warning threshold. A share below this
|
|
# fires on the very first degraded classification in the window, which is
|
|
# never useful — the operator already knows one failure happened.
|
|
DEGRADED_WARN_THRESHOLD_FLOOR = 0.001
|
|
|
|
|
|
class ClassifierConfig(StrictModel):
|
|
provider: str
|
|
base_url: str
|
|
# Env var holding the API key, for a classifier served by a provider that
|
|
# actually checks one. None means unauthenticated, which is the local
|
|
# Ollama case — it ignores the key entirely but the SDK requires one.
|
|
api_key_env: Optional[str] = None
|
|
model: str
|
|
# Ceiling on the text handed to the classifier. 0 disables clamping.
|
|
max_input_chars: int = 8000
|
|
# When a preceding turn is available as context, frame the classifier
|
|
# input as llmrouter does — "Context: <prev>\n---\nMessage: <task>" — so a
|
|
# short follow-up ("Yes", "Try now?") can inherit the complexity of the
|
|
# turn it continues instead of being classified in isolation as trivial.
|
|
context_framing: bool = True
|
|
timeout_seconds: int
|
|
temperature: float = 0.0
|
|
max_output_tokens: int = 1024
|
|
fallback_tier: int = 2
|
|
fallback_category: str = "general_chat"
|
|
response_format: str
|
|
system_prompt: str
|
|
# The labels a classifier is ALLOWED to return -- deliberately a separate
|
|
# list from proficiency.categories, which is the scoring axis.
|
|
#
|
|
# One list was serving two jobs, and they are not the same job. The
|
|
# scoring axis answers "what is this model good at"; the candidate set
|
|
# answers "what should a classifier be asked to distinguish". A category
|
|
# can be genuinely useful for the first and harmful for the second:
|
|
# tool_use_agentic describes what a turn mechanically DOES rather than
|
|
# what it is for, and every agent turn does it, so a per-turn classifier
|
|
# collapses onto it (31 of 31 consecutive live turns, 2026-09-14) and
|
|
# routing loses every other signal. The tool competence that category
|
|
# exists to protect is read from the request's own `tools` array by
|
|
# routing.min_tool_proficiency, which is exact and free -- it never
|
|
# needed the classifier's opinion.
|
|
#
|
|
# None means "every proficiency category", i.e. the old behaviour. The
|
|
# shipped config/config.yaml sets the list explicitly; see the comment
|
|
# there for what excluding a category costs.
|
|
#
|
|
# Validated as a SUBSET of proficiency.categories by RouterConfig, for
|
|
# the same reason routing.tool_use_category is: a label that joins
|
|
# against nothing in the proficiency table routes on a phantom join.
|
|
candidate_categories: Optional[list[str]] = None
|
|
# Optional cloud classifier, tried only after the local one and the two
|
|
# free session-derived steps have all failed. Absent — the default —
|
|
# means the cascade degrades straight to the static fallback and the
|
|
# whole feature costs nothing.
|
|
cloud_fallback: Optional["CloudFallbackConfig"] = None
|
|
# Global backoff after a classifier failure. This is what bounds cloud
|
|
# spend during a sustained local outage: at most one cloud attempt per
|
|
# window across ALL requests, not one per request.
|
|
cooldown_seconds: int = 30
|
|
# Degradation warning: how many classifications the window needs before
|
|
# the degraded share means anything, and the share that trips it.
|
|
degraded_warn_min: int = 20
|
|
degraded_warn_threshold: float = 0.5
|
|
|
|
# --- which implementation answers PRIMARY, as opposed to the cascade's
|
|
# fallback steps above, which are unaffected by this and still apply on
|
|
# a primary failure regardless of which mode is primary. --------------
|
|
#
|
|
# "local_llm" is every existing deployment's behavior, unchanged, and
|
|
# stays the default so this feature costs nothing until opted into.
|
|
mode: Literal["local_llm", "cloud_llm", "local_encoder", "local_decision"] = "local_llm"
|
|
# Only read when mode == "cloud_llm", and mutually exclusive with
|
|
# cloud_primary_auto (RouterConfig validator enforces exactly one).
|
|
# Reuses CloudFallbackConfig's shape verbatim rather than a near-copy —
|
|
# a pinned primary cloud classifier has the same failure-handling needs
|
|
# (short timeout, optional api_key_env) as the cascade's cloud step.
|
|
cloud_primary: Optional["CloudFallbackConfig"] = None
|
|
# "auto_classifier": resolve the cheapest currently-routable model from
|
|
# the live catalog instead of a pinned id (routing.cheapest_classifier_
|
|
# candidate). Only read when mode == "cloud_llm".
|
|
cloud_primary_auto: bool = False
|
|
# Only read when mode == "local_encoder".
|
|
encoder: Optional["LocalEncoderConfig"] = None
|
|
# Only read when mode == "local_decision".
|
|
decision: Optional["LocalDecisionConfig"] = None
|
|
|
|
# Refused at load rather than accepted and inert. A threshold above 1 can
|
|
# never be reached by a share, and one at or below 0 is always reached, so
|
|
# either reads as a working detector that has quietly stopped warning (or
|
|
# never stops). The minimum has to be a real sample size for the same
|
|
# reason: 0 would compute a share over an empty window.
|
|
@field_validator("degraded_warn_threshold")
|
|
@classmethod
|
|
def degraded_warn_threshold_in_range(cls, v: float) -> float:
|
|
if not (DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE < v <= DEGRADED_WARN_THRESHOLD_MAX):
|
|
raise ValueError(
|
|
"classifier.degraded_warn_threshold is a share of decisions "
|
|
"and must be in (0, 1]; a value above 1 can never fire"
|
|
)
|
|
return v
|
|
|
|
@field_validator("degraded_warn_min")
|
|
@classmethod
|
|
def degraded_warn_min_positive(cls, v: int) -> int:
|
|
if v < DEGRADED_WARN_MIN_FLOOR:
|
|
raise ValueError(
|
|
"classifier.degraded_warn_min must be at least "
|
|
f"{DEGRADED_WARN_MIN_FLOOR}, got {v}"
|
|
)
|
|
if v > DEGRADED_WARN_MIN_CEILING:
|
|
raise ValueError(
|
|
"classifier.degraded_warn_min must be at most "
|
|
f"{DEGRADED_WARN_MIN_CEILING}, got {v}"
|
|
)
|
|
return v
|
|
|
|
@field_validator("cooldown_seconds")
|
|
@classmethod
|
|
def cooldown_seconds_in_range(cls, v: int) -> int:
|
|
if not (COOLDOWN_FLOOR <= v <= COOLDOWN_CEILING):
|
|
raise ValueError(
|
|
"classifier.cooldown_seconds must be between "
|
|
f"{COOLDOWN_FLOOR} and {COOLDOWN_CEILING}, got {v}"
|
|
)
|
|
return v
|
|
|
|
@field_validator("fallback_tier")
|
|
@classmethod
|
|
def fallback_tier_in_range(cls, v: int) -> int:
|
|
if not (FALLBACK_TIER_MIN <= v <= FALLBACK_TIER_MAX):
|
|
raise ValueError(
|
|
"classifier.fallback_tier must be between "
|
|
f"{FALLBACK_TIER_MIN} and {FALLBACK_TIER_MAX}, got {v}"
|
|
)
|
|
return v
|
|
|
|
|
|
class LocalComputeConfig(StrictModel):
|
|
"""The outer gate over every call this router makes to local hardware.
|
|
|
|
"Gaming mode": the operator stops Ollama to give the GPU back to something
|
|
else. Without a flag the router discovers that one timeout at a time, so
|
|
this exists to be TOLD rather than to find out.
|
|
|
|
Deliberately ONE flag that the call sites read, not a macro that writes
|
|
``verification.local_llm_enabled``, ``local_vision.enabled``,
|
|
``local_energy.enabled`` and friends. A macro is hard to undo cleanly,
|
|
drifts the moment a sixth local call site appears, and leaves nobody able
|
|
to answer "why isn't the classifier running?" from one place. Those keys
|
|
keep their own meanings; this is an outer AND over all of them.
|
|
"""
|
|
|
|
enabled: bool = True
|
|
|
|
|
|
class LocalEnergyConfig(StrictModel):
|
|
"""Optional metering for the router's own local Ollama calls.
|
|
|
|
Cloud providers expose per-request energy, but classifier / verifier / local
|
|
vision run on your own hardware and the router's ledger ignores them by
|
|
default. Enabling it without a real tariff is an error — a null rate would
|
|
make every local call appear free rather than unknown.
|
|
|
|
Only call sites that resolve to a loopback hostname are metered. Pointing
|
|
local inference at a non-loopback host (VPN, another machine) skips metering
|
|
with a startup warning, because the machine being metered must be the one
|
|
running the code.
|
|
"""
|
|
|
|
enabled: bool = False
|
|
meter: Literal["nvidia_smi"] = "nvidia_smi"
|
|
sample_interval_seconds: float = 0.25
|
|
tariff_usd_per_kwh: Optional[float] = None
|
|
grid_intensity_g_per_kwh: Optional[float] = None
|
|
|
|
@field_validator("sample_interval_seconds")
|
|
@classmethod
|
|
def sample_interval_positive(cls, v: float) -> float:
|
|
if v <= 0:
|
|
raise ValueError("local_energy.sample_interval_seconds must be > 0")
|
|
return v
|
|
|
|
@field_validator("tariff_usd_per_kwh")
|
|
@classmethod
|
|
def tariff_positive(cls, v: Optional[float]) -> Optional[float]:
|
|
if v is not None and v < 0:
|
|
raise ValueError("local_energy.tariff_usd_per_kwh must be >= 0, or null")
|
|
return v
|
|
|
|
@field_validator("grid_intensity_g_per_kwh")
|
|
@classmethod
|
|
def grid_intensity_positive(cls, v: Optional[float]) -> Optional[float]:
|
|
if v is not None and v < 0:
|
|
raise ValueError("local_energy.grid_intensity_g_per_kwh must be >= 0, or null")
|
|
return v
|
|
|
|
|
|
class LocalDispatchModel(StrictModel):
|
|
"""A local (non-cloud) model that the router can dispatch to on eligible tasks.
|
|
|
|
Configured alongside cloud models in ``dispatch_providers``; entries whose
|
|
``base_url`` resolves to a loopback address are additionally metered by the
|
|
local-energy accounting path. An entry that points to a non-loopback host
|
|
emits a startup warning — local energy metering can only run on the machine
|
|
executing this code.
|
|
"""
|
|
|
|
model_id: str
|
|
base_url: str = "http://localhost:11434/v1"
|
|
api_key_env: Optional[str] = None
|
|
timeout_seconds: float = 120.0
|
|
context_window: int
|
|
max_output_tokens: int = 2048
|
|
tier: int = Field(ge=1, le=3)
|
|
eligible_categories: list[str]
|
|
|
|
@field_validator("eligible_categories")
|
|
@classmethod
|
|
def eligible_categories_non_empty(cls, v: list[str]) -> list[str]:
|
|
if not v:
|
|
raise ValueError("eligible_categories must contain at least one category")
|
|
return v
|
|
|
|
@field_validator("eligible_categories")
|
|
@classmethod
|
|
def eligible_categories_no_duplicates(cls, v: list[str]) -> list[str]:
|
|
seen: set[str] = set()
|
|
for name in v:
|
|
if name in seen:
|
|
raise ValueError(
|
|
f"eligible_categories contains duplicate: {name!r}"
|
|
)
|
|
seen.add(name)
|
|
return v
|
|
|
|
@field_validator("timeout_seconds")
|
|
@classmethod
|
|
def timeout_positive(cls, v: float) -> float:
|
|
if v <= 0:
|
|
raise ValueError("local_dispatch_models.timeout_seconds must be > 0")
|
|
return v
|
|
|
|
|
|
class DispatchProvider(StrictModel):
|
|
base_url: str
|
|
api_key_env: str
|
|
balance_url: Optional[str] = None
|
|
has_energy_telemetry: bool = False
|
|
reports_cost_in_usage: bool = False
|
|
enabled: bool = True
|
|
require_allowlist: bool = False
|
|
|
|
@field_validator("balance_url")
|
|
@classmethod
|
|
def balance_url_requires_https(cls, v: Optional[str]) -> Optional[str]:
|
|
if v is None:
|
|
return v
|
|
if urlparse(v).scheme != "https":
|
|
raise ValueError(
|
|
f"dispatch_providers[...].balance_url must use https://, got {v!r}"
|
|
)
|
|
return v
|
|
|
|
|
|
# Names-only balance-parser registry. poller imports config, so config cannot
|
|
# import poller's parser dict; the poller implements the parser and tests pin
|
|
# the two sets equal.
|
|
PROVIDERS_WITH_BALANCE_PARSERS: frozenset[str] = frozenset({"openrouter"})
|
|
|
|
|
|
class LoggingConfig(StrictModel):
|
|
# log_path is gone. Nothing ever wrote a file: the dispatcher logs to
|
|
# stderr and systemd captures that to the journal, so the setting named a
|
|
# destination that did not exist.
|
|
log_energy_observations: bool
|
|
# Whether to write a row to route_decisions for every routing decision
|
|
# (kind route | dispatch | chat | passthrough | local_vision). Off means
|
|
# the monitoring TUI's decision history is empty; it does not affect
|
|
# routing itself.
|
|
log_route_decisions: bool = True
|
|
# LLM_ROUTER_LOG_LEVEL overrides this at runtime — see logs.resolve_level.
|
|
level: str = "info"
|
|
|
|
@field_validator("level")
|
|
@classmethod
|
|
def level_known(cls, v: str) -> str:
|
|
known = ("debug", "info", "warning", "error")
|
|
if v.strip().lower() not in known:
|
|
raise ValueError(f"logging.level must be one of {known}, got {v!r}")
|
|
return v.strip().lower()
|
|
|
|
|
|
class DetectorConfig(StrictModel):
|
|
"""Heuristic thresholds for the watchdog's model-response anomaly detector.
|
|
|
|
Each threshold is a rough boundary tuned on the reference deployment's
|
|
traffic; you should expect to adjust them for your own workload.
|
|
"""
|
|
|
|
window: int = 60
|
|
dup_min: float = 0.25
|
|
top_min: int = 12
|
|
top_min_ro: int = 8
|
|
cum_min: int = 15
|
|
cover_min: float = 4.0
|
|
min_calls: int = 40
|
|
|
|
|
|
class ChannelConfig(StrictModel):
|
|
"""A single notification channel that the watchdog can alert through.
|
|
|
|
The ``type`` field is deliberately restricted — unknown types fail at
|
|
load rather than being silently ignored.
|
|
"""
|
|
|
|
name: str
|
|
type: Literal["desktop"]
|
|
enabled: bool = True
|
|
min_severity: Literal["info", "warning", "critical"] = "warning"
|
|
|
|
|
|
class NotificationsConfig(StrictModel):
|
|
"""Watchdog alert routing: which channels receive which severities."""
|
|
|
|
channels: list[ChannelConfig] = [
|
|
{"name": "default", "type": "desktop", "enabled": True, "min_severity": "warning"}
|
|
]
|
|
|
|
|
|
class WatchdogConfig(StrictModel):
|
|
"""Runtime model-response watchdog that monitors for anomalous behavior.
|
|
|
|
Watches the live decision stream for duplicate/truncated/repetitive
|
|
output patterns and sends desktop notifications when thresholds are
|
|
crossed.
|
|
"""
|
|
|
|
enabled: bool = True
|
|
local_llm_enabled: bool = True
|
|
# Defaults to verification.model at runtime when None.
|
|
model: str | None = None
|
|
read_only_agents: list[str] = ["explore", "librarian", "oracle"]
|
|
detector: DetectorConfig = DetectorConfig()
|
|
# Base URL for alert links back to the admin dashboard.
|
|
dashboard_base_url: str = "http://127.0.0.1:8080/admin"
|
|
|
|
|
|
def _is_loopback_host(raw_url: str) -> bool:
|
|
"""Return True if the URL's hostname is a loopback address.
|
|
|
|
Strips a trailing ``/v1`` path before parsing, because classifier and local
|
|
vision configs use OpenAI-compatible endpoints that look like
|
|
``http://localhost:11434/v1`` while verification uses the native
|
|
``http://localhost:11434`` root.
|
|
"""
|
|
url = raw_url.rstrip("/").removesuffix("/v1")
|
|
hostname = urlparse(url).hostname
|
|
if hostname is None:
|
|
return False
|
|
return hostname in ("localhost", "127.0.0.1", "::1")
|
|
|
|
|
|
class DispatchSettingsConfig(StrictModel):
|
|
default_provider: str = "neuralwatt"
|
|
|
|
|
|
class RouterConfig(StrictModel):
|
|
objective: Objective
|
|
context: ContextConfig
|
|
tiers: dict[int, str]
|
|
tiering: TieringConfig
|
|
proficiency: ProficiencyConfig
|
|
routing: RoutingConfig
|
|
verification: VerificationConfig = VerificationConfig()
|
|
local_vision: LocalVisionConfig = LocalVisionConfig()
|
|
escalation: EscalationConfig
|
|
iteration: IterationConfig = IterationConfig()
|
|
pinch: PinchConfig = PinchConfig()
|
|
session_cache: SessionCacheConfig = SessionCacheConfig()
|
|
circuit_breaker: CircuitBreakerConfig = CircuitBreakerConfig()
|
|
exploration: ExplorationConfig = ExplorationConfig()
|
|
freshness: FreshnessConfig
|
|
database: DatabaseConfig
|
|
classifier: ClassifierConfig
|
|
dispatch_providers: dict[str, DispatchProvider]
|
|
dispatch_settings: DispatchSettingsConfig = DispatchSettingsConfig()
|
|
logging: LoggingConfig
|
|
local_compute: LocalComputeConfig = LocalComputeConfig()
|
|
local_energy: LocalEnergyConfig = LocalEnergyConfig()
|
|
local_dispatch_models: list[LocalDispatchModel] = []
|
|
profiles: dict[str, RoutingProfile] = {}
|
|
watchdog: WatchdogConfig = WatchdogConfig()
|
|
notifications: NotificationsConfig = NotificationsConfig()
|
|
|
|
@model_validator(mode="after")
|
|
def local_energy_needs_tariff_when_enabled(self) -> "RouterConfig":
|
|
"""A missing tariff with metering enabled would silently log zeros.
|
|
|
|
The whole point of local energy accounting is cost/carbon awareness. If
|
|
the operator enables it without setting a real per-kWh rate, every local
|
|
Ollama call would record cost_usd = 0 instead of failing — that is the
|
|
same silent wrong-default bug that this codebase uses strict config to
|
|
prevent.
|
|
"""
|
|
if self.local_energy.enabled and self.local_energy.tariff_usd_per_kwh is None:
|
|
raise ValueError(
|
|
"local_energy.enabled is true but local_energy.tariff_usd_per_kwh "
|
|
"is null. Set your real per-kWh rate before enabling metering."
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def cloud_llm_mode_names_exactly_one_primary(self) -> "RouterConfig":
|
|
"""classifier.mode == "cloud_llm" needs to know WHICH cloud model.
|
|
|
|
Exactly one of cloud_primary (pinned) or cloud_primary_auto (resolve
|
|
the cheapest routable model live) must be set — not both, which would
|
|
leave "which one wins" unspecified, and not neither, which would
|
|
leave the primary classifier undefined even though a mode was chosen
|
|
that promises one.
|
|
"""
|
|
if self.classifier.mode != "cloud_llm":
|
|
return self
|
|
pinned = self.classifier.cloud_primary is not None
|
|
auto = self.classifier.cloud_primary_auto
|
|
if pinned and auto:
|
|
raise ValueError(
|
|
"classifier.mode is 'cloud_llm' with BOTH classifier.cloud_primary "
|
|
"and classifier.cloud_primary_auto set — exactly one must be "
|
|
"given so it is unambiguous which cloud model is primary."
|
|
)
|
|
if not pinned and not auto:
|
|
raise ValueError(
|
|
"classifier.mode is 'cloud_llm' but neither classifier.cloud_primary "
|
|
"nor classifier.cloud_primary_auto is set. Configure one: a pinned "
|
|
"cloud_primary block, or cloud_primary_auto: true to resolve the "
|
|
"cheapest routable model live."
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def local_encoder_mode_needs_encoder_config(self) -> "RouterConfig":
|
|
"""classifier.mode == "local_encoder" needs to know which model.
|
|
|
|
LocalEncoderConfig ships sensible defaults, so this never forces an
|
|
operator to write out every field -- but the block itself must exist,
|
|
the same way local_encoder.py refuses to guess a model id.
|
|
"""
|
|
if self.classifier.mode == "local_encoder" and self.classifier.encoder is None:
|
|
raise ValueError(
|
|
"classifier.mode is 'local_encoder' but classifier.encoder is not "
|
|
"set. Add a classifier.encoder block (its fields all have "
|
|
"defaults, so `classifier.encoder: {}` is enough)."
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def local_decision_mode_needs_decision_config(self) -> "RouterConfig":
|
|
"""classifier.mode == 'local_decision' needs to know which model.
|
|
|
|
LocalDecisionConfig ships sensible defaults, so this never forces an
|
|
operator to write out every field -- but the block itself must exist,
|
|
the same way local_encoder.py refuses to guess a model id.
|
|
"""
|
|
if self.classifier.mode == "local_decision" and self.classifier.decision is None:
|
|
raise ValueError(
|
|
"classifier.mode is 'local_decision' but classifier.decision is not "
|
|
"set. Add a classifier.decision block (its fields all have "
|
|
"defaults, so `classifier.decision: {}` is enough)."
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def gaming_mode_requires_a_cloud_classifier(self) -> "RouterConfig":
|
|
"""Local compute off with no cloud classifier is a worse router.
|
|
|
|
Turning the local classifier off does not make classification cheaper
|
|
or remote — it drops every request through stale-session, then session
|
|
history, then a fixed static guess. That guess is recorded as
|
|
``general_chat``, a fully scored category, so the traffic it mislabels
|
|
looks like real classification on the dashboard.
|
|
|
|
The rule is a refusal rather than a warning because a warning is
|
|
exactly what nobody reads while their game is loading. Refusing names
|
|
the key, and uncommenting the shipped ``classifier.cloud_fallback``
|
|
example is the intended fix — it points at the cheapest routable
|
|
model, which is the right shape for a short prompt with a short JSON
|
|
answer. Nothing here auto-writes that block: silently switching on a
|
|
paid classifier is not a favour.
|
|
"""
|
|
if self.local_compute.enabled:
|
|
return self
|
|
if self.classifier.cloud_fallback is None:
|
|
raise ValueError(
|
|
"local_compute.enabled is false but classifier.cloud_fallback "
|
|
"is not configured. Gaming mode skips the local classifier, so "
|
|
"without a cloud classifier every request degrades to a static "
|
|
"guess. Configure classifier.cloud_fallback (an example ships "
|
|
"commented out in config/config.yaml) or leave local compute on."
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def tool_use_category_is_a_real_category(self) -> "RouterConfig":
|
|
"""The tool filter joins on a category name; a typo would disable it.
|
|
|
|
A name that matches nothing produces NULL for every row, and NULL
|
|
means "unproven, do not disqualify" — so the filter would silently
|
|
pass everything. Failing at load beats a guard that quietly stops
|
|
guarding.
|
|
"""
|
|
if self.routing.min_tool_proficiency is None:
|
|
return self
|
|
if self.routing.tool_use_category not in self.proficiency.categories:
|
|
raise ValueError(
|
|
f"routing.tool_use_category "
|
|
f"({self.routing.tool_use_category!r}) is not in "
|
|
f"proficiency.categories — the tool-competence filter would "
|
|
f"join against nothing and silently pass every model."
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def classifier_candidates_are_real_categories(self) -> "RouterConfig":
|
|
"""Every classifier label must exist on the proficiency scoring axis.
|
|
|
|
The two lists are separate on purpose (see
|
|
ClassifierConfig.candidate_categories), but the separation is one
|
|
directional: the candidate set is a SUBSET of the scoring axis, never
|
|
a parallel vocabulary. A label outside proficiency.categories joins
|
|
against nothing in the proficiency table, so routing would rank on
|
|
NULL scores and POST /outcome would attribute to a category no model
|
|
is ever scored on — the same phantom-join failure
|
|
tool_use_category_is_a_real_category exists to stop, arriving from
|
|
the other side.
|
|
"""
|
|
candidates = self.classifier.candidate_categories
|
|
if candidates is None:
|
|
return self
|
|
if not candidates:
|
|
raise ValueError(
|
|
"classifier.candidate_categories is empty — the classifier "
|
|
"would have no label to return. Omit the key entirely to "
|
|
"mean 'every proficiency category'."
|
|
)
|
|
unknown = [c for c in candidates if c not in self.proficiency.categories]
|
|
if unknown:
|
|
raise ValueError(
|
|
f"classifier.candidate_categories contains "
|
|
f"{sorted(unknown)!r}, which "
|
|
f"{'are' if len(unknown) > 1 else 'is'} not in "
|
|
f"proficiency.categories — a classifier label that is not a "
|
|
f"scoring category joins against nothing in the proficiency "
|
|
f"table, so routing would rank on NULLs."
|
|
)
|
|
return self
|
|
|
|
@property
|
|
def classifier_candidate_categories(self) -> list[str]:
|
|
"""The resolved label set a classifier may return.
|
|
|
|
A property rather than a filled-in field so call sites get a concrete
|
|
``list[str]`` instead of an Optional they each have to unwrap, and so
|
|
the "None means all of them" default lives in exactly one place.
|
|
"""
|
|
if self.classifier.candidate_categories is None:
|
|
return list(self.proficiency.categories)
|
|
return list(self.classifier.candidate_categories)
|
|
|
|
@model_validator(mode="after")
|
|
def profile_names_do_not_shadow_builtins(self) -> "RouterConfig":
|
|
"""Configured profile names must not collide with built-in profiles.
|
|
|
|
Built-ins are the shared enumeration that admin.py, RouterConfig, and
|
|
dispatcher.py all consult; allowing a config profile to shadow one would
|
|
make the merge semantics order-dependent and surprise callers.
|
|
"""
|
|
reserved = set(BUILTIN_PROFILES)
|
|
for name in self.profiles:
|
|
if name in reserved:
|
|
raise ValueError(
|
|
f"profiles[{name!r}] collides with a built-in profile. "
|
|
f"Reserved names: {sorted(reserved)}"
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def default_profile_names_a_known_profile(self) -> "RouterConfig":
|
|
"""routing.default_profile must resolve to an existing profile.
|
|
|
|
Bare ``auto`` and the implicit default profile both resolve through
|
|
this name, so a typo or deletion here would silently change routing.
|
|
Valid names are the built-in profiles plus any configured ones; the
|
|
non-empty check is handled by the field-level validator above.
|
|
"""
|
|
valid = sorted(set(BUILTIN_PROFILES) | set(self.profiles))
|
|
if self.routing.default_profile not in valid:
|
|
raise ValueError(
|
|
f"routing.default_profile {self.routing.default_profile!r} "
|
|
f"is not a known profile. Valid: {valid}"
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def default_provider_is_a_configured_provider(self) -> "RouterConfig":
|
|
"""dispatch_settings.default_provider must name a configured provider.
|
|
|
|
A typo here would make every passthrough, poll, eval, and seed-energy
|
|
lookup fail under a key that does not exist in dispatch_providers. Fail
|
|
at config load instead of at runtime.
|
|
"""
|
|
provider = self.dispatch_settings.default_provider
|
|
if provider not in self.dispatch_providers:
|
|
valid = sorted(self.dispatch_providers)
|
|
raise ValueError(
|
|
f"dispatch_settings.default_provider {provider!r} "
|
|
f"is not a key in dispatch_providers. Valid: {valid}"
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def local_dispatch_categories_are_real_categories(
|
|
self,
|
|
) -> "RouterConfig":
|
|
"""Every eligible_category must be a known proficiency category.
|
|
|
|
A name that matches nothing produces NULL for every row, and NULL
|
|
means "unproven, do not disqualify" — so the filter would silently
|
|
pass everything. Failing at load beats a guard that quietly stops
|
|
guarding.
|
|
|
|
Also enforces that model_ids are unique across local dispatch entries.
|
|
"""
|
|
seen_ids: set[str] = set()
|
|
for entry in self.local_dispatch_models:
|
|
if entry.model_id in seen_ids:
|
|
raise ValueError(
|
|
f"local_dispatch_models contains duplicate model_id: "
|
|
f"{entry.model_id!r}"
|
|
)
|
|
seen_ids.add(entry.model_id)
|
|
for cat in entry.eligible_categories:
|
|
if cat not in self.proficiency.categories:
|
|
raise ValueError(
|
|
f"local_dispatch_models[{entry.model_id}].eligible_categories "
|
|
f"contains {cat!r}, which is not in "
|
|
f"proficiency.categories — the filter would join against "
|
|
f"nothing and silently pass every model."
|
|
)
|
|
return self
|
|
|
|
_dispatch_meterable_cache: frozenset[str] = PrivateAttr(default=frozenset())
|
|
_sites_cache: dict[str, bool] = PrivateAttr(default_factory=dict)
|
|
|
|
def model_post_init(self, __context: object) -> None:
|
|
"""Compute local_energy_call_sites + dispatch meterable set once at load.
|
|
|
|
Both properties return their caches and emit warnings during this
|
|
single-pass computation. This avoids the per-request warning storm
|
|
that occurred when every dispatcher request called the property.
|
|
"""
|
|
sites: dict[str, bool]
|
|
if not self.local_energy.enabled:
|
|
sites = {"classify": False, "verify": False, "local_vision": False}
|
|
else:
|
|
# classifier.base_url describes an Ollama endpoint, so it only
|
|
# answers the metering question in local_llm mode. A local_encoder
|
|
# classifier places no HTTP call at all -- the model runs
|
|
# IN-PROCESS on this host's own GPU -- so base_url describes
|
|
# nothing about where that work happens, and loopback is
|
|
# unconditionally the right answer.
|
|
#
|
|
# The distinction is load-bearing because docs/local-models.md
|
|
# recommends pointing classifier.base_url at a VPN address rather
|
|
# than 0.0.0.0. Do that while in encoder mode and the old gate
|
|
# silently stopped metering a GPU whose electricity is on this
|
|
# machine's own bill -- while warning about a URL nothing calls.
|
|
encoder_is_in_process = self.classifier.mode == "local_encoder"
|
|
classify_is_loopback = True if encoder_is_in_process else (
|
|
_is_loopback_host(self.classifier.decision.base_url)
|
|
if self.classifier.mode == "local_decision"
|
|
and self.classifier.decision is not None
|
|
else _is_loopback_host(self.classifier.base_url)
|
|
)
|
|
sites = {
|
|
"classify": classify_is_loopback,
|
|
"verify": _is_loopback_host(self.verification.base_url),
|
|
"local_vision": _is_loopback_host(self.local_vision.base_url),
|
|
}
|
|
for name, loopback in sites.items():
|
|
if not loopback:
|
|
section = {
|
|
"classify": (
|
|
self.classifier.decision
|
|
if self.classifier.mode == "local_decision"
|
|
and self.classifier.decision is not None
|
|
else self.classifier
|
|
),
|
|
"verify": self.verification,
|
|
"local_vision": self.local_vision,
|
|
}[name]
|
|
logging.getLogger("router.config").warning(
|
|
"local_energy is enabled but %s.base_url points to a "
|
|
"non-loopback host (%s); skipping local metering for that call "
|
|
"site. Local energy can only meter the machine this code runs on.",
|
|
name,
|
|
section.base_url,
|
|
)
|
|
self._sites_cache = sites
|
|
|
|
meterable_ids: list[str] = []
|
|
if self.local_energy.enabled:
|
|
for entry in self.local_dispatch_models:
|
|
if _is_loopback_host(entry.base_url):
|
|
meterable_ids.append(entry.model_id)
|
|
else:
|
|
logging.getLogger("router.config").warning(
|
|
"local_energy is enabled but local_dispatch_models[%s] "
|
|
"base_url points to a non-loopback host (%s); skipping "
|
|
"local metering for that model. Local energy can only meter "
|
|
"the machine this code runs on.",
|
|
entry.model_id,
|
|
entry.base_url,
|
|
)
|
|
self._dispatch_meterable_cache = frozenset(meterable_ids)
|
|
|
|
@property
|
|
def local_energy_call_sites(self) -> dict[str, bool]:
|
|
"""Which local-Ollama call sites are eligible for local energy metering.
|
|
|
|
Each site is checked against its configured base_url: if the resolved
|
|
hostname is loopback, the site can be wrapped by the dispatcher's local
|
|
energy meter. If local_energy.enabled is false, every value is false.
|
|
|
|
The dispatcher must still decide whether a call site is actually used
|
|
(e.g. verification.local_llm_enabled gates the verifier), but this
|
|
property tells it whether metering is *safe* for the URL.
|
|
|
|
Warning for non-loopback hosts is emitted once at config load time,
|
|
not on every property access.
|
|
"""
|
|
return self._sites_cache
|
|
|
|
@property
|
|
def local_energy_dispatch_models(self) -> frozenset[str]:
|
|
"""Local dispatch model IDs eligible for local energy metering.
|
|
|
|
Returns a frozenset of model_ids whose ``base_url`` resolves to a
|
|
loopback hostname. When ``local_energy.enabled`` is false the set is
|
|
empty — metering is only meaningful when the timer is actually running.
|
|
|
|
Warning for non-loopback hosts is emitted once at config load time,
|
|
not on every property access.
|
|
"""
|
|
return self._dispatch_meterable_cache
|
|
|
|
@model_validator(mode="after")
|
|
def balance_url_on_provider_with_parser_and_no_telemetry(
|
|
self,
|
|
) -> "RouterConfig":
|
|
"""balance_url may only be set for providers with a parser and no telemetry.
|
|
|
|
A balance_url asks the poller to query a provider account endpoint.
|
|
Providers that already report balance per completion via
|
|
has_energy_telemetry would create two contradictory sources; fail at
|
|
load rather than silently picking one. A provider whose key has no
|
|
parser implementation would fail quietly every poll cycle.
|
|
"""
|
|
for provider, prov_cfg in self.dispatch_providers.items():
|
|
if prov_cfg.balance_url is None:
|
|
continue
|
|
if getattr(prov_cfg, "has_energy_telemetry", False):
|
|
raise ValueError(
|
|
f"dispatch_providers[{provider!r}] has both "
|
|
f"balance_url and has_energy_telemetry=true. "
|
|
f"Telemetry providers already report balance per completion; "
|
|
f"configure one source or the other."
|
|
)
|
|
if provider not in PROVIDERS_WITH_BALANCE_PARSERS:
|
|
valid = sorted(PROVIDERS_WITH_BALANCE_PARSERS)
|
|
raise ValueError(
|
|
f"dispatch_providers[{provider!r}].balance_url is set, "
|
|
f"but {provider!r} has no balance parser implementation. "
|
|
f"Providers with parsers: {valid}"
|
|
)
|
|
return self
|
|
|
|
@model_validator(mode="after")
|
|
def verifier_model_is_stated_once_the_hosts_differ(self) -> "RouterConfig":
|
|
"""A remote classifier must not lend its model name to the verifier.
|
|
|
|
``verification.model`` falling back to ``classifier.model`` is correct
|
|
only while both point at the same Ollama. Once classification moves
|
|
off-host the fallback names a model the local Ollama has never heard
|
|
of, and the failure is SILENT: the verifier 404s, catches it, logs
|
|
"local verification unavailable" and records no sample. Verification
|
|
would appear to be on while producing nothing.
|
|
|
|
Caught by pointing the classifier at NeuralWatt and watching the
|
|
verifier POST ``deepseek-v4-flash`` to localhost:11434. Failing at
|
|
config load instead means the misconfiguration is impossible rather
|
|
than merely documented.
|
|
"""
|
|
if not self.verification.local_llm_enabled or self.verification.model:
|
|
return self
|
|
classifier_host = urlparse(self.classifier.base_url).hostname
|
|
verifier_host = urlparse(self.verification.base_url).hostname
|
|
if classifier_host != verifier_host:
|
|
raise ValueError(
|
|
"verification.model must be set explicitly when the classifier "
|
|
f"runs on a different host ({classifier_host} vs "
|
|
f"{verifier_host}). It would otherwise fall back to "
|
|
f"classifier.model ({self.classifier.model!r}), which the local "
|
|
"Ollama does not serve — and the verifier fails silently. "
|
|
"Set verification.model, or verification.local_llm_enabled: false."
|
|
)
|
|
return self
|
|
|
|
|
|
def _merge_overlay(base: dict, overlay: dict) -> dict:
|
|
"""Deep-merge *overlay* on top of *base* and return the merged dict.
|
|
|
|
Merge semantics:
|
|
|
|
* **Mappings** (``dict``): merged recursively, key by key. Keys present
|
|
only in *overlay* are inserted; keys only in *base* are preserved.
|
|
* **Lists** (``list``): replaced wholesale — the overlay list entirely
|
|
overwrites the base list at that key. Lists are never concatenated.
|
|
* **Scalars**: overlay value wins.
|
|
|
|
An absent overlay file leaves base unchanged, so this helper is
|
|
strictly additive. The merged dict is validated once by the caller
|
|
so that unknown keys in the overlay surface as Pydantic errors.
|
|
|
|
Precedence chain (highest → lowest): environment variables >
|
|
overlay > base (the base file).
|
|
"""
|
|
result = {}
|
|
# Start with all base keys
|
|
for key, base_val in base.items():
|
|
if (
|
|
key in overlay
|
|
and isinstance(base_val, dict)
|
|
and isinstance(overlay[key], dict)
|
|
):
|
|
# Both sides are dicts → recurse
|
|
result[key] = _merge_overlay(base_val, overlay[key])
|
|
else:
|
|
# Scalars, lists, or overlay-only keys: overlay wins (or base if absent)
|
|
result[key] = overlay.get(key, base_val)
|
|
# Insert overlay-only keys that had no counterpart in base
|
|
for key, overlay_val in overlay.items():
|
|
if key not in result:
|
|
result[key] = overlay_val
|
|
return result
|
|
|
|
|
|
def load_config(
|
|
path: str | Path = "config/config.yaml",
|
|
*,
|
|
include_overlay: Optional[bool] = None,
|
|
) -> RouterConfig:
|
|
"""Load config.yaml, merging the sibling config.local.yaml overlay.
|
|
|
|
``include_overlay=None`` (the default) consults
|
|
``ROUTER_IGNORE_LOCAL_CONFIG``; pass True or False to decide explicitly,
|
|
which is what a test exercising overlay merging on its own tmp files
|
|
wants -- that behaviour is the thing under test, not an accident of the
|
|
developer's machine.
|
|
"""
|
|
path = Path(path)
|
|
if not path.exists():
|
|
raise FileNotFoundError(f"Config file not found: {path}")
|
|
raw = yaml.safe_load(path.read_text())
|
|
|
|
# ROUTER_IGNORE_LOCAL_CONFIG exists for the test suite, and for exactly
|
|
# one reason: without it the suite reads whichever overlay the machine
|
|
# happens to have. On this deployment `classifier.mode: local_encoder` in
|
|
# config.local.yaml made nine tests fail -- five in
|
|
# test_classifier_backoff, three in test_gaming_mode, one in
|
|
# test_classifier_modes_dispatch -- with assertions like
|
|
# `assert 'local_encoder' == 'local_llm'`. Those tests are about what
|
|
# config.yaml declares, so a machine-local override is not input to them.
|
|
#
|
|
# Reproduced both directions on the same tree: overlay absent, 1523
|
|
# passed; overlay present, 9 failed. A deployment that has configured
|
|
# anything could not run its own tests clean.
|
|
#
|
|
# Deliberately an explicit opt-out rather than pytest detection: the
|
|
# skip is a decision the caller makes, and it stays visible in the
|
|
# environment rather than depending on how the process was started.
|
|
if include_overlay is None:
|
|
include_overlay = not os.environ.get("ROUTER_IGNORE_LOCAL_CONFIG")
|
|
if include_overlay:
|
|
local_path = path.with_name("config.local.yaml")
|
|
if local_path.exists():
|
|
local_raw = yaml.safe_load(local_path.read_text())
|
|
raw = _merge_overlay(raw, local_raw)
|
|
|
|
return RouterConfig(**raw)
|
|
|
|
|
|
def summary_lines(cfg: RouterConfig) -> list[str]:
|
|
"""What ``python config.py`` prints.
|
|
|
|
A function rather than inline prints so a test can pin the attribute names.
|
|
The previous version read ``cfg.weights``, which had been replaced by
|
|
``cfg.objective`` -- so the setup step documented in both README.md and
|
|
CLAUDE.md said "Config loaded OK" and then died with AttributeError on a
|
|
config that had in fact loaded perfectly.
|
|
"""
|
|
return [
|
|
"Config loaded OK",
|
|
f" objective: quality_tolerance={cfg.objective.quality_tolerance}, "
|
|
f"max_energy_per_request={cfg.objective.max_energy_per_request}, "
|
|
f"plan_kwh_per_period={cfg.objective.plan_kwh_per_period}, "
|
|
f"billing_reset_day={cfg.objective.billing_reset_day}",
|
|
f" categories: {cfg.proficiency.categories}",
|
|
# The two lists differ on purpose; printing only the scoring axis
|
|
# would hide which labels the classifier can actually return.
|
|
f" classifier labels: {cfg.classifier_candidate_categories}"
|
|
+ (
|
|
""
|
|
if len(cfg.classifier_candidate_categories)
|
|
== len(cfg.proficiency.categories)
|
|
else " (excluded: "
|
|
+ ", ".join(
|
|
c
|
|
for c in cfg.proficiency.categories
|
|
if c not in cfg.classifier_candidate_categories
|
|
)
|
|
+ ")"
|
|
),
|
|
f" classifier: {cfg.classifier.model} @ {cfg.classifier.base_url}",
|
|
f" verifier: {cfg.verification.model or cfg.classifier.model} "
|
|
f"@ {cfg.verification.base_url}"
|
|
+ ("" if cfg.verification.local_llm_enabled else " (disabled)"),
|
|
f" tool filter: min_tool_proficiency={cfg.routing.min_tool_proficiency}",
|
|
f" dispatch providers: {list(cfg.dispatch_providers)}",
|
|
]
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import sys
|
|
|
|
cfg_path = sys.argv[1] if len(sys.argv) > 1 else "config/config.yaml"
|
|
print("\n".join(summary_lines(load_config(cfg_path))))
|