Files
6krrt/src/config.py
adlee-was-taken 9aba0fa11d feat(config): add ClassifierConfig validators and named constants for knob coverage
Add COOLDOWN_FLOOR, FALLBACK_TIER_MIN, DEGRADED_WARN_MIN_CEILING, etc. constants and field_validators for cooldown_seconds, fallback_tier, degraded_warn_min. Update test_admin_config.py for category/advanced fields.

Ultraworked with Sisyphus

Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
2026-10-05 00:12:34 -04:00

2127 lines
88 KiB
Python

"""
Loads and validates config.yaml for the local LLM router.
Usage:
from config import load_config
cfg = load_config("config/config.yaml")
cfg.objective.quality_tolerance # etc.
"""
from __future__ import annotations
from enum import Enum
import logging
import os
from pathlib import Path
from typing import Literal, Optional
from urllib.parse import urlparse
import yaml
from pydantic import BaseModel, ConfigDict, Field, PrivateAttr, field_validator, model_validator
class StrictModel(BaseModel):
"""Base for every config section: an unknown key is an error.
Pydantic ignores extra keys by default, which makes a typo or a
misplaced setting silently do nothing while the file still loads and
still looks configured. That is not hypothetical here — `max_input_chars`
was written into the `verification:` block instead of `classifier:`,
where it was accepted, ignored, and had no effect. It happened to carry
the same value as the code default, so nothing visibly broke; editing it
would simply have done nothing.
Anyone tuning this file needs a wrong key to say so.
"""
model_config = ConfigDict(extra="forbid")
class CreditAttenuationConfig(StrictModel):
"""Optional cost-inflation tiebreak for providers with a polled account balance.
When enabled, a provider whose prepaid balance is near depletion gets its
comparison cost multiplied inside the quality-first ranking. Quality bands
still win; the multiplier only breaks ties. Telemetry providers (those
billed per-completion via ``allowance_remaining_usd``) are always treated
as multiplier 1.0 — their "low" reading is normal overage-invoice noise,
not a depleting pool, so attenuation never applies to them.
"""
enabled: bool = False
soft_floor_usd: float = 5.0
zero_floor_usd: float = 0.0
max_multiplier: float = 5.0
refresh_seconds: int = 300
@field_validator("soft_floor_usd")
@classmethod
def soft_floor_positive(cls, v: float) -> float:
if v <= 0:
raise ValueError("objective.credit_attenuation.soft_floor_usd must be > 0")
return v
@field_validator("zero_floor_usd")
@classmethod
def zero_floor_non_negative(cls, v: float) -> float:
if v < 0:
raise ValueError(
"objective.credit_attenuation.zero_floor_usd must be >= 0"
)
return v
@field_validator("max_multiplier")
@classmethod
def max_multiplier_above_one(cls, v: float) -> float:
if v <= 1.0:
raise ValueError(
"objective.credit_attenuation.max_multiplier must be > 1.0"
)
return v
@field_validator("refresh_seconds")
@classmethod
def refresh_seconds_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError(
"objective.credit_attenuation.refresh_seconds must be > 0"
)
return v
@model_validator(mode="after")
def zero_below_soft(self) -> "CreditAttenuationConfig":
if self.zero_floor_usd >= self.soft_floor_usd:
raise ValueError(
"objective.credit_attenuation.zero_floor_usd "
"must be < soft_floor_usd"
)
return self
class Objective(StrictModel):
"""What the router optimizes: quality, bounded by cost.
Replaced a three-way weighted blend. See config.yaml for why — briefly,
the cost weight was measured to be nearly inert while consuming 40% of
every decision.
"""
quality_tolerance: float = 0.10
plan_pace_warn_ratio: float = 1.25
assumed_cache_rate: float = 0.917
assumed_completion_tokens: int = 500
max_energy_per_request: Optional[float] = None
plan_kwh_per_period: Optional[float] = None
billing_reset_day: Optional[int] = None
quota_burn_window_hours: Optional[int] = None
quota_runway_warning_hours: Optional[int] = None
quota_burn_min_segment_samples: Optional[int] = None
quota_burn_min_segment_hours: Optional[float] = None
rejection_warning_window_hours: Optional[int] = None
rejection_warning_baseline_hours: Optional[int] = None
rejection_warning_min_count: Optional[int] = None
selection_coverage_window_hours: Optional[int] = None
cache_rate_window_hours: Optional[int] = None
cache_rate_warn_margin: Optional[float] = None
cache_rate_warn_min_observations: Optional[int] = None
# Premise-expiry checks for the quality_tolerance and cost-as-tiebreak
# justifications in config.yaml. These are /metrics warning thresholds,
# not dispatch controls — they fire once the comment's premise is clearly
# expired, so the operator knows to re-read that justification.
proficiency_depth_warn_min_samples: Optional[int] = None
proficiency_depth_warn_min_rows: Optional[int] = None
cumulative_spend_warn_usd: Optional[float] = None
cumulative_spend_warn_min_rows: Optional[int] = None
# Report-only measurement windows. Neither series is read by routing;
# see metrics.cost_estimate_calibration / metrics.latency_series.
cost_calibration_window_hours: Optional[int] = None
cost_calibration_min_observations: Optional[int] = None
latency_window_hours: Optional[int] = None
latency_min_observations: Optional[int] = None
# Gate: with this off, routing is byte-identical to today.
# Flip on only after the Wave 1 post-restart baseline day.
incumbent_cache_pricing: bool = False
# Challenger cache-rate dial.
# - None (default): neutral, follows assumed_cache_rate.
# - 0.0: challengers priced as fully cold prompts (maximum incumbent advantage).
incumbent_challenger_cache_rate: Optional[float] = None
# Seconds between refreshes of measured per-(provider, model) cache rates.
incumbent_rate_refresh_seconds: int = 300
# Minimum reported-cache observations before a per-(provider, model)
# cache rate is trusted for pricing decisions.
incumbent_rate_min_observations: int = 25
# Window (in seconds) for the conversation adoption counter in /metrics.
# Recent route_decisions rows within this window whose session_key starts
# with "c:" are counted. Null or absent means "all time" (no window);
# 0 is rejected by the adoption_window_positive validator below.
adoption_window_seconds: Optional[int] = None
credit_attenuation: CreditAttenuationConfig = CreditAttenuationConfig()
@field_validator("quality_tolerance")
@classmethod
def tolerance_in_range(cls, v: float) -> float:
if not (0.0 <= v < 1.0):
raise ValueError("objective.quality_tolerance must be in [0, 1)")
return v
@field_validator("plan_pace_warn_ratio")
@classmethod
def pace_warn_ratio_positive(cls, v: float) -> float:
if v <= 0:
raise ValueError("objective.plan_pace_warn_ratio must be > 0")
return v
@field_validator("billing_reset_day")
@classmethod
def billing_reset_day_range(cls, v: Optional[int]) -> Optional[int]:
if v is not None and not (1 <= v <= 28):
raise ValueError(
"objective.billing_reset_day must be in 1..28, or null to disable"
)
return v
@field_validator("proficiency_depth_warn_min_samples")
@classmethod
def depth_min_samples_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.proficiency_depth_warn_min_samples must be > 0"
)
return v
@field_validator("proficiency_depth_warn_min_rows")
@classmethod
def depth_min_rows_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.proficiency_depth_warn_min_rows must be > 0"
)
return v
@field_validator("cumulative_spend_warn_usd")
@classmethod
def spend_warn_usd_positive(cls, v: Optional[float]) -> Optional[float]:
if v is not None and v <= 0:
raise ValueError(
"objective.cumulative_spend_warn_usd must be > 0"
)
return v
@field_validator("cumulative_spend_warn_min_rows")
@classmethod
def spend_min_rows_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.cumulative_spend_warn_min_rows must be > 0"
)
return v
@field_validator("max_energy_per_request")
@classmethod
def ceiling_positive(cls, v: Optional[float]) -> Optional[float]:
if v is not None and v <= 0:
raise ValueError(
"objective.max_energy_per_request must be > 0 kWh, or null to disable"
)
return v
@field_validator("quota_burn_window_hours")
@classmethod
def burn_window_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.quota_burn_window_hours must be > 0"
)
return v
@field_validator("quota_runway_warning_hours")
@classmethod
def runway_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.quota_runway_warning_hours must be > 0"
)
return v
@field_validator("quota_burn_min_segment_samples")
@classmethod
def segment_samples_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.quota_burn_min_segment_samples must be > 0"
)
return v
@field_validator("quota_burn_min_segment_hours")
@classmethod
def segment_hours_positive(cls, v: Optional[float]) -> Optional[float]:
if v is not None and v <= 0:
raise ValueError(
"objective.quota_burn_min_segment_hours must be > 0"
)
return v
@field_validator("rejection_warning_window_hours")
@classmethod
def rejection_window_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.rejection_warning_window_hours must be > 0"
)
return v
@field_validator("rejection_warning_baseline_hours")
@classmethod
def rejection_baseline_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.rejection_warning_baseline_hours must be > 0"
)
return v
@field_validator("rejection_warning_min_count")
@classmethod
def rejection_min_count_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.rejection_warning_min_count must be > 0"
)
return v
@field_validator("selection_coverage_window_hours")
@classmethod
def selection_window_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.selection_coverage_window_hours must be > 0"
)
return v
@field_validator("cache_rate_window_hours")
@classmethod
def cache_rate_window_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.cache_rate_window_hours must be > 0"
)
return v
@field_validator("cache_rate_warn_margin")
@classmethod
def cache_rate_margin_in_range(cls, v: Optional[float]) -> Optional[float]:
# A margin of 0 warns on any float noise at all; a margin of 1 can
# never be exceeded, because both the measured rate and the assumed
# one live in [0, 1] and their absolute difference cannot reach 1
# without one of them being an exact 0 against an exact 1.
if v is not None and not (0.0 < v < 1.0):
raise ValueError(
"objective.cache_rate_warn_margin must be in (0, 1)"
)
return v
@field_validator("cache_rate_warn_min_observations")
@classmethod
def cache_rate_min_obs_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.cache_rate_warn_min_observations must be > 0"
)
return v
@field_validator(
"cost_calibration_window_hours",
"cost_calibration_min_observations",
"latency_window_hours",
"latency_min_observations",
)
@classmethod
def report_window_positive(cls, v: Optional[int], info) -> Optional[int]:
# One validator for four knobs of the same shape. The field name comes
# from the ValidationInfo rather than being hardcoded, so a renamed
# knob cannot keep reporting the old name -- the failure mode that
# makes a strict-config error message actively misleading.
if v is not None and v <= 0:
raise ValueError(f"objective.{info.field_name} must be > 0")
return v
@field_validator("incumbent_rate_refresh_seconds")
@classmethod
def incumbent_refresh_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("objective.incumbent_rate_refresh_seconds must be > 0")
return v
@field_validator("incumbent_challenger_cache_rate")
@classmethod
def challenger_cache_rate_in_range(cls, v: Optional[float]) -> Optional[float]:
if v is not None and not (0.0 <= v <= 1.0):
raise ValueError("objective.incumbent_challenger_cache_rate must be in [0, 1]")
return v
@field_validator("incumbent_rate_min_observations")
@classmethod
def incumbent_min_obs_non_negative(cls, v: int) -> int:
if v < 0:
raise ValueError("objective.incumbent_rate_min_observations must be >= 0")
return v
@field_validator("adoption_window_seconds")
@classmethod
def adoption_window_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.adoption_window_seconds must be > 0, or null/absent for all time"
)
return v
@model_validator(mode="after")
def _resolve_challenger_cache_rate(self) -> "Objective":
if self.incumbent_challenger_cache_rate is None:
self.incumbent_challenger_cache_rate = self.assumed_cache_rate
return self
class ContextOverride(StrictModel):
"""Per-model context handling, for a row whose real limits are known.
Typed rather than a bare ``dict`` so a typo INSIDE an override is an error
too. That is the whole point of StrictModel, and it was not true here: the
block validated, nothing read it, and config.yaml shipped a worked example
for it -- so anyone who followed that example got silence.
Both fields are optional; whichever is absent falls back to the global.
"""
safety_factor: Optional[float] = None
output_reserve_tokens: Optional[int] = None
@field_validator("safety_factor")
@classmethod
def factor_in_range(cls, v: Optional[float]) -> Optional[float]:
if v is not None and not (0.0 < v <= 1.0):
raise ValueError(
"context.per_model_overrides[...].safety_factor must be in (0, 1]"
)
return v
@field_validator("output_reserve_tokens")
@classmethod
def reserve_not_negative(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v < 0:
raise ValueError(
"context.per_model_overrides[...].output_reserve_tokens must be >= 0"
)
return v
class ContextConfig(StrictModel):
safety_factor: float
default_output_reserve_tokens: int
# Ceiling on the output reserve, as a fraction of the usable window. See
# poller.ModelRow.effective_context_window for what this exists to stop.
max_output_reserve_fraction: float
# Read by poller.ModelRow.effective_context_window.
per_model_overrides: dict[str, ContextOverride] = {}
@field_validator("safety_factor")
@classmethod
def factor_in_range(cls, v: float) -> float:
if not (0.0 < v <= 1.0):
raise ValueError("context.safety_factor must be in (0, 1]")
return v
@field_validator("max_output_reserve_fraction")
@classmethod
def reserve_fraction_in_range(cls, v: float) -> float:
# Exclusive at both ends: 0 would reserve nothing at all for the
# answer, 1 would allow a reserve that consumes the whole window --
# which is the exact failure this ceiling exists to prevent.
if not (0.0 < v < 1.0):
raise ValueError(
"context.max_output_reserve_fraction must be in (0, 1)"
)
return v
class ProficiencyConfig(StrictModel):
self_eval_min_samples: int
leaderboard_weight: float
self_eval_weight: float
categories: list[str]
# Prior strength (k) for empirical-Bayes shrunken estimate: the number of
# pseudo-observations the peer prior contributes, controlling how far a
# noisy per-model score is pulled toward the global average. 20 was chosen
# as a conservative starting point — thin self-eval data on a single model
# (n < 100) would dominate without any shrinkage; 20 pseudo-observations
# dampens that without erasing the per-model signal.
outcome_prior_strength: int = 20
@model_validator(mode="after")
def blend_weights_sum_to_one(self) -> "ProficiencyConfig":
total = round(self.leaderboard_weight + self.self_eval_weight, 6)
if total != 1.0:
raise ValueError(
"proficiency.leaderboard_weight + self_eval_weight must sum to 1.0, "
f"got {total}"
)
return self
class TieringConfig(StrictModel):
cheap_completion_max: float
tier1_context_max: float = float("inf")
model_tiers: dict[str, int]
@field_validator("cheap_completion_max")
@classmethod
def max_must_be_positive(cls, v: float) -> float:
if v <= 0:
raise ValueError("tiering.cheap_completion_max must be > 0")
return v
@field_validator("tier1_context_max")
@classmethod
def context_max_must_be_positive(cls, v: float) -> float:
if v <= 0:
raise ValueError("tiering.tier1_context_max must be > 0")
return v
@field_validator("model_tiers")
@classmethod
def overrides_in_range(cls, v: dict[str, int]) -> dict[str, int]:
for model_id, tier in v.items():
if tier not in (1, 2, 3):
raise ValueError(
f"tiering.model_tiers[{model_id!r}] must be in {{1, 2, 3}}, "
f"got {tier}"
)
return v
class FlexPreference(Enum):
"""Operator's stance on routing to ``-flex`` serving-class rows.
A 4-position scale:
- ``no-flex``: never route to a flex row.
- ``auto``: decide per request (the default).
- ``prefer-flex``: flex first, standard as fallback.
- ``force-flex``: flex only.
"""
no_flex = "no-flex"
auto = "auto"
prefer_flex = "prefer-flex"
force_flex = "force-flex"
class RoutingProfile(StrictModel):
"""A named candidate-set filter requested through ``auto:<name>``.
Profiles restrict the candidate set; they do NOT change ranking. All
fields are optional, and any absent field falls back to the request's
own value or the global default. This keeps profiles small rewrite
rules rather than a parallel config layer.
"""
provider: Optional[str] = None
min_tier: Optional[int] = None
max_tier: Optional[int] = None
latency_tolerance: Optional[Literal["interactive", "batch"]] = None
max_cost_per_1m_completion: Optional[float] = None
allowed_model_ids: Optional[set[str]] = None
@field_validator("min_tier", "max_tier")
@classmethod
def tier_in_range(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v not in (1, 2, 3):
raise ValueError(
f"profiles[...].{cls.__name__}.tier must be in {{1, 2, 3}}, got {v}"
)
return v
@model_validator(mode="after")
def min_not_above_max(self) -> "RoutingProfile":
if self.min_tier is not None and self.max_tier is not None:
if self.min_tier > self.max_tier:
raise ValueError(
f"profiles[...].{self.__class__.__name__}.min_tier ({self.min_tier}) "
f"must be <= max_tier ({self.max_tier})"
)
return self
@field_validator("max_cost_per_1m_completion")
@classmethod
def cost_positive(cls, v: Optional[float]) -> Optional[float]:
if v is not None and v <= 0:
raise ValueError(
"profiles[...].max_cost_per_1m_completion must be > 0, or null"
)
return v
@field_validator("allowed_model_ids")
@classmethod
def allowed_model_ids_non_empty(cls, v: Optional[set[str]]) -> Optional[set[str]]:
if v is not None and not v:
raise ValueError("profiles[...].allowed_model_ids must be non-empty when set")
return v
# Built-in routing profiles. Profiles are selectable as ``auto:<name>``; they
# restrict the candidate set and may override the effective latency_tolerance.
# Defined in config.py so RouterConfig validators and modules that may not
# import dispatcher (e.g., admin.py, metrics.py) can enumerate them directly.
BUILTIN_PROFILES: dict[str, RoutingProfile] = {
"default": RoutingProfile(),
"batch": RoutingProfile(latency_tolerance="batch"),
"locality": RoutingProfile(provider="ollama-local"),
"bigboybritches": RoutingProfile(min_tier=3),
"onlycheaps": RoutingProfile(max_cost_per_1m_completion=0.5),
}
class RoutingConfig(StrictModel):
allowed_access_levels: list[str]
default_latency_tolerance: str
# Operator's default stance on flex serving-class rows for requests
# that do not state one explicitly.
default_flex_preference: FlexPreference = FlexPreference.auto
# Bare ``auto`` resolves to this profile. It names a built-in profile by
# default; operators may override it via config (or the admin Controls page
# in a later wave) to change the implicit behavior of unqualified ``auto``.
default_profile: str = "default"
# Applied only when the REQUEST carries tool definitions. None disables it.
min_tool_proficiency: Optional[float] = 0.5
tool_use_category: str = "tool_use_agentic"
# Whether a request carrying image parts is hard-restricted to vision-capable
# models. A wrong guess here is a guaranteed 400, so this gates by default
# and routing fails closed when the catalog flag is unknown.
require_vision: bool = True
# Whether a response_format requiring json_object/json_schema is hard-restricted
# to JSON-mode-capable models. Same guaranteed-failure argument.
require_json_mode: bool = True
@field_validator("allowed_access_levels")
@classmethod
def levels_known(cls, v: list[str]) -> list[str]:
known = {"public", "preview", "canary"}
unknown = set(v) - known
if unknown:
raise ValueError(
f"routing.allowed_access_levels contains unknown levels {sorted(unknown)}; "
f"must be a subset of {sorted(known)}"
)
if not v:
raise ValueError("routing.allowed_access_levels must not be empty")
return v
@field_validator("default_latency_tolerance")
@classmethod
def tolerance_known(cls, v: str) -> str:
if v not in ("interactive", "batch"):
raise ValueError(
f"routing.default_latency_tolerance must be 'interactive' or 'batch', got {v!r}"
)
return v
@field_validator("default_profile")
@classmethod
def default_profile_non_empty(cls, v: str) -> str:
if not v or not v.strip():
raise ValueError("routing.default_profile must be a non-empty string")
return v
class VerificationConfig(StrictModel):
local_llm_enabled: bool = True
min_completion_tokens: int = 600
timeout_seconds: int = 60
max_output_tokens: int = 1024
outcome_attribution_window_seconds: int = 120
# The local checker's OWN endpoint, no longer derived from the
# classifier's. It speaks Ollama's NATIVE API (/api/chat, think=False),
# which no cloud provider offers, so it must keep pointing at an Ollama
# instance even when classification has been moved off this machine.
base_url: str = "http://localhost:11434"
# None means "whatever the classifier uses", which is correct only while
# both run on the same local Ollama. Set it explicitly once they diverge.
model: Optional[str] = None
class LocalVisionConfig(StrictModel):
enabled: bool = True
base_url: str = "http://localhost:11434/v1" # OpenAI-compatible (classifier shape)
api_key_env: Optional[str] = None
model: str = "qwen3-vl:4b"
timeout_seconds: int = 60
max_images: int = 4
max_image_bytes: int = 9 * 1024 * 1024 # 9 MiB, Ollama default cap
@field_validator("timeout_seconds")
@classmethod
def timeout_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("local_vision.timeout_seconds must be > 0")
return v
@field_validator("max_images")
@classmethod
def images_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("local_vision.max_images must be > 0")
return v
@field_validator("max_image_bytes")
@classmethod
def image_bytes_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("local_vision.max_image_bytes must be > 0")
return v
class EscalationConfig(StrictModel):
enabled: bool
max_tier: int
min_confidence_before_bump: float
# Off by default: the iteration budget escalates on evidence instead.
preemptive_on_low_confidence: bool = False
class IterationConfig(StrictModel):
"""A tier's budget for corrective attempts after a verification failure."""
enabled: bool = True
attempts_by_tier: dict[int, int] = {1: 0, 2: 1, 3: 2}
max_attempts_interactive: int = 1
# Maximum conversation prompt-tokens for which escalation (cache-destroying
# re-bill on a different model) is allowed. Same-model retries preserve
# the cache and are not gated. 0 = no limit (backward-compatible default).
max_rebill_prompt_tokens: int = 0
@field_validator("attempts_by_tier")
@classmethod
def attempts_sane(cls, v: dict[int, int]) -> dict[int, int]:
for tier, attempts in v.items():
if attempts < 0:
raise ValueError(f"iteration.attempts_by_tier[{tier}] must be >= 0")
if attempts > 5:
raise ValueError(
f"iteration.attempts_by_tier[{tier}]={attempts} is implausibly "
"high; each attempt spends energy against a fixed quota"
)
return v
class FreshnessConfig(StrictModel):
stale_after_days: int
exclude_stale: bool
exclude_deprecated: bool
# Seconds to wait after an admin allowlist edit before re-polling the
# catalog. Debounce, not delay: each further edit restarts the clock, so
# adding five models one at a time costs one poll rather than five.
# 0 disables the trigger entirely and the scheduled timer stays the only
# path. See admin._schedule_catalog_poll.
repoll_after_allowlist_change_seconds: float = 0.0
@field_validator("repoll_after_allowlist_change_seconds")
@classmethod
def repoll_non_negative(cls, v: float) -> float:
if v < 0:
raise ValueError(
"freshness.repoll_after_allowlist_change_seconds must be >= 0"
)
return v
class PinchRelevanceConfig(StrictModel):
"""Embedding-model relevance scoring for pinch trimming.
When enabled (and ``pinch.enabled`` is also true), the dispatcher embeds
the current-turn query with the old tool-result candidates and trims the
least relevant first, so a relevant-but-old result survives. On by
default; any failure reverts to uniform trimming. This must point at an
EMBEDDING model, never ``classifier.model`` or ``verification.model``.
"""
enabled: bool = True
# OpenAI-compatible embeddings endpoint on the same local Ollama.
model: str = "nomic-embed-text"
base_url: str = "http://localhost:11434/v1"
timeout_seconds: int = 10
# Below this many trim-eligible candidates, skip the embedding round-trip.
min_candidates: int = 2
@field_validator("timeout_seconds")
@classmethod
def timeout_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("pinch.relevance.timeout_seconds must be > 0")
return v
@field_validator("min_candidates")
@classmethod
def min_candidates_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("pinch.relevance.min_candidates must be > 0")
return v
class PinchConfig(StrictModel):
"""Optional relevance-based context pruning (Port of llmrouter's pinch).
Prunes the provider-bound conversation — not the classifier input — when it
exceeds ``budget_tokens``, so a long agent session ships fewer prompt tokens
upstream. User/assistant/system messages are always kept; only tool results
are summarized or dropped (they carry the bulk of a long session's tokens).
"""
enabled: bool = True
budget_tokens: int = 50000
# How many recent user turns (plus their assistant replies and tool results)
# are protected from pruning.
keep_last_turns: int = 4
# Tool results longer than this many characters are summarized in place.
max_summarize_chars: int = 4000
# keep_last_turns has no size limit inside it -- an entire autonomous
# tool-call loop with no new user message can be one protected turn, and
# one outsized tool result inside it (a full verbose test run, a huge
# file read) ships verbatim regardless of size. Measured live: a
# 324k-token conversation shrank only ~8% because nearly all of it sat
# inside the protected window. This is deliberately a much higher bar
# than max_summarize_chars -- recent results are more likely to still
# matter -- so it only catches true outliers. None disables it.
protected_max_chars: Optional[int] = 20000
# Record, on each decision row, where this turn's pruned payload stopped
# matching the previous turn's in the same session -- the provider bills
# the longest byte-identical prefix at the cached rate, so a divergence
# early in the payload re-bills everything after it. HASHES ONLY, held in
# process memory for one turn; what reaches route_decisions is three
# integers. Measured on the live median payload (98k tokens in, 74k out):
# ~1.0 ms per turn, against a request path whose floor is a provider
# round-trip of 1.4-2.0 s. On by default because a probe that is off
# measures nothing, and Wave 3 of plans/token-waste-waves.md is waiting on
# what it says.
prefix_probe: bool = True
relevance: PinchRelevanceConfig = PinchRelevanceConfig()
@field_validator("budget_tokens")
@classmethod
def budget_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("pinch.budget_tokens must be > 0")
return v
@field_validator("keep_last_turns")
@classmethod
def turns_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("pinch.keep_last_turns must be > 0")
return v
@field_validator("max_summarize_chars")
@classmethod
def summarize_chars_valid(cls, v: int) -> int:
if v < 3000:
raise ValueError(
"pinch.max_summarize_chars must be >= 3000 (below this, "
"summarization grows the message)"
)
return v
@field_validator("protected_max_chars")
@classmethod
def protected_max_chars_valid(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v < 3000:
raise ValueError(
"pinch.protected_max_chars must be >= 3000, or null to "
"disable (below this, elision grows the message, same "
"floor as max_summarize_chars)"
)
return v
# The legal range for ``session_cache.staleness_seconds``, kept as module
# constants because three surfaces have to agree on it: this validator (file
# and overlay edits), ``admin._INT_KNOBS`` (the runtime write, which bypasses
# every Pydantic validator), and the number field's min/max in
# ``admin/frontend/controls.html``. Two of the three import these; the third is
# pinned by a test.
#
# The floor is 5, not 0, and that is the pre-existing ``> 0`` rule rather than
# a new opinion. It matters more than it looks: 0 would make ``session_cache.get``
# miss on every turn, which LOOKS like "stop reusing labels" but is not —
# ``session_cache.put`` still writes, and the classifier-failure cascade's
# ``stale_read`` ignores staleness entirely, so a 0-second window still replays
# a session's label whenever the classifier is down. The knob that actually
# stops reuse is ``session_cache.enabled``, and it has its own control.
#
# The ceiling is a judgement, and the judgement is that an unbounded window is
# a cache that never expires. 7200 is 6x the shipped default and ~8.5x the
# longest single-classification run measured on live traffic (107 consecutive
# turns across 840 seconds), so it is far above any value there is
# a reason to try, while still guaranteeing a label cannot outlive the working
# session that produced it. See CLAUDE.md, north star rule 2.
STALENESS_SECONDS_MIN = 5
STALENESS_SECONDS_MAX = 7200
class SessionCacheConfig(StrictModel):
"""Per-session classification cache (in-memory, process lifetime).
Remembers the last task_category/task_tier decision for each session for
``staleness_seconds``, so a long agent session skips the classifier
round-trip on every turn. Capability flags (tools/images/json) are NEVER
cached — they are read fresh from each request body. Fallback
classifications are NEVER cached. No persistence: the cache lives only in
process memory, so a restart just reclassifies once per session.
"""
enabled: bool = False
# Seconds since a cached classification was written before it is treated
# as expired and the next turn reclassifies from scratch.
staleness_seconds: int = 1200
@field_validator("staleness_seconds")
@classmethod
def staleness_in_range(cls, v: int) -> int:
if not (STALENESS_SECONDS_MIN <= v <= STALENESS_SECONDS_MAX):
raise ValueError(
f"session_cache.staleness_seconds must be between "
f"{STALENESS_SECONDS_MIN} and {STALENESS_SECONDS_MAX} "
f"(to stop reusing classifications set session_cache.enabled "
f"to false; 0 is not that, it still caches and the "
f"classifier-failure cascade still replays the entry)"
)
return v
class ExplorationConfig(StrictModel):
"""Epsilon-greedy exploration over ranked candidates.
On the epsilon share of requests, the router picks the hard-filter-eligible
candidate with the fewest outcome samples (tie-break: lowest cost) instead
of the highest-score model. This injects exploration into a system that
would otherwise converge on a single winner, which is how exposure bias
creeps in — the model that happens to be sampled more looks better, and
gets sampled more. Off by default: ship it, watch route_decisions
with ``was_exploration=True`` on real traffic, then enable it.
"""
enabled: bool = True
# Probability of exploration per request. 0 = never explore, 1 = always
# explore. 0.03 means ~3% of requests take an exploratory path. Above 1.0
# is an error — epsilon-greedy with epsilon > 1 is silently self-defeating;
# it stops being "mostly-exploit" and becomes random selection.
epsilon: float = 0.03
# Maximum cost ratio for an exploration candidate. The explorer picks the
# least-sampled eligible row; this caps its cost at this multiple of the
# ranking winner so an unproven frontier row doesn't eat the explore budget.
# A ratio below 1.0 is an error — you cannot go below the winner's own cost.
max_cost_ratio: float = 4.0
# Tier ceiling for exploration: only models in tiers 1..max_tier are
# eligible for the explore path. Tier 2 is a safety gate — you do not
# want exploration spending frontier budgets on random, unproven calls.
max_tier: int = 2
@field_validator("epsilon")
@classmethod
def epsilon_in_range(cls, v: float) -> float:
if not (0.0 <= v <= 1.0):
raise ValueError(
"exploration.epsilon must be in [0, 1], got "
f"{v} — above 1, epsilon-greedy stops being "
"mostly-exploit and becomes random selection"
)
return v
@field_validator("max_cost_ratio")
@classmethod
def cost_ratio_positive(cls, v: float) -> float:
if v < 1.0:
raise ValueError(
"exploration.max_cost_ratio must be >= 1.0, got "
f"{v} — a ratio below the winner's own cost "
"would always reject the exploration candidate"
)
return v
@field_validator("max_tier")
@classmethod
def tier_in_range(cls, v: int) -> int:
if v not in (1, 2, 3):
raise ValueError(
"exploration.max_tier must be in {1, 2, 3}, "
f"got {v}"
)
return v
class CircuitBreakerConfig(StrictModel):
"""Passive circuit breaker for upstream model availability.
When enabled, a model that returns 5xx is temporarily skipped by routing
(with exponential backoff). Recovery is passive: a real request that would
have picked it becomes the probe once the cooldown passes. On by default.
"""
enabled: bool = True
initial_cooldown_seconds: int = 30
max_cooldown_seconds: int = 600
backoff_multiplier: float = 2.0
@field_validator("initial_cooldown_seconds")
@classmethod
def initial_positive(cls, v: int) -> int:
if v <= 0:
raise ValueError("circuit_breaker.initial_cooldown_seconds must be > 0")
return v
@field_validator("max_cooldown_seconds")
@classmethod
def max_at_least_initial(cls, v: int) -> int:
if v <= 0:
raise ValueError("circuit_breaker.max_cooldown_seconds must be > 0")
return v
@model_validator(mode="after")
def max_gte_initial(self) -> "CircuitBreakerConfig":
if self.max_cooldown_seconds < self.initial_cooldown_seconds:
raise ValueError(
"circuit_breaker.max_cooldown_seconds must be >= "
"initial_cooldown_seconds"
)
return self
@field_validator("backoff_multiplier")
@classmethod
def multiplier_above_one(cls, v: float) -> float:
if v <= 1.0:
raise ValueError("circuit_breaker.backoff_multiplier must be > 1.0")
return v
class DatabaseConfig(StrictModel):
path: str
class CloudFallbackConfig(StrictModel):
"""A cloud classifier used ONLY when the local one is unreachable.
Deliberately its own block rather than reusing ``ClassifierConfig``: the
two differ in the ways that matter under failure. This one has a short
timeout because it sits on the latency floor of every request, and a
missing ``api_key_env`` degrades to the next cascade step instead of
raising — an optional fallback that fails the request would be worse than
the outage it exists to soften.
"""
base_url: str
model: str
# Short by design: this runs after the local attempt has already spent
# its own timeout, so it adds to a latency budget that is already over.
timeout_seconds: int = 2
api_key_env: Optional[str] = None
max_output_tokens: int = 1024
class LocalEncoderConfig(StrictModel):
"""A non-generative encoder model used for zero-shot category classification.
Embedding+centroid architecture, not NLI cross-encoding: the input runs
through the encoder once and is scored by cosine similarity against
precomputed embeddings of the description text, rather than cross-encoded
pairwise against each label.
This still structurally rules out the one failure mode that has cost this
project two classifier generations already (see docs/local-models.md):
a generative model spending its budget on an unbounded reasoning trace
and returning no parseable JSON. An encoder scored against a fixed label
set cannot produce that failure — it isn't generative, so there is no
trace to run away.
Zero-shot, not fine-tuned, and deliberately so for now: this router
never stores raw task text anywhere (see ``docs/operations.md`` and the
test that enforces it), so a supervised model has no training corpus to
learn from without a new, separate opt-in data-capture feature.
"""
# BAAI/bge-large-en-v1.5 -- a strong general-purpose English embedding
# model, CPU-viable at this size (~1.3 GB), and well-established in the
# embedding model space. This is Wave 5.1 of the token-waste plan
# (plans/token-waste-waves.md): replacing the previous NLI cross-encoding
# approach with an embedding+centroid one.
model: str = "BAAI/bge-large-en-v1.5"
device: Literal["cpu", "cuda"] = "cpu"
# minimum raw similarity score to accept the encoder's verdict
# (not a probability -- the softmax-amplified cosine similarity
# _SOFTMAX_TEMPERATURE produces is monotonic but has no probabilistic
# meaning, so the minimum is a heuristic threshold, not a confidence
# calibration). Below this threshold the classification is treated as
# a FAILURE rather than a low-confidence answer. Caught live 2026-09-06:
# the admin UI took a raw number with no conversion or bound, so typing
# the intuitive "80" broke classification on every request. The UI now
# converts 0-100 to 0.0-1.0 before saving; this validator is the
# fail-closed backstop for any other caller.
confidence_min: float = 0.5
@field_validator("confidence_min")
@classmethod
def confidence_min_in_range(cls, v: float) -> float:
if not (0.0 <= v <= 1.0):
raise ValueError(
f"classifier.encoder.confidence_min must be in [0.0, 1.0], "
f"got {v!r}"
)
return v
@model_validator(mode="before")
@classmethod
def deprecate_confidence_threshold(cls, data: dict) -> dict:
"""Accept old confidence_threshold key with a logged deprecation warning."""
if isinstance(data, dict) and "confidence_threshold" in data:
if "confidence_min" not in data:
data["confidence_min"] = data.pop("confidence_threshold")
import logs
logs.warning(
"confidence_threshold_deprecated",
key="confidence_threshold",
replacement="confidence_min",
suggestion="rename classifier.encoder.confidence_threshold "
"to confidence_min in config",
)
else:
del data["confidence_threshold"]
import logs
logs.warning(
"confidence_threshold_deprecated_both",
key="confidence_threshold",
message="both confidence_threshold and confidence_min "
"provided; using confidence_min",
)
return data
# Design sketch (see _TierFeatureClassifier in local_encoder.py): predict
# task_tier from non-textual request features (prompt token count, tools
# array length, fenced-block count, conversation depth, diff presence)
# instead of from task text. Not wired — config keys reserved for future
# implementation. No admin control because both fields are read nowhere
# outside config.py; an admin toggle would control nothing real until
# this feature (item E in the plan) is implemented.
tier_from_features: bool = False
tier_feature_fields: list[str] = Field(default_factory=list)
class LocalDecisionConfig(StrictModel):
"""A generative local model that picks a task category by choice.
Unlike the embedding+centroid ``LocalEncoderConfig``, this is a small
generative LLM asked to return one of a fixed set of category labels
(see ``classify_choice`` in local_decision.py). Confidence comes from the
model's logprobs on the chosen label, not from a similarity score.
"""
base_url: str = "http://localhost:11434"
model: str = "qwen3.5:4b"
num_ctx: int = 8192
timeout_s: int = 10
# minimum logprob-derived confidence to accept the model's verdict.
# Below this threshold the classification is treated as a FAILURE
# rather than a low-confidence answer, mirroring the encoder's
# confidence_min semantics.
confidence_min: float = 0.5
# minimum total probability mass on option letters for one call;
# below this threshold parse_logprobs raises RuntimeError and the
# classification cascades as a failure.
coverage_min: float = 0.3
# whether the classifier may also decide task_tier (vs. only category).
tier_enabled: bool = False
@field_validator("confidence_min")
@classmethod
def confidence_min_in_range(cls, v: float) -> float:
if not (0.0 <= v <= 1.0):
raise ValueError(
f"classifier.decision.confidence_min must be in [0.0, 1.0], "
f"got {v!r}"
)
return v
# Bounds for the degraded-share warning. Named, not inlined in the validators,
# because a runtime write goes straight past Pydantic and the admin registry's
# bounds are then the only check: it imports these so the runtime path and the
# load validator cannot disagree about what a legal value is. The threshold is
# exclusive of 0 (a share of 0 is always reached) and inclusive of 1.
DEGRADED_WARN_MIN_FLOOR = 1
DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE = 0.0
DEGRADED_WARN_THRESHOLD_MAX = 1.0
# Bounds for the classifier cooldown window. At least 1s, at most 1h —
# a cloud-classifier retry during a sustained local outage cannot burn
# more than one attempt per hour per process.
COOLDOWN_FLOOR = 1
COOLDOWN_CEILING = 3600
# Bounds for the fallback tier. Must correspond to a real tier in the
# routing table (tier 1 = cheap+small, tier 3 = frontier).
FALLBACK_TIER_MIN = 1
FALLBACK_TIER_MAX = 3
# Upper bound on the degradation-warning sample-size floor. A value past
# 10 000 classifications per window is a configuration mistake regardless
# of traffic volume.
DEGRADED_WARN_MIN_CEILING = 10000
# Minimum value for the degradation-warning threshold. A share below this
# fires on the very first degraded classification in the window, which is
# never useful — the operator already knows one failure happened.
DEGRADED_WARN_THRESHOLD_FLOOR = 0.001
class ClassifierConfig(StrictModel):
provider: str
base_url: str
# Env var holding the API key, for a classifier served by a provider that
# actually checks one. None means unauthenticated, which is the local
# Ollama case — it ignores the key entirely but the SDK requires one.
api_key_env: Optional[str] = None
model: str
# Ceiling on the text handed to the classifier. 0 disables clamping.
max_input_chars: int = 8000
# When a preceding turn is available as context, frame the classifier
# input as llmrouter does — "Context: <prev>\n---\nMessage: <task>" — so a
# short follow-up ("Yes", "Try now?") can inherit the complexity of the
# turn it continues instead of being classified in isolation as trivial.
context_framing: bool = True
timeout_seconds: int
temperature: float = 0.0
max_output_tokens: int = 1024
fallback_tier: int = 2
fallback_category: str = "general_chat"
response_format: str
system_prompt: str
# The labels a classifier is ALLOWED to return -- deliberately a separate
# list from proficiency.categories, which is the scoring axis.
#
# One list was serving two jobs, and they are not the same job. The
# scoring axis answers "what is this model good at"; the candidate set
# answers "what should a classifier be asked to distinguish". A category
# can be genuinely useful for the first and harmful for the second:
# tool_use_agentic describes what a turn mechanically DOES rather than
# what it is for, and every agent turn does it, so a per-turn classifier
# collapses onto it (31 of 31 consecutive live turns, 2026-09-14) and
# routing loses every other signal. The tool competence that category
# exists to protect is read from the request's own `tools` array by
# routing.min_tool_proficiency, which is exact and free -- it never
# needed the classifier's opinion.
#
# None means "every proficiency category", i.e. the old behaviour. The
# shipped config/config.yaml sets the list explicitly; see the comment
# there for what excluding a category costs.
#
# Validated as a SUBSET of proficiency.categories by RouterConfig, for
# the same reason routing.tool_use_category is: a label that joins
# against nothing in the proficiency table routes on a phantom join.
candidate_categories: Optional[list[str]] = None
# Optional cloud classifier, tried only after the local one and the two
# free session-derived steps have all failed. Absent — the default —
# means the cascade degrades straight to the static fallback and the
# whole feature costs nothing.
cloud_fallback: Optional["CloudFallbackConfig"] = None
# Global backoff after a classifier failure. This is what bounds cloud
# spend during a sustained local outage: at most one cloud attempt per
# window across ALL requests, not one per request.
cooldown_seconds: int = 30
# Degradation warning: how many classifications the window needs before
# the degraded share means anything, and the share that trips it.
degraded_warn_min: int = 20
degraded_warn_threshold: float = 0.5
# --- which implementation answers PRIMARY, as opposed to the cascade's
# fallback steps above, which are unaffected by this and still apply on
# a primary failure regardless of which mode is primary. --------------
#
# "local_llm" is every existing deployment's behavior, unchanged, and
# stays the default so this feature costs nothing until opted into.
mode: Literal["local_llm", "cloud_llm", "local_encoder", "local_decision"] = "local_llm"
# Only read when mode == "cloud_llm", and mutually exclusive with
# cloud_primary_auto (RouterConfig validator enforces exactly one).
# Reuses CloudFallbackConfig's shape verbatim rather than a near-copy —
# a pinned primary cloud classifier has the same failure-handling needs
# (short timeout, optional api_key_env) as the cascade's cloud step.
cloud_primary: Optional["CloudFallbackConfig"] = None
# "auto_classifier": resolve the cheapest currently-routable model from
# the live catalog instead of a pinned id (routing.cheapest_classifier_
# candidate). Only read when mode == "cloud_llm".
cloud_primary_auto: bool = False
# Only read when mode == "local_encoder".
encoder: Optional["LocalEncoderConfig"] = None
# Only read when mode == "local_decision".
decision: Optional["LocalDecisionConfig"] = None
# Refused at load rather than accepted and inert. A threshold above 1 can
# never be reached by a share, and one at or below 0 is always reached, so
# either reads as a working detector that has quietly stopped warning (or
# never stops). The minimum has to be a real sample size for the same
# reason: 0 would compute a share over an empty window.
@field_validator("degraded_warn_threshold")
@classmethod
def degraded_warn_threshold_in_range(cls, v: float) -> float:
if not (DEGRADED_WARN_THRESHOLD_MIN_EXCLUSIVE < v <= DEGRADED_WARN_THRESHOLD_MAX):
raise ValueError(
"classifier.degraded_warn_threshold is a share of decisions "
"and must be in (0, 1]; a value above 1 can never fire"
)
return v
@field_validator("degraded_warn_min")
@classmethod
def degraded_warn_min_positive(cls, v: int) -> int:
if v < DEGRADED_WARN_MIN_FLOOR:
raise ValueError(
"classifier.degraded_warn_min must be at least "
f"{DEGRADED_WARN_MIN_FLOOR}, got {v}"
)
if v > DEGRADED_WARN_MIN_CEILING:
raise ValueError(
"classifier.degraded_warn_min must be at most "
f"{DEGRADED_WARN_MIN_CEILING}, got {v}"
)
return v
@field_validator("cooldown_seconds")
@classmethod
def cooldown_seconds_in_range(cls, v: int) -> int:
if not (COOLDOWN_FLOOR <= v <= COOLDOWN_CEILING):
raise ValueError(
"classifier.cooldown_seconds must be between "
f"{COOLDOWN_FLOOR} and {COOLDOWN_CEILING}, got {v}"
)
return v
@field_validator("fallback_tier")
@classmethod
def fallback_tier_in_range(cls, v: int) -> int:
if not (FALLBACK_TIER_MIN <= v <= FALLBACK_TIER_MAX):
raise ValueError(
"classifier.fallback_tier must be between "
f"{FALLBACK_TIER_MIN} and {FALLBACK_TIER_MAX}, got {v}"
)
return v
class LocalComputeConfig(StrictModel):
"""The outer gate over every call this router makes to local hardware.
"Gaming mode": the operator stops Ollama to give the GPU back to something
else. Without a flag the router discovers that one timeout at a time, so
this exists to be TOLD rather than to find out.
Deliberately ONE flag that the call sites read, not a macro that writes
``verification.local_llm_enabled``, ``local_vision.enabled``,
``local_energy.enabled`` and friends. A macro is hard to undo cleanly,
drifts the moment a sixth local call site appears, and leaves nobody able
to answer "why isn't the classifier running?" from one place. Those keys
keep their own meanings; this is an outer AND over all of them.
"""
enabled: bool = True
class LocalEnergyConfig(StrictModel):
"""Optional metering for the router's own local Ollama calls.
Cloud providers expose per-request energy, but classifier / verifier / local
vision run on your own hardware and the router's ledger ignores them by
default. Enabling it without a real tariff is an error — a null rate would
make every local call appear free rather than unknown.
Only call sites that resolve to a loopback hostname are metered. Pointing
local inference at a non-loopback host (VPN, another machine) skips metering
with a startup warning, because the machine being metered must be the one
running the code.
"""
enabled: bool = False
meter: Literal["nvidia_smi"] = "nvidia_smi"
sample_interval_seconds: float = 0.25
tariff_usd_per_kwh: Optional[float] = None
grid_intensity_g_per_kwh: Optional[float] = None
@field_validator("sample_interval_seconds")
@classmethod
def sample_interval_positive(cls, v: float) -> float:
if v <= 0:
raise ValueError("local_energy.sample_interval_seconds must be > 0")
return v
@field_validator("tariff_usd_per_kwh")
@classmethod
def tariff_positive(cls, v: Optional[float]) -> Optional[float]:
if v is not None and v < 0:
raise ValueError("local_energy.tariff_usd_per_kwh must be >= 0, or null")
return v
@field_validator("grid_intensity_g_per_kwh")
@classmethod
def grid_intensity_positive(cls, v: Optional[float]) -> Optional[float]:
if v is not None and v < 0:
raise ValueError("local_energy.grid_intensity_g_per_kwh must be >= 0, or null")
return v
class LocalDispatchModel(StrictModel):
"""A local (non-cloud) model that the router can dispatch to on eligible tasks.
Configured alongside cloud models in ``dispatch_providers``; entries whose
``base_url`` resolves to a loopback address are additionally metered by the
local-energy accounting path. An entry that points to a non-loopback host
emits a startup warning — local energy metering can only run on the machine
executing this code.
"""
model_id: str
base_url: str = "http://localhost:11434/v1"
api_key_env: Optional[str] = None
timeout_seconds: float = 120.0
context_window: int
max_output_tokens: int = 2048
tier: int = Field(ge=1, le=3)
eligible_categories: list[str]
@field_validator("eligible_categories")
@classmethod
def eligible_categories_non_empty(cls, v: list[str]) -> list[str]:
if not v:
raise ValueError("eligible_categories must contain at least one category")
return v
@field_validator("eligible_categories")
@classmethod
def eligible_categories_no_duplicates(cls, v: list[str]) -> list[str]:
seen: set[str] = set()
for name in v:
if name in seen:
raise ValueError(
f"eligible_categories contains duplicate: {name!r}"
)
seen.add(name)
return v
@field_validator("timeout_seconds")
@classmethod
def timeout_positive(cls, v: float) -> float:
if v <= 0:
raise ValueError("local_dispatch_models.timeout_seconds must be > 0")
return v
class DispatchProvider(StrictModel):
base_url: str
api_key_env: str
balance_url: Optional[str] = None
has_energy_telemetry: bool = False
reports_cost_in_usage: bool = False
enabled: bool = True
require_allowlist: bool = False
@field_validator("balance_url")
@classmethod
def balance_url_requires_https(cls, v: Optional[str]) -> Optional[str]:
if v is None:
return v
if urlparse(v).scheme != "https":
raise ValueError(
f"dispatch_providers[...].balance_url must use https://, got {v!r}"
)
return v
# Names-only balance-parser registry. poller imports config, so config cannot
# import poller's parser dict; the poller implements the parser and tests pin
# the two sets equal.
PROVIDERS_WITH_BALANCE_PARSERS: frozenset[str] = frozenset({"openrouter"})
class LoggingConfig(StrictModel):
# log_path is gone. Nothing ever wrote a file: the dispatcher logs to
# stderr and systemd captures that to the journal, so the setting named a
# destination that did not exist.
log_energy_observations: bool
# Whether to write a row to route_decisions for every routing decision
# (kind route | dispatch | chat | passthrough | local_vision). Off means
# the monitoring TUI's decision history is empty; it does not affect
# routing itself.
log_route_decisions: bool = True
# LLM_ROUTER_LOG_LEVEL overrides this at runtime — see logs.resolve_level.
level: str = "info"
@field_validator("level")
@classmethod
def level_known(cls, v: str) -> str:
known = ("debug", "info", "warning", "error")
if v.strip().lower() not in known:
raise ValueError(f"logging.level must be one of {known}, got {v!r}")
return v.strip().lower()
class DetectorConfig(StrictModel):
"""Heuristic thresholds for the watchdog's model-response anomaly detector.
Each threshold is a rough boundary tuned on the reference deployment's
traffic; you should expect to adjust them for your own workload.
"""
window: int = 60
dup_min: float = 0.25
top_min: int = 12
top_min_ro: int = 8
cum_min: int = 15
cover_min: float = 4.0
min_calls: int = 40
class ChannelConfig(StrictModel):
"""A single notification channel that the watchdog can alert through.
The ``type`` field is deliberately restricted — unknown types fail at
load rather than being silently ignored.
"""
name: str
type: Literal["desktop"]
enabled: bool = True
min_severity: Literal["info", "warning", "critical"] = "warning"
class NotificationsConfig(StrictModel):
"""Watchdog alert routing: which channels receive which severities."""
channels: list[ChannelConfig] = [
{"name": "default", "type": "desktop", "enabled": True, "min_severity": "warning"}
]
class WatchdogConfig(StrictModel):
"""Runtime model-response watchdog that monitors for anomalous behavior.
Watches the live decision stream for duplicate/truncated/repetitive
output patterns and sends desktop notifications when thresholds are
crossed.
"""
enabled: bool = True
local_llm_enabled: bool = True
# Defaults to verification.model at runtime when None.
model: str | None = None
read_only_agents: list[str] = ["explore", "librarian", "oracle"]
detector: DetectorConfig = DetectorConfig()
# Base URL for alert links back to the admin dashboard.
dashboard_base_url: str = "http://127.0.0.1:8080/admin"
def _is_loopback_host(raw_url: str) -> bool:
"""Return True if the URL's hostname is a loopback address.
Strips a trailing ``/v1`` path before parsing, because classifier and local
vision configs use OpenAI-compatible endpoints that look like
``http://localhost:11434/v1`` while verification uses the native
``http://localhost:11434`` root.
"""
url = raw_url.rstrip("/").removesuffix("/v1")
hostname = urlparse(url).hostname
if hostname is None:
return False
return hostname in ("localhost", "127.0.0.1", "::1")
class DispatchSettingsConfig(StrictModel):
default_provider: str = "neuralwatt"
class RouterConfig(StrictModel):
objective: Objective
context: ContextConfig
tiers: dict[int, str]
tiering: TieringConfig
proficiency: ProficiencyConfig
routing: RoutingConfig
verification: VerificationConfig = VerificationConfig()
local_vision: LocalVisionConfig = LocalVisionConfig()
escalation: EscalationConfig
iteration: IterationConfig = IterationConfig()
pinch: PinchConfig = PinchConfig()
session_cache: SessionCacheConfig = SessionCacheConfig()
circuit_breaker: CircuitBreakerConfig = CircuitBreakerConfig()
exploration: ExplorationConfig = ExplorationConfig()
freshness: FreshnessConfig
database: DatabaseConfig
classifier: ClassifierConfig
dispatch_providers: dict[str, DispatchProvider]
dispatch_settings: DispatchSettingsConfig = DispatchSettingsConfig()
logging: LoggingConfig
local_compute: LocalComputeConfig = LocalComputeConfig()
local_energy: LocalEnergyConfig = LocalEnergyConfig()
local_dispatch_models: list[LocalDispatchModel] = []
profiles: dict[str, RoutingProfile] = {}
watchdog: WatchdogConfig = WatchdogConfig()
notifications: NotificationsConfig = NotificationsConfig()
@model_validator(mode="after")
def local_energy_needs_tariff_when_enabled(self) -> "RouterConfig":
"""A missing tariff with metering enabled would silently log zeros.
The whole point of local energy accounting is cost/carbon awareness. If
the operator enables it without setting a real per-kWh rate, every local
Ollama call would record cost_usd = 0 instead of failing — that is the
same silent wrong-default bug that this codebase uses strict config to
prevent.
"""
if self.local_energy.enabled and self.local_energy.tariff_usd_per_kwh is None:
raise ValueError(
"local_energy.enabled is true but local_energy.tariff_usd_per_kwh "
"is null. Set your real per-kWh rate before enabling metering."
)
return self
@model_validator(mode="after")
def cloud_llm_mode_names_exactly_one_primary(self) -> "RouterConfig":
"""classifier.mode == "cloud_llm" needs to know WHICH cloud model.
Exactly one of cloud_primary (pinned) or cloud_primary_auto (resolve
the cheapest routable model live) must be set — not both, which would
leave "which one wins" unspecified, and not neither, which would
leave the primary classifier undefined even though a mode was chosen
that promises one.
"""
if self.classifier.mode != "cloud_llm":
return self
pinned = self.classifier.cloud_primary is not None
auto = self.classifier.cloud_primary_auto
if pinned and auto:
raise ValueError(
"classifier.mode is 'cloud_llm' with BOTH classifier.cloud_primary "
"and classifier.cloud_primary_auto set — exactly one must be "
"given so it is unambiguous which cloud model is primary."
)
if not pinned and not auto:
raise ValueError(
"classifier.mode is 'cloud_llm' but neither classifier.cloud_primary "
"nor classifier.cloud_primary_auto is set. Configure one: a pinned "
"cloud_primary block, or cloud_primary_auto: true to resolve the "
"cheapest routable model live."
)
return self
@model_validator(mode="after")
def local_encoder_mode_needs_encoder_config(self) -> "RouterConfig":
"""classifier.mode == "local_encoder" needs to know which model.
LocalEncoderConfig ships sensible defaults, so this never forces an
operator to write out every field -- but the block itself must exist,
the same way local_encoder.py refuses to guess a model id.
"""
if self.classifier.mode == "local_encoder" and self.classifier.encoder is None:
raise ValueError(
"classifier.mode is 'local_encoder' but classifier.encoder is not "
"set. Add a classifier.encoder block (its fields all have "
"defaults, so `classifier.encoder: {}` is enough)."
)
return self
@model_validator(mode="after")
def local_decision_mode_needs_decision_config(self) -> "RouterConfig":
"""classifier.mode == 'local_decision' needs to know which model.
LocalDecisionConfig ships sensible defaults, so this never forces an
operator to write out every field -- but the block itself must exist,
the same way local_encoder.py refuses to guess a model id.
"""
if self.classifier.mode == "local_decision" and self.classifier.decision is None:
raise ValueError(
"classifier.mode is 'local_decision' but classifier.decision is not "
"set. Add a classifier.decision block (its fields all have "
"defaults, so `classifier.decision: {}` is enough)."
)
return self
@model_validator(mode="after")
def gaming_mode_requires_a_cloud_classifier(self) -> "RouterConfig":
"""Local compute off with no cloud classifier is a worse router.
Turning the local classifier off does not make classification cheaper
or remote — it drops every request through stale-session, then session
history, then a fixed static guess. That guess is recorded as
``general_chat``, a fully scored category, so the traffic it mislabels
looks like real classification on the dashboard.
The rule is a refusal rather than a warning because a warning is
exactly what nobody reads while their game is loading. Refusing names
the key, and uncommenting the shipped ``classifier.cloud_fallback``
example is the intended fix — it points at the cheapest routable
model, which is the right shape for a short prompt with a short JSON
answer. Nothing here auto-writes that block: silently switching on a
paid classifier is not a favour.
"""
if self.local_compute.enabled:
return self
if self.classifier.cloud_fallback is None:
raise ValueError(
"local_compute.enabled is false but classifier.cloud_fallback "
"is not configured. Gaming mode skips the local classifier, so "
"without a cloud classifier every request degrades to a static "
"guess. Configure classifier.cloud_fallback (an example ships "
"commented out in config/config.yaml) or leave local compute on."
)
return self
@model_validator(mode="after")
def tool_use_category_is_a_real_category(self) -> "RouterConfig":
"""The tool filter joins on a category name; a typo would disable it.
A name that matches nothing produces NULL for every row, and NULL
means "unproven, do not disqualify" — so the filter would silently
pass everything. Failing at load beats a guard that quietly stops
guarding.
"""
if self.routing.min_tool_proficiency is None:
return self
if self.routing.tool_use_category not in self.proficiency.categories:
raise ValueError(
f"routing.tool_use_category "
f"({self.routing.tool_use_category!r}) is not in "
f"proficiency.categories — the tool-competence filter would "
f"join against nothing and silently pass every model."
)
return self
@model_validator(mode="after")
def classifier_candidates_are_real_categories(self) -> "RouterConfig":
"""Every classifier label must exist on the proficiency scoring axis.
The two lists are separate on purpose (see
ClassifierConfig.candidate_categories), but the separation is one
directional: the candidate set is a SUBSET of the scoring axis, never
a parallel vocabulary. A label outside proficiency.categories joins
against nothing in the proficiency table, so routing would rank on
NULL scores and POST /outcome would attribute to a category no model
is ever scored on — the same phantom-join failure
tool_use_category_is_a_real_category exists to stop, arriving from
the other side.
"""
candidates = self.classifier.candidate_categories
if candidates is None:
return self
if not candidates:
raise ValueError(
"classifier.candidate_categories is empty — the classifier "
"would have no label to return. Omit the key entirely to "
"mean 'every proficiency category'."
)
unknown = [c for c in candidates if c not in self.proficiency.categories]
if unknown:
raise ValueError(
f"classifier.candidate_categories contains "
f"{sorted(unknown)!r}, which "
f"{'are' if len(unknown) > 1 else 'is'} not in "
f"proficiency.categories — a classifier label that is not a "
f"scoring category joins against nothing in the proficiency "
f"table, so routing would rank on NULLs."
)
return self
@property
def classifier_candidate_categories(self) -> list[str]:
"""The resolved label set a classifier may return.
A property rather than a filled-in field so call sites get a concrete
``list[str]`` instead of an Optional they each have to unwrap, and so
the "None means all of them" default lives in exactly one place.
"""
if self.classifier.candidate_categories is None:
return list(self.proficiency.categories)
return list(self.classifier.candidate_categories)
@model_validator(mode="after")
def profile_names_do_not_shadow_builtins(self) -> "RouterConfig":
"""Configured profile names must not collide with built-in profiles.
Built-ins are the shared enumeration that admin.py, RouterConfig, and
dispatcher.py all consult; allowing a config profile to shadow one would
make the merge semantics order-dependent and surprise callers.
"""
reserved = set(BUILTIN_PROFILES)
for name in self.profiles:
if name in reserved:
raise ValueError(
f"profiles[{name!r}] collides with a built-in profile. "
f"Reserved names: {sorted(reserved)}"
)
return self
@model_validator(mode="after")
def default_profile_names_a_known_profile(self) -> "RouterConfig":
"""routing.default_profile must resolve to an existing profile.
Bare ``auto`` and the implicit default profile both resolve through
this name, so a typo or deletion here would silently change routing.
Valid names are the built-in profiles plus any configured ones; the
non-empty check is handled by the field-level validator above.
"""
valid = sorted(set(BUILTIN_PROFILES) | set(self.profiles))
if self.routing.default_profile not in valid:
raise ValueError(
f"routing.default_profile {self.routing.default_profile!r} "
f"is not a known profile. Valid: {valid}"
)
return self
@model_validator(mode="after")
def default_provider_is_a_configured_provider(self) -> "RouterConfig":
"""dispatch_settings.default_provider must name a configured provider.
A typo here would make every passthrough, poll, eval, and seed-energy
lookup fail under a key that does not exist in dispatch_providers. Fail
at config load instead of at runtime.
"""
provider = self.dispatch_settings.default_provider
if provider not in self.dispatch_providers:
valid = sorted(self.dispatch_providers)
raise ValueError(
f"dispatch_settings.default_provider {provider!r} "
f"is not a key in dispatch_providers. Valid: {valid}"
)
return self
@model_validator(mode="after")
def local_dispatch_categories_are_real_categories(
self,
) -> "RouterConfig":
"""Every eligible_category must be a known proficiency category.
A name that matches nothing produces NULL for every row, and NULL
means "unproven, do not disqualify" — so the filter would silently
pass everything. Failing at load beats a guard that quietly stops
guarding.
Also enforces that model_ids are unique across local dispatch entries.
"""
seen_ids: set[str] = set()
for entry in self.local_dispatch_models:
if entry.model_id in seen_ids:
raise ValueError(
f"local_dispatch_models contains duplicate model_id: "
f"{entry.model_id!r}"
)
seen_ids.add(entry.model_id)
for cat in entry.eligible_categories:
if cat not in self.proficiency.categories:
raise ValueError(
f"local_dispatch_models[{entry.model_id}].eligible_categories "
f"contains {cat!r}, which is not in "
f"proficiency.categories — the filter would join against "
f"nothing and silently pass every model."
)
return self
_dispatch_meterable_cache: frozenset[str] = PrivateAttr(default=frozenset())
_sites_cache: dict[str, bool] = PrivateAttr(default_factory=dict)
def model_post_init(self, __context: object) -> None:
"""Compute local_energy_call_sites + dispatch meterable set once at load.
Both properties return their caches and emit warnings during this
single-pass computation. This avoids the per-request warning storm
that occurred when every dispatcher request called the property.
"""
sites: dict[str, bool]
if not self.local_energy.enabled:
sites = {"classify": False, "verify": False, "local_vision": False}
else:
# classifier.base_url describes an Ollama endpoint, so it only
# answers the metering question in local_llm mode. A local_encoder
# classifier places no HTTP call at all -- the model runs
# IN-PROCESS on this host's own GPU -- so base_url describes
# nothing about where that work happens, and loopback is
# unconditionally the right answer.
#
# The distinction is load-bearing because docs/local-models.md
# recommends pointing classifier.base_url at a VPN address rather
# than 0.0.0.0. Do that while in encoder mode and the old gate
# silently stopped metering a GPU whose electricity is on this
# machine's own bill -- while warning about a URL nothing calls.
encoder_is_in_process = self.classifier.mode == "local_encoder"
classify_is_loopback = True if encoder_is_in_process else (
_is_loopback_host(self.classifier.decision.base_url)
if self.classifier.mode == "local_decision"
and self.classifier.decision is not None
else _is_loopback_host(self.classifier.base_url)
)
sites = {
"classify": classify_is_loopback,
"verify": _is_loopback_host(self.verification.base_url),
"local_vision": _is_loopback_host(self.local_vision.base_url),
}
for name, loopback in sites.items():
if not loopback:
section = {
"classify": (
self.classifier.decision
if self.classifier.mode == "local_decision"
and self.classifier.decision is not None
else self.classifier
),
"verify": self.verification,
"local_vision": self.local_vision,
}[name]
logging.getLogger("router.config").warning(
"local_energy is enabled but %s.base_url points to a "
"non-loopback host (%s); skipping local metering for that call "
"site. Local energy can only meter the machine this code runs on.",
name,
section.base_url,
)
self._sites_cache = sites
meterable_ids: list[str] = []
if self.local_energy.enabled:
for entry in self.local_dispatch_models:
if _is_loopback_host(entry.base_url):
meterable_ids.append(entry.model_id)
else:
logging.getLogger("router.config").warning(
"local_energy is enabled but local_dispatch_models[%s] "
"base_url points to a non-loopback host (%s); skipping "
"local metering for that model. Local energy can only meter "
"the machine this code runs on.",
entry.model_id,
entry.base_url,
)
self._dispatch_meterable_cache = frozenset(meterable_ids)
@property
def local_energy_call_sites(self) -> dict[str, bool]:
"""Which local-Ollama call sites are eligible for local energy metering.
Each site is checked against its configured base_url: if the resolved
hostname is loopback, the site can be wrapped by the dispatcher's local
energy meter. If local_energy.enabled is false, every value is false.
The dispatcher must still decide whether a call site is actually used
(e.g. verification.local_llm_enabled gates the verifier), but this
property tells it whether metering is *safe* for the URL.
Warning for non-loopback hosts is emitted once at config load time,
not on every property access.
"""
return self._sites_cache
@property
def local_energy_dispatch_models(self) -> frozenset[str]:
"""Local dispatch model IDs eligible for local energy metering.
Returns a frozenset of model_ids whose ``base_url`` resolves to a
loopback hostname. When ``local_energy.enabled`` is false the set is
empty — metering is only meaningful when the timer is actually running.
Warning for non-loopback hosts is emitted once at config load time,
not on every property access.
"""
return self._dispatch_meterable_cache
@model_validator(mode="after")
def balance_url_on_provider_with_parser_and_no_telemetry(
self,
) -> "RouterConfig":
"""balance_url may only be set for providers with a parser and no telemetry.
A balance_url asks the poller to query a provider account endpoint.
Providers that already report balance per completion via
has_energy_telemetry would create two contradictory sources; fail at
load rather than silently picking one. A provider whose key has no
parser implementation would fail quietly every poll cycle.
"""
for provider, prov_cfg in self.dispatch_providers.items():
if prov_cfg.balance_url is None:
continue
if getattr(prov_cfg, "has_energy_telemetry", False):
raise ValueError(
f"dispatch_providers[{provider!r}] has both "
f"balance_url and has_energy_telemetry=true. "
f"Telemetry providers already report balance per completion; "
f"configure one source or the other."
)
if provider not in PROVIDERS_WITH_BALANCE_PARSERS:
valid = sorted(PROVIDERS_WITH_BALANCE_PARSERS)
raise ValueError(
f"dispatch_providers[{provider!r}].balance_url is set, "
f"but {provider!r} has no balance parser implementation. "
f"Providers with parsers: {valid}"
)
return self
@model_validator(mode="after")
def verifier_model_is_stated_once_the_hosts_differ(self) -> "RouterConfig":
"""A remote classifier must not lend its model name to the verifier.
``verification.model`` falling back to ``classifier.model`` is correct
only while both point at the same Ollama. Once classification moves
off-host the fallback names a model the local Ollama has never heard
of, and the failure is SILENT: the verifier 404s, catches it, logs
"local verification unavailable" and records no sample. Verification
would appear to be on while producing nothing.
Caught by pointing the classifier at NeuralWatt and watching the
verifier POST ``deepseek-v4-flash`` to localhost:11434. Failing at
config load instead means the misconfiguration is impossible rather
than merely documented.
"""
if not self.verification.local_llm_enabled or self.verification.model:
return self
classifier_host = urlparse(self.classifier.base_url).hostname
verifier_host = urlparse(self.verification.base_url).hostname
if classifier_host != verifier_host:
raise ValueError(
"verification.model must be set explicitly when the classifier "
f"runs on a different host ({classifier_host} vs "
f"{verifier_host}). It would otherwise fall back to "
f"classifier.model ({self.classifier.model!r}), which the local "
"Ollama does not serve — and the verifier fails silently. "
"Set verification.model, or verification.local_llm_enabled: false."
)
return self
def _merge_overlay(base: dict, overlay: dict) -> dict:
"""Deep-merge *overlay* on top of *base* and return the merged dict.
Merge semantics:
* **Mappings** (``dict``): merged recursively, key by key. Keys present
only in *overlay* are inserted; keys only in *base* are preserved.
* **Lists** (``list``): replaced wholesale — the overlay list entirely
overwrites the base list at that key. Lists are never concatenated.
* **Scalars**: overlay value wins.
An absent overlay file leaves base unchanged, so this helper is
strictly additive. The merged dict is validated once by the caller
so that unknown keys in the overlay surface as Pydantic errors.
Precedence chain (highest → lowest): environment variables >
overlay > base (the base file).
"""
result = {}
# Start with all base keys
for key, base_val in base.items():
if (
key in overlay
and isinstance(base_val, dict)
and isinstance(overlay[key], dict)
):
# Both sides are dicts → recurse
result[key] = _merge_overlay(base_val, overlay[key])
else:
# Scalars, lists, or overlay-only keys: overlay wins (or base if absent)
result[key] = overlay.get(key, base_val)
# Insert overlay-only keys that had no counterpart in base
for key, overlay_val in overlay.items():
if key not in result:
result[key] = overlay_val
return result
def load_config(
path: str | Path = "config/config.yaml",
*,
include_overlay: Optional[bool] = None,
) -> RouterConfig:
"""Load config.yaml, merging the sibling config.local.yaml overlay.
``include_overlay=None`` (the default) consults
``ROUTER_IGNORE_LOCAL_CONFIG``; pass True or False to decide explicitly,
which is what a test exercising overlay merging on its own tmp files
wants -- that behaviour is the thing under test, not an accident of the
developer's machine.
"""
path = Path(path)
if not path.exists():
raise FileNotFoundError(f"Config file not found: {path}")
raw = yaml.safe_load(path.read_text())
# ROUTER_IGNORE_LOCAL_CONFIG exists for the test suite, and for exactly
# one reason: without it the suite reads whichever overlay the machine
# happens to have. On this deployment `classifier.mode: local_encoder` in
# config.local.yaml made nine tests fail -- five in
# test_classifier_backoff, three in test_gaming_mode, one in
# test_classifier_modes_dispatch -- with assertions like
# `assert 'local_encoder' == 'local_llm'`. Those tests are about what
# config.yaml declares, so a machine-local override is not input to them.
#
# Reproduced both directions on the same tree: overlay absent, 1523
# passed; overlay present, 9 failed. A deployment that has configured
# anything could not run its own tests clean.
#
# Deliberately an explicit opt-out rather than pytest detection: the
# skip is a decision the caller makes, and it stays visible in the
# environment rather than depending on how the process was started.
if include_overlay is None:
include_overlay = not os.environ.get("ROUTER_IGNORE_LOCAL_CONFIG")
if include_overlay:
local_path = path.with_name("config.local.yaml")
if local_path.exists():
local_raw = yaml.safe_load(local_path.read_text())
raw = _merge_overlay(raw, local_raw)
return RouterConfig(**raw)
def summary_lines(cfg: RouterConfig) -> list[str]:
"""What ``python config.py`` prints.
A function rather than inline prints so a test can pin the attribute names.
The previous version read ``cfg.weights``, which had been replaced by
``cfg.objective`` -- so the setup step documented in both README.md and
CLAUDE.md said "Config loaded OK" and then died with AttributeError on a
config that had in fact loaded perfectly.
"""
return [
"Config loaded OK",
f" objective: quality_tolerance={cfg.objective.quality_tolerance}, "
f"max_energy_per_request={cfg.objective.max_energy_per_request}, "
f"plan_kwh_per_period={cfg.objective.plan_kwh_per_period}, "
f"billing_reset_day={cfg.objective.billing_reset_day}",
f" categories: {cfg.proficiency.categories}",
# The two lists differ on purpose; printing only the scoring axis
# would hide which labels the classifier can actually return.
f" classifier labels: {cfg.classifier_candidate_categories}"
+ (
""
if len(cfg.classifier_candidate_categories)
== len(cfg.proficiency.categories)
else " (excluded: "
+ ", ".join(
c
for c in cfg.proficiency.categories
if c not in cfg.classifier_candidate_categories
)
+ ")"
),
f" classifier: {cfg.classifier.model} @ {cfg.classifier.base_url}",
f" verifier: {cfg.verification.model or cfg.classifier.model} "
f"@ {cfg.verification.base_url}"
+ ("" if cfg.verification.local_llm_enabled else " (disabled)"),
f" tool filter: min_tool_proficiency={cfg.routing.min_tool_proficiency}",
f" dispatch providers: {list(cfg.dispatch_providers)}",
]
if __name__ == "__main__":
import sys
cfg_path = sys.argv[1] if len(sys.argv) > 1 else "config/config.yaml"
print("\n".join(summary_lines(load_config(cfg_path))))