Files
6krrt/tests/test_admin_knob_coverage.py
2026-10-05 01:28:39 -04:00

812 lines
37 KiB
Python

"""Tripwire: an operator-facing config knob reaches the admin portal on purpose.
Wave 2 of ``plans/token-waste-waves.md`` shipped ``objective.incumbent_cache_pricing``
and ``objective.incumbent_challenger_cache_rate`` with no admin control. Nobody
decided they should not have one; it simply never came up. That matters more
than usual for a dial whose whole rationale is tuning from neutral to full
penalty WITHOUT reverting code — if turning it means hand-editing a tracked file
and restarting the service, the tuning loop is too slow to walk.
So this file is the forcing function, in the shape ``test_tui_schema_drift`` and
``test_tui_warnings`` already established here: two registries and a failure
message that NAMES the thing, so whoever broke it learns what they broke without
reading the test.
- covered — ``admin._CONFIG_ALLOWLIST`` (persisted), the runtime registries
``_BOOL_KNOBS``, ``_FLOAT_KNOBS``, ``_INT_KNOBS`` (in-memory), and
``_CARD_BACKED_PATHS`` (dedicated admin cards). Derived, never hand-copied:
a hand-copy is one more thing to drift.
- ``DELIBERATELY_NOT_IN_ADMIN`` — in-scope knobs with no control, each with a
REASON STRING. The escape clause is load-bearing rather than hedging:
``objective.credit_attenuation.enabled`` is deliberately off the allowlist
and off provider edits, because enabling it must be a config edit plus a
restart (CLAUDE.md, "Routing notes"). The rule is not "everything must have a
control" — it is that the ABSENCE of one is a decision someone made. A bare
name with no reason is indistinguishable from an oversight, which is the one
thing this test exists to eliminate, so reasons are asserted non-empty.
Both directions are checked. A knob that is neither covered nor excused fails
naming the knob; an excused knob that has quietly GAINED a control fails too,
because otherwise the excuse list rots into a rubber stamp.
Scope — the judgement call, drawn explicitly so a reader can tell why a key is
out of scope rather than merely absent
--------------------------------------------------------------------------
``RouterConfig`` has ~130 scalar leaves. Demanding a decision on all of them
produces a baseline nobody reads, which IS the rubber stamp this is guarding
against. Two clauses cut it to the knobs an operator would plausibly turn:
1. **Section already reached by the portal.** A top-level section is in scope
iff at least one knob under it already has an admin control. The rule is
derived from the registries, not listed here, and it reads: where the portal
reaches, it must reach COMPLETELY. Today that is objective, routing,
verification, pinch, session_cache, circuit_breaker, local_compute, logging.
Out by this clause, and why: ``database`` (a file path), ``dispatch_providers``
/ ``dispatch_settings`` / ``local_dispatch_models`` (provider and model
definitions), ``profiles`` (its own CRUD surface, not a scalar knob),
``local_energy`` (machine-specific, lives in the gitignored overlay by
design), ``tiers`` / ``tiering`` / ``proficiency`` / ``context`` (catalog and
scoring structure), and ``classifier`` — the last has its
own dedicated admin card and endpoint pair (``/admin/api/classifier-config``,
``/admin/api/cloud-fallback-config``), so measuring it against the GENERIC
allowlist would report every field as missing while the card covers it.
``local_vision`` is reached now that ``local_vision.enabled`` has a persisted
control; the rest of the section (``base_url``, ``api_key_env``, ``model``)
is deployment wiring out by clause 2, and the timeout/image limits are
excused in ``DELIBERATELY_NOT_IN_ADMIN``.
``classifier`` has three coverage sources. **Registry knobs** (``_BOOL_KNOBS``,
``_INT_KNOBS``, ``_FLOAT_KNOBS``): runtime toggles like context_framing,
cooldown_seconds and fallback_tier. **Allowlist knobs** (``_CONFIG_ALLOWLIST``):
the same seven persisted to the overlay for restart durability. **Card-backed
paths** (``_CARD_BACKED_PATHS``): knobs with their own dedicated admin card
and endpoint pair (classifier mode, cloud primary, cloud fallback, encoder
and decision fields). The remaining classifier scalars (timeout_seconds,
temperature, max_output_tokens and encoder.tier_from_features) are wiring or
reproducibility settings excused by name in ``DELIBERATELY_NOT_IN_ADMIN``.
The known limitation: a brand-new section with no control at all is out of
scope and unchecked. The first control added under it drags every one of its
knobs into scope at once, which is the intended moment to decide.
2. **Not deployment wiring.** Within an in-scope section, leaves named for an
endpoint, a model, a credential, a path or a device are configuration of
WHERE the router points, not of how it behaves, and they are deliberately
off the allowlist already (admin.py's comment above ``_CONFIG_ALLOWLIST``).
Containers (dict/list/set) are out: the brief is scalar knobs, and every
container here is a structural mapping with its own editing surface.
Offline and pure: it reflects over the Pydantic model and the admin registries.
No database, no network, no config file is read or written.
"""
from __future__ import annotations
import enum
import sqlite3
import types
import typing
from pathlib import Path
from pydantic import BaseModel
from starlette.testclient import TestClient
import admin
import dispatcher
from config import RouterConfig, load_config
# Clause 2 of the scope rule. A leaf with one of these names configures WHERE
# the router points (endpoint, model id, credential env var, filesystem path,
# accelerator) rather than how it behaves, and admin.py already keeps this
# whole class off the persisted allowlist on purpose.
DEPLOYMENT_LEAF_NAMES = frozenset(
{
"api_key_env",
"base_url",
"device",
"meter",
"model",
"model_id",
"path",
"provider",
"response_format",
"system_prompt",
}
)
# In-scope knobs with NO admin control, each with the reason. Add here only
# when the absence is a decision; if it is merely undone work, do the work.
DELIBERATELY_NOT_IN_ADMIN: dict[str, str] = {
# `objective.incumbent_cache_pricing` and
# `objective.incumbent_challenger_cache_rate` were listed here while this
# test was written against origin/main, where their controls did not exist
# yet. On merging feat/admin-incumbent-knobs the exactness check below
# failed on both by name and demanded their removal, which is exactly what
# it is for. They are covered now, so they are gone.
# --- config-edit-plus-restart on purpose --------------------------------
"objective.credit_attenuation.enabled": (
"deliberately absent from the admin allowlist and from provider edits: "
"turning it on must be a config.yaml edit plus a restart, so that "
"biasing routing by provider balance is never a one-click change "
"(CLAUDE.md, 'Routing notes')."
),
"objective.credit_attenuation.soft_floor_usd": (
"inert unless credit_attenuation.enabled, which is config-only by "
"design; exposing the shape of a switched-off feature is surface "
"without a decision behind it."
),
"objective.credit_attenuation.zero_floor_usd": (
"inert unless credit_attenuation.enabled, which is config-only by design."
),
"objective.credit_attenuation.max_multiplier": (
"inert unless credit_attenuation.enabled, which is config-only by design."
),
"objective.credit_attenuation.refresh_seconds": (
"inert unless credit_attenuation.enabled, which is config-only by design."
),
"routing.require_vision": (
"fail-closed capability gate. Turning it off routes image requests to "
"models that cannot see them, and the failure surfaces as a provider "
"error rather than as a routing refusal — a deliberate config edit, not "
"a toggle to flip while debugging."
),
"routing.require_json_mode": (
"fail-closed capability gate, same asymmetry as require_vision: a "
"request asking for JSON must not silently reach a model that cannot "
"emit it."
),
"routing.tool_use_category": (
"names a proficiency category, and a name matching nothing yields NULL "
"for every row, which means 'do not disqualify' — the filter would "
"silently stop filtering. Config load validates it against the real "
"category list; that check is the right place for it, not a web form."
),
"routing.min_tool_proficiency": (
"held at null pending the POST /outcome experiment described in "
"CLAUDE.md, 'Tool competence is read from the request'. Flipping it is "
"an experiment with a recorded protocol, not an operational dial."
),
# --- the decision is the master switch, which IS covered -----------------
"routing.default_latency_tolerance": (
"only the default for requests that omit the field; every agent client "
"sends it per request, so the config value decides almost nothing."
),
# `session_cache.staleness_seconds` was listed here as "TTL internal to a
# covered master switch". That excuse did not survive north star rule 2:
# the window is not an implementation detail of the switch, it is the size
# of the concession -- 96.6% of classifications are replays, and one
# classification drove 107 consecutive turns. The only choices the portal
# offered were a 1200-second window or no cache at all, and the answer is
# almost certainly in between. It has both controls now, so it is gone.
"circuit_breaker.initial_cooldown_seconds": (
"backoff shape internal to a covered master switch "
"(circuit_breaker.enabled)."
),
"circuit_breaker.max_cooldown_seconds": (
"backoff shape internal to a covered master switch "
"(circuit_breaker.enabled)."
),
"circuit_breaker.backoff_multiplier": (
"backoff shape internal to a covered master switch "
"(circuit_breaker.enabled)."
),
"verification.min_completion_tokens": (
"the pay-off threshold for the local checker, fitted to the measured "
"~600-token break-even in CLAUDE.md. The operator's decision is whether "
"the local check runs at all (verification.local_llm_enabled, covered)."
),
"verification.timeout_seconds": (
"tuning for the configured local checker, which is itself off the "
"allowlist as deployment wiring; the two are changed together."
),
"verification.max_output_tokens": (
"tuning for the configured local checker, changed with the model."
),
"verification.outcome_attribution_window_seconds": (
"how long a POST /outcome report may lag its decision. An attribution "
"rule, not an operating dial: widening it from a web form would "
"retroactively change which reports count."
),
"pinch.budget_tokens": (
"pruning shape internal to a covered master switch (pinch.enabled). "
"Changing it live would also change context between turns of one "
"conversation, which is a restart-scoped decision."
),
"pinch.keep_last_turns": (
"pruning shape internal to a covered master switch (pinch.enabled)."
),
"pinch.max_summarize_chars": (
"pruning shape internal to a covered master switch (pinch.enabled)."
),
"pinch.protected_max_chars": (
"pruning shape internal to a covered master switch (pinch.enabled); see "
"docs/pinch.md before changing how much prefix context is guarded."
),
"pinch.relevance.timeout_seconds": (
"tuning for the configured local embedding model, which is off the "
"allowlist as deployment wiring; the two are changed together."
),
"pinch.relevance.min_candidates": (
"relevance-path shape internal to a covered master switch "
"(pinch.relevance.enabled)."
),
# --- local_vision: the master switch (local_vision.enabled) is covered;
# --- the timeout and image limits are tuning for the configured local
# --- Ollama vision model, changed with the deployment it points at.
"local_vision.timeout_seconds": (
"tuning for the configured local vision model; the operator's decision "
"is whether the fallback runs at all (local_vision.enabled, covered)."
),
"local_vision.max_images": (
"request-shape limit for the local vision fallback; tuning for the "
"configured local model, changed with the deployment."
),
"local_vision.max_image_bytes": (
"request-shape limit for the local vision fallback; tuning for the "
"configured local model, changed with the deployment."
),
# --- measured constants, re-derived rather than dialled ------------------
"objective.assumed_cache_rate": (
"a measured property of this deployment's traffic (0.917, token-weighted "
"over 40.7M tokens), not a preference. It is re-derived from the cache "
"telemetry when it drifts, and /metrics already warns when the measured "
"rate leaves the configured band."
),
"objective.assumed_completion_tokens": (
"a measured workload constant feeding estimated_cost, re-derived from "
"traffic rather than dialled; a guessed value silently re-prices every "
"candidate."
),
"objective.incumbent_rate_refresh_seconds": (
"sampling parameter for the measured per-model cache rates, not the "
"dial; the dial is the challenger rate."
),
"objective.incumbent_rate_min_observations": (
"sampling floor for the measured per-model cache rates: below it the "
"pricing falls back to the assumed rate. A statistical guard, not a "
"preference."
),
# --- /metrics detector thresholds: they change what the router WARNS
# --- about, never what it dispatches. An operator reads these warnings;
# --- they do not steer traffic, and a wrong value degrades a warning
# --- rather than a request.
"objective.plan_pace_warn_ratio": (
"quota-alarm sensitivity; tunes the /metrics plan-pace warning, not "
"dispatch."
),
"objective.billing_reset_day": (
"a calendar fact about the account, not a preference — it decides "
"whether quota is read as pace or as rolling 30d usage."
),
"objective.quota_burn_window_hours": (
"burn-rate estimator window; tunes the quota warning, not dispatch."
),
"objective.quota_runway_warning_hours": (
"runway alarm threshold; tunes the quota warning, not dispatch."
),
"objective.quota_burn_min_segment_samples": (
"noise floor for the burn-rate estimator; tunes the quota warning."
),
"objective.quota_burn_min_segment_hours": (
"noise floor for the burn-rate estimator; tunes the quota warning."
),
"objective.rejection_warning_window_hours": (
"alert window for the reactive rejection detector; tunes a warning. Its "
"novelty-OR-rate logic is what makes it fire, not this number."
),
"objective.rejection_warning_baseline_hours": (
"baseline window for the reactive rejection detector; tunes a warning."
),
"objective.rejection_warning_min_count": (
"count threshold for a FAMILIAR rejection group; tunes a warning. A "
"novel group warns at n>=2 regardless."
),
"objective.selection_coverage_window_hours": (
"lookback for the unroutable-models warning; tunes a warning."
),
"objective.cache_rate_window_hours": (
"lookback for the cache-rate premise warning; tunes a warning."
),
"objective.cache_rate_warn_margin": (
"how far measured cache rate may drift from assumed_cache_rate before "
"/metrics says so; tunes a warning."
),
"objective.cache_rate_warn_min_observations": (
"sample floor for the cache-rate warning; tunes a warning."
),
# --- premise-expiry check thresholds: they tune a /metrics warning
# --- class, never dispatch. Same category as the cache-rate warn knobs.
"objective.proficiency_depth_warn_min_samples": (
"minimum average sample depth threshold for the proficiency-depth "
"premise-expiry warning; tunes a /metrics warning, not dispatch."
),
"objective.proficiency_depth_warn_min_rows": (
"observation floor for the proficiency-depth premise-expiry warning; "
"tunes a /metrics warning, not dispatch."
),
"objective.cumulative_spend_warn_usd": (
"dollar threshold for the cumulative-spend premise-expiry warning; "
"tunes a /metrics warning, not dispatch."
),
"objective.cumulative_spend_warn_min_rows": (
"priced-row floor for the cumulative-spend premise-expiry warning; "
"tunes a /metrics warning, not dispatch."
),
# --- report-only series: they change what /metrics SHOWS, and not even
# --- what it warns about. metrics.cost_estimate_calibration and
# --- metrics.latency_series emit no warning class at all and are read by
# --- nothing on the routing path, so a wrong value here degrades a
# --- number on a dashboard.
"objective.cost_calibration_window_hours": (
"lookback for the estimate-vs-bill report; shapes a /metrics series "
"that nothing applies to estimated_cost or to ranking."
),
"objective.cost_calibration_min_observations": (
"sample floor deciding which per-model correction factors are marked "
"sufficient and may set the reported spread; a statistical guard on a "
"report, not a dial."
),
"objective.latency_window_hours": (
"lookback for the router-observed latency percentiles; shapes a "
"/metrics series. Latency is deliberately not a scored axis."
),
"objective.latency_min_observations": (
"sample floor for the latency percentiles, applied separately to the "
"wall and TTFT counts; a statistical guard on a report."
),
"objective.adoption_window_seconds": (
"lookback for the conversation adoption counter in /metrics; a "
"read-only window that shapes a report, not a routing dial."
),
# --- classifier: wiring with no control or reproducibility scalars -------
"classifier.timeout_seconds": (
"deployment tuning paired with model; the operator's decision is "
"which classifier fits the deployment, and the timeout follows it."
),
"classifier.temperature": (
"reproducibility, not tunable (user Q6); per-request temperature 0 "
"is what makes the classification deterministic, and deviating from "
"it in a form would produce nondeterministic routing decisions."
),
"classifier.max_output_tokens": (
"safety cap against reasoning cascade (user Q6); a classifier that "
"spends its generation budget on reasoning returns no parsible JSON, "
"and the cascade catches that silently -- widening the cap from the "
"UI would make the hidden failure more expensive, not fix it."
),
"classifier.cloud_primary.max_output_tokens": (
"not an input in the card UI (encoder-level setting); only meaningful "
"when mode is cloud_llm and cloud_primary_auto is off, and even then "
"it is a cap inherited from the card backend, not a knob the mode "
"selector exposes for independent tuning."
),
"classifier.encoder.tier_from_features": (
"design sketch only, not wired; the field and its companion "
"tier_feature_fields exist in the Pydantic model as a placeholder "
"for a future feature that is not implemented anywhere outside "
"config.py -- no admin control because it would control nothing real."
),
# --- deployment wiring: set in config, never in admin --------------------
"watchdog.dashboard_base_url": (
"deployment wiring, set in config, not admin"
),
}
# --- reflection over RouterConfig ------------------------------------------
def _strip_optional(annotation: typing.Any) -> typing.Any:
"""``Optional[X]`` -> ``X``; anything else unchanged.
Only single-argument unions are unwrapped. A genuine multi-type union is
left alone so it lands in the ``unclassified`` bucket rather than being
silently taken for one of its arms.
"""
origin = typing.get_origin(annotation)
if origin is typing.Union or origin is types.UnionType:
rest = [a for a in typing.get_args(annotation) if a is not type(None)]
if len(rest) == 1:
return _strip_optional(rest[0])
return annotation
def _walk(model: type[BaseModel], prefix: str = "") -> tuple[list[str], list[str]]:
"""Dotted paths of every scalar leaf under *model*, plus the unclassified.
Scalar means bool / int / float / str, an ``Enum`` (``FlexPreference``), or
a ``Literal`` of those (``classifier.mode``) — including Optional forms.
Nested models recurse; containers are skipped per the scope rule. Anything
that matches none of those is reported so a new annotation shape cannot
quietly drop a knob out of the walk.
"""
scalars: list[str] = []
unclassified: list[str] = []
for name, field in model.model_fields.items():
path = prefix + name
annotation = _strip_optional(field.annotation)
origin = typing.get_origin(annotation)
if origin is typing.Literal:
scalars.append(path)
elif isinstance(annotation, type) and issubclass(annotation, BaseModel):
nested_scalars, nested_unclassified = _walk(annotation, path + ".")
scalars.extend(nested_scalars)
unclassified.extend(nested_unclassified)
elif isinstance(annotation, type) and (
issubclass(annotation, enum.Enum) or annotation in (bool, int, float, str)
):
scalars.append(path)
elif origin in (dict, list, set, frozenset, tuple):
continue
else:
unclassified.append(f"{path} ({annotation!r})")
return scalars, unclassified
def _covered_paths() -> set[str]:
"""Dotted paths an operator can reach from the admin portal, derived.
All four runtime registries count, not just the boolean one. ``_FLOAT_KNOBS``
and ``_INT_KNOBS`` hold ``(path, ...)`` tuples rather than bare paths, and
reading only ``_BOOL_KNOBS`` would report a runtime-only numeric knob as
having no control at all — a false failure that invites exactly the wrong
fix, an excuse entry for a knob that is in fact covered.
Card-backed paths (``_CARD_BACKED_PATHS``) are also covered: each maps a
config key to an HTML element id on a dedicated admin card.
"""
return (
set(admin._CONFIG_ALLOWLIST)
| {".".join(path) for path in admin._BOOL_KNOBS.values()}
| {".".join(entry[0]) for entry in admin._FLOAT_KNOBS.values()}
| {".".join(entry[0]) for entry in admin._INT_KNOBS.values()}
| set(admin._CARD_BACKED_PATHS)
)
def _in_scope(paths: list[str]) -> list[str]:
"""Clause 1 (covered section) and clause 2 (not deployment wiring)."""
sections = {path.split(".")[0] for path in _covered_paths()}
return [
path
for path in paths
if path.split(".")[0] in sections
and path.rsplit(".", 1)[-1] not in DEPLOYMENT_LEAF_NAMES
]
def _scalars() -> list[str]:
scalars, unclassified = _walk(RouterConfig)
assert not unclassified, (
"RouterConfig grew a field whose annotation this walk cannot classify, "
"so it is silently exempt from the coverage check. Teach _walk about "
"it (or skip it explicitly):\n " + "\n ".join(unclassified)
)
return scalars
# --- the tests --------------------------------------------------------------
def test_every_in_scope_knob_is_reachable_or_deliberately_not():
"""The gate: a new knob needs a control, or a written reason for having none.
Fails NAMING the knob, like the precedent tests, because whoever adds it is
the person who should not have to read this file to find out what broke.
"""
covered = _covered_paths()
undecided = [
path
for path in _in_scope(_scalars())
if path not in covered and path not in DELIBERATELY_NOT_IN_ADMIN
]
assert not undecided, (
"config knob(s) reached config.yaml with no admin control and no "
"recorded decision to have none:\n "
+ "\n ".join(undecided)
+ "\n\nEither give it a control — admin._CONFIG_ALLOWLIST to persist it "
"to the overlay, admin._BOOL_KNOBS for a runtime toggle, or both as the "
"mechanism warrants — or add it to DELIBERATELY_NOT_IN_ADMIN in "
"tests/test_admin_knob_coverage.py with the reason. A control the "
"operator cannot reach without hand-editing a tracked file and "
"restarting is a knob that will not get turned."
)
def test_card_backed_gate_is_real():
"""Prove the gate actually uses _CARD_BACKED_PATHS by removing one entry.
Temporarily remove a card-backed path that is NOT in the excuse list,
run the coverage check, verify it fails naming that leaf, then restore it.
"""
import admin as admin_mod
original = dict(admin_mod._CARD_BACKED_PATHS)
# Pick a card-backed path not already excused (classifier.mode is excused
# but still card-backed — pick one we deleted from the excuse list).
test_key = None
for key in original:
if key not in DELIBERATELY_NOT_IN_ADMIN:
test_key = key
break
if test_key is None:
# Fallback: any card-backed path will do
test_key = next(iter(original))
try:
del admin_mod._CARD_BACKED_PATHS[test_key]
scalars_set = set(_scalars())
in_scope = _in_scope(sorted(scalars_set))
covered = _covered_paths()
undecided = [
p for p in in_scope
if p not in covered and p not in DELIBERATELY_NOT_IN_ADMIN
]
assert test_key in undecided, (
f"Removing {test_key!r} from _CARD_BACKED_PATHS should cause "
f"the gate to flag it, but it did not"
)
finally:
admin_mod._CARD_BACKED_PATHS = original
def test_excused_knobs_are_not_already_covered():
"""Excused knobs should NOT be covered, otherwise the excuse is false.
A card-backed leaf in DELIBERATELY_NOT_IN_ADMIN is a false excuse:
the card IS its control. The excuse should be removed, not kept.
"""
covered = _covered_paths()
for path in DELIBERATELY_NOT_IN_ADMIN:
assert path not in covered, (
f"{path!r} is in DELIBERATELY_NOT_IN_ADMIN but is already "
f"covered by the admin portal. Remove the excuse."
)
def test_excused_knobs_still_exist_and_are_in_scope():
"""No excuse for a knob that was renamed, removed, or scoped out."""
in_scope = set(_in_scope(_scalars()))
orphans = sorted(set(DELIBERATELY_NOT_IN_ADMIN) - in_scope)
assert not orphans, (
"DELIBERATELY_NOT_IN_ADMIN names knob(s) that are no longer in scope — "
"renamed, deleted, or moved into a section the portal does not "
"reach:\n " + "\n ".join(orphans) + "\n\nDelete or re-path those entries."
)
def test_every_excuse_carries_a_reason():
"""A bare name is indistinguishable from an oversight, which is the point."""
empty = sorted(k for k, v in DELIBERATELY_NOT_IN_ADMIN.items() if not v.strip())
assert not empty, (
"DELIBERATELY_NOT_IN_ADMIN entries with no reason: "
+ ", ".join(empty)
+ ". The reason IS the decision; without it the entry records only that "
"someone wanted the test to pass."
)
def test_admin_registries_point_at_real_config_knobs():
"""The covered side is not a lie either: every registered path resolves.
A typo in _CONFIG_ALLOWLIST or _BOOL_KNOBS would otherwise mark a knob
covered while the control writes a path RouterConfig has never heard of.
"""
scalars = set(_scalars())
phantom = sorted(_covered_paths() - scalars)
assert not phantom, (
"admin registry path(s) that are not scalar fields of RouterConfig:\n "
+ "\n ".join(phantom)
)
# --- Phase A Step 3: taxonomy coverage ----------------------------------------
# The test below checks two-way coverage between _KNOB_CATEGORY and the covered
# registries (_CONFIG_ALLOWLIST + runtime + card-backed). Card-backed knobs
# (with their own dedicated admin card/endpoint pair) are excluded from both
# directions — they are covered by their card but don't need a taxonomy entry.
# _CARD_BACKED_PATHS is used throughout this module: in _covered_paths(),
# test_knob_taxonomy_coverage(), and test_card_backed_gate_is_real().
def _cfg_paths_from_runtime() -> set[str]:
"""Dotted config paths from every runtime registry entry.
``_RUNTIME_KNOB_PATHS`` is the single source of truth (it is derived
from the same registries that drive the POST handlers) but we rebuild
the path set directly from the registries to avoid importing a
dict whose values are the same as ``_CONFIG_ALLOWLIST``'s and would
obscure the separate provenance this check needs.
Unlike ``_covered_paths()`` this function exists solely for the taxonomy
two-way check — it does not deduplicate against the allowlist, because
both directions start from a different origin.
"""
return (
{".".join(path) for path in admin._BOOL_KNOBS.values()}
| {".".join(entry[0]) for entry in admin._FLOAT_KNOBS.values()}
| {".".join(entry[0]) for entry in admin._INT_KNOBS.values()}
)
def _card_backed_set() -> set[str]:
"""Card-backed config paths to exclude from two-way coverage check.
Reads directly from ``admin._CARD_BACKED_PATHS``, which maps classifier
config keys to the HTML element ids of their dedicated admin card controls.
"""
return set(admin._CARD_BACKED_PATHS)
def test_knob_taxonomy_coverage():
"""Two-way coverage between _KNOB_CATEGORY and the covered registries.
Direction 1 (allowlist + runtime + card-backed -> taxonomy): every knob
that appears in ``_CONFIG_ALLOWLIST``, any runtime registry, or
``_CARD_BACKED_PATHS`` must have an entry in ``_KNOB_CATEGORY``.
Direction 2 (taxonomy -> registries + card-backed): every key in
``_KNOB_CATEGORY`` must resolve to a path covered by ``_CONFIG_ALLOWLIST``,
a runtime registry, or ``_CARD_BACKED_PATHS``.
Both directions exclude card-backed paths from the taxonomy check — they
have dedicated admin cards and don't belong in the generic knob table.
"""
# --- prepare covered set --------------------------------------------------
allowlist_paths: set[str] = set(admin._CONFIG_ALLOWLIST)
runtime_paths: set[str] = _cfg_paths_from_runtime()
card_backed: set[str] = _card_backed_set()
covered = allowlist_paths | runtime_paths | card_backed
# --- card-backed exclusion ------------------------------------------------
card_backed = _card_backed_set()
# --- Direction 1 ----------------------------------------------------------
# Every covered knob (minus card-backed) needs a category assignment.
candidates_d1 = covered - card_backed
missing = sorted(candidates_d1 - set(admin._KNOB_CATEGORY))
assert not missing, (
"Covered knob(s) with no entry in _KNOB_CATEGORY:\n "
+ "\n ".join(missing)
+ "\n\nAdd an entry to _KNOB_CATEGORY in src/admin.py with the "
"appropriate category id and advanced flag."
)
# --- Direction 2 ----------------------------------------------------------
# Every taxonomy entry (minus card-backed) must be a real covered path.
candidates_d2 = set(admin._KNOB_CATEGORY) - card_backed
orphaned = sorted(candidates_d2 - covered)
assert not orphaned, (
"_KNOB_CATEGORY key(s) that are in neither _CONFIG_ALLOWLIST nor any "
"runtime registry:\n "
+ "\n ".join(orphaned)
+ "\n\nEither add the path to _CONFIG_ALLOWLIST or a runtime registry, "
"or --- if the knob has its own dedicated admin card --- add it to "
"_CARD_BACKED_PATHS in src/admin.py."
)
# --- well-formed entries --------------------------------------------------
valid_categories = {
"routing_quality_cost",
"classification",
"local_hardware",
"caching_context",
"safety_nets",
"watchdog",
}
for path, entry in admin._KNOB_CATEGORY.items():
assert isinstance(entry, dict), (
f"_KNOB_CATEGORY[{path!r}] is not a dict: {type(entry).__name__}"
)
assert "category" in entry, (
f"_KNOB_CATEGORY[{path!r}] has no 'category' key"
)
assert entry["category"] in valid_categories, (
f"_KNOB_CATEGORY[{path!r}].category={entry['category']!r} "
f"is not in {sorted(valid_categories)}"
)
assert "advanced" in entry, (
f"_KNOB_CATEGORY[{path!r}] has no 'advanced' key"
)
assert isinstance(entry["advanced"], bool), (
f"_KNOB_CATEGORY[{path!r}].advanced is not bool: "
f"{type(entry['advanced']).__name__}"
)
# --- Component 6: one knob table -----------------------------------------------
# The two tests below step outside the pure-reflection contract documented
# above: the first drives the real app (still offline -- a temp SQLite file,
# no network, no port), the second reads admin/frontend/controls.html as
# text. Both exist because the merged knob table pairs runtime rows with
# persisted rows by a config key the API labels and the frontend source
# selects on, and neither half of that contract is visible to reflection.
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
CONTROLS_HTML = ROOT / "admin" / "frontend" / "controls.html"
def test_runtime_api_labels_every_knob_with_a_real_config_key(tmp_path, monkeypatch):
"""GET /admin/api/runtime labels every knob with a resolvable config path.
The controls page pairs each runtime row with its persisted twin by
config_key alone, so a missing or mistyped label renders one-sided rows
that cannot be fixed from the UI. Walking getattr down a loaded
RouterConfig checks each label against the model itself, independently
of the registry that produced it.
"""
conn = sqlite3.connect(str(tmp_path / "test.db"))
conn.row_factory = sqlite3.Row
conn.executescript(SCHEMA_SQL)
conn.close()
monkeypatch.setattr(dispatcher.cfg.database, "path", str(tmp_path / "test.db"))
monkeypatch.setenv("NEURALWATT_API_KEY", "test-key")
with TestClient(dispatcher.app) as client:
resp = client.get("/admin/api/runtime")
assert resp.status_code == 200
body = resp.json()
assert body
cfg = load_config(str(ROOT / "config" / "config.yaml"))
for knob, item in body.items():
assert set(item) == {"persisted", "runtime", "config_key", "category", "advanced"}, knob
# An AttributeError here names the broken label directly.
obj: typing.Any = cfg
for part in item["config_key"].split("."):
obj = getattr(obj, part)
keys = [item["config_key"] for item in body.values()]
assert len(set(keys)) == len(keys), (
"runtime knobs sharing one config_key would fuse into one table row: "
+ ", ".join(sorted(k for k in keys if keys.count(k) > 1))
)
assert set(body) == set(admin._RUNTIME_KNOB_PATHS)
def test_controls_page_keeps_the_table_selectors_its_js_depends_on():
"""controls.html still serves the strings the dependent JS sites select.
markDirty and the save collector re-query
``#config-list .setting-row[data-key]``, the default-profile hint looks
its row up by exact data-key, init() attaches delegated listeners by the
list's id, and the profile select is found via data-profile-select. A
rename in any of these silently kills dirty tracking, saving and the
hint. The tbody fallback must stay table-shaped -- a bare <div> inside
a <tbody> is invalid markup the browser will hoist out of the table.
"""
html = CONTROLS_HTML.read_text(encoding="utf-8")
assert 'id="config-list"' in html
for selector in (
'#config-list .setting-row[data-key="routing.default_profile"]',
"#config-list .setting-row[data-key]",
"getElementById('config-list')",
):
assert selector in html, selector
# The persisted row template still marks rows for the collectors.
assert 'class="setting-row"' in html
assert " data-key=" in html
assert " data-orig=" in html
assert "data-config-input" in html
assert "data-profile-select" in html
assert '<tr><td colspan="4"' in html
def test_card_backed_paths_are_real_config_scalars_and_element_ids_exist():
"""Every card-backed path is a real scalar field and its element id exists.
(a) Walk ``RouterConfig`` via ``_walk`` to verify every key in
``admin._CARD_BACKED_PATHS`` resolves to a scalar leaf of the config model.
(b) Read ``controls.html`` and verify every element id value appears as an
``id="..."`` attribute somewhere in the source.
"""
scalars = set(_scalars())
html = CONTROLS_HTML.read_text(encoding="utf-8")
for path, element_id in admin._CARD_BACKED_PATHS.items():
assert path in scalars, (
f"_CARD_BACKED_PATHS key {path!r} is not a scalar field of "
f"RouterConfig"
)
assert element_id in html, (
f"_CARD_BACKED_PATHS value {element_id!r} (for {path!r}) not found "
f"in controls.html as an element id"
)