refactor(session-cache): rename staleness_minutes to staleness_seconds #94
@@ -99,7 +99,7 @@ Then edit `config/config.yaml` for your own setup — at minimum:
|
||||
| `classifier.base_url` | where that Ollama actually is |
|
||||
| `objective.plan_kwh_per_period` | your plan's quota; `/health` reports burn against it |
|
||||
| `objective.assumed_cache_rate` | 0.917 was measured from one client's traffic (40.7M tokens). Check yours against the provider's per-session cache-hit figures |
|
||||
| `session_cache.enabled` | off by default; caches category/tier per session for `staleness_minutes` to skip repeat classifier round-trips on long agent sessions |
|
||||
| `session_cache.enabled` | off by default; caches category/tier per session for `staleness_seconds` to skip repeat classifier round-trips on long agent sessions |
|
||||
|
||||
`PYTHONPATH=src python -m seed_energy` is optional — it populates energy data logged but not used by routing. It costs real money and quota, so it is not in the install path.
|
||||
|
||||
|
||||
@@ -639,7 +639,7 @@ const SESSION_CACHE_GATE_NOTE = {
|
||||
};
|
||||
const SESSION_CACHE_TTL_NOTE = {
|
||||
text: 'how long one label steers routing',
|
||||
title: 'Minutes a session reuses its last classification, 1 to 120. No blank and no zero: to stop reusing labels turn session_cache off, because 0 still caches and the classifier-failure cascade still replays the entry. Measured on live traffic: one classification drove 107 consecutive turns.',
|
||||
title: 'Seconds a session reuses its last classification, 5 to 7200. No blank and no zero: to stop reusing labels turn session_cache off, because 0 still caches and the classifier-failure cascade still replays the entry. Measured on live traffic: one classification drove 107 consecutive turns.',
|
||||
cls: 'bg-info',
|
||||
glyph: 'info',
|
||||
};
|
||||
@@ -648,8 +648,8 @@ const KNOB_NOTES = {
|
||||
'pinch.prefix_probe': PROBE_NOTE,
|
||||
session_cache_enabled: SESSION_CACHE_GATE_NOTE,
|
||||
'session_cache.enabled': SESSION_CACHE_GATE_NOTE,
|
||||
session_cache_staleness_minutes: SESSION_CACHE_TTL_NOTE,
|
||||
'session_cache.staleness_minutes': SESSION_CACHE_TTL_NOTE,
|
||||
session_cache_staleness_seconds: SESSION_CACHE_TTL_NOTE,
|
||||
'session_cache.staleness_seconds': SESSION_CACHE_TTL_NOTE,
|
||||
incumbent_cache_pricing: INCUMBENT_GATE_NOTE,
|
||||
'objective.incumbent_cache_pricing': INCUMBENT_GATE_NOTE,
|
||||
incumbent_challenger_cache_rate: CHALLENGER_DIAL_NOTE,
|
||||
@@ -671,9 +671,9 @@ const NUMBER_BOUNDS = {
|
||||
// No `placeholder`, deliberately: an empty field means nothing here, and a
|
||||
// placeholder is how the dial above says that it does. Blank posts null and
|
||||
// the server refuses it, naming session_cache_enabled as the switch the
|
||||
// operator was reaching for. min is 1 rather than 0 -- 0 is not "off", it
|
||||
// operator was reaching for. min is 5 rather than 0 -- 0 is not "off", it
|
||||
// still caches and the classifier-failure cascade still replays the entry.
|
||||
session_cache_staleness_minutes: { min: 1, max: 120, step: 1 },
|
||||
session_cache_staleness_seconds: { min: 5, max: 7200, step: 1 },
|
||||
};
|
||||
|
||||
function settingRow({ key, meta = '', control, dirtyAttrs = '' }) {
|
||||
|
||||
@@ -455,7 +455,7 @@ session_cache:
|
||||
# project: ship it, watch route_decisions.source="cached" on real traffic,
|
||||
# then decide the right default.
|
||||
enabled: true
|
||||
staleness_minutes: 20
|
||||
staleness_seconds: 1200
|
||||
|
||||
circuit_breaker:
|
||||
# Passive availability circuit breaker, on by default. When enabled, a model
|
||||
|
||||
@@ -314,10 +314,10 @@ per type, not one shared table:
|
||||
| table | knob | range | blank |
|
||||
|---|---|---|---|
|
||||
| `_FLOAT_KNOBS` | `incumbent_challenger_cache_rate` | 0 - 1 | neutral |
|
||||
| `_INT_KNOBS` | `session_cache_staleness_minutes` | 1 - 120 | refused |
|
||||
| `_INT_KNOBS` | `session_cache_staleness_seconds` | 5 - 7200 | refused |
|
||||
|
||||
They are separate because the declared type has to survive the write.
|
||||
`_FLOAT_KNOBS` coerces with `float(value)`, and `session_cache.staleness_minutes`
|
||||
`_FLOAT_KNOBS` coerces with `float(value)`, and `session_cache.staleness_seconds`
|
||||
is declared `int` — a float there would leave the running `cfg` holding a value
|
||||
`load_config` could never produce, and a fractional one would 422 the persisted
|
||||
twin at load. The float tuple also carries a fourth element, `neutral_path`,
|
||||
@@ -328,9 +328,9 @@ A runtime write bypasses every Pydantic validator — `StrictModel` sets
|
||||
`extra="forbid"`, not `validate_assignment` — so the endpoint does the type and
|
||||
range checking itself: a numeric knob refuses anything that is not a number
|
||||
(booleans included, since `True` is an `int` in Python and would land as `1.0`
|
||||
or a 1-minute window) and anything outside its declared bounds. `_INT_KNOBS`
|
||||
or a 1-second window) and anything outside its declared bounds. `_INT_KNOBS`
|
||||
also refuses floats rather than truncating them, and imports its bounds from
|
||||
`config.STALENESS_MINUTES_MIN`/`MAX` so the runtime path and the config
|
||||
`config.STALENESS_SECONDS_MIN`/`MAX` so the runtime path and the config
|
||||
validator cannot drift into disagreeing about what the service will boot with.
|
||||
|
||||
**Persisted config edits**
|
||||
@@ -388,20 +388,20 @@ in both panels:
|
||||
| key | runtime knob | persisted key |
|
||||
|---|---|---|
|
||||
| switch | `session_cache_enabled` | `session_cache.enabled` |
|
||||
| window | `session_cache_staleness_minutes` | `session_cache.staleness_minutes` |
|
||||
| window | `session_cache_staleness_seconds` | `session_cache.staleness_seconds` |
|
||||
|
||||
The window decides how long **one** classification keeps steering routing, so
|
||||
it is the knob that sets the size of the concession CLAUDE.md's north star rule
|
||||
2 describes, not an implementation detail of the switch. Measured on live
|
||||
traffic: 96.6% of all classifications are cache replays, and one classification
|
||||
drove 107 consecutive turns across 14 minutes. Before this control the only
|
||||
available settings were a 20-minute window or no cache at all, and the answer
|
||||
drove 107 consecutive turns across 840s. Before this control the only
|
||||
available settings were a 1200-second window or no cache at all, and the answer
|
||||
is almost certainly in between — which is why the runtime half matters.
|
||||
`dispatcher.py` reads `cfg.session_cache.staleness_minutes` per request, so a
|
||||
`dispatcher.py` reads `cfg.session_cache.staleness_seconds` per request, so a
|
||||
narrower window is live on the next one and the replay share can be re-measured
|
||||
without a restart.
|
||||
|
||||
**Units are minutes and the UI does no scaling** — what is typed is what is
|
||||
**Units are seconds and the UI does no scaling** — what is typed is what is
|
||||
stored, the `classifier.confidence_threshold` lesson applied rather than
|
||||
relearned. The field is an `int`, and a float is refused rather than truncated:
|
||||
`20.5` quietly becoming `20` is a window nobody chose.
|
||||
@@ -411,23 +411,24 @@ off, and the difference is easy to miss:
|
||||
|
||||
- `session_cache.put` still writes on every turn, so the entry exists.
|
||||
- The classifier-failure cascade reads it with `session_cache.stale_read`,
|
||||
which **ignores staleness entirely** — so a 0-minute window still replays a
|
||||
which **ignores staleness entirely** — so a 0-second window still replays a
|
||||
session's label whenever the classifier is down.
|
||||
|
||||
So 0 is out of range (floor 1, the pre-existing `> 0` rule) and an empty field
|
||||
So 0 is out of range (floor 5, the pre-existing `> 0` rule) and an empty field
|
||||
posts `null`, which the runtime endpoint refuses with a message naming
|
||||
`session_cache_enabled` as the switch the operator was reaching for. The
|
||||
persisted path refuses both through whole-config validation of the merged
|
||||
base + overlay, before any byte reaches disk.
|
||||
|
||||
The ceiling, 120, is a judgement: an unbounded window is a cache that never
|
||||
expires. It is 6x the shipped default and ~8.5x the longest single-classification
|
||||
The ceiling, 7200, is a judgement: an unbounded window is a cache that never
|
||||
expires. It is 7200/1200 = 6x (same ratio) the shipped default and 7200/840 ≈
|
||||
8.57x the longest single-classification
|
||||
run measured, so it sits above every value there is a reason to try while still
|
||||
guaranteeing a label cannot outlive the working session that produced it. The
|
||||
bound lives in `config.py` (`STALENESS_MINUTES_MIN`/`MAX`) so a file edit, an
|
||||
bound lives in `config.py` (`STALENESS_SECONDS_MIN`/`MAX`) so a file edit, an
|
||||
overlay write and a runtime POST all enforce the same range.
|
||||
|
||||
The shipped default is unchanged: `enabled: true`, `staleness_minutes: 20`.
|
||||
The shipped default is unchanged: `enabled: true`, `staleness_seconds: 1200`.
|
||||
|
||||
**Classifier mode**
|
||||
|
||||
|
||||
@@ -93,8 +93,8 @@ Two more settings are graceful degradation rather than prevention:
|
||||
local model does not silently promote every request to the frontier tier.
|
||||
- **Session classification cache** (`session_cache.enabled` in
|
||||
`config/config.yaml`, off by default) remembers the last
|
||||
`task_category`/`task_tier` decision per session for `staleness_minutes`
|
||||
(default 20), so a long agent session skips the classifier round-trip on
|
||||
`task_category`/`task_tier` decision per session for `staleness_seconds`
|
||||
(default 1200), so a long agent session skips the classifier round-trip on
|
||||
every turn. It only ever short-circuits the classifier call — capability
|
||||
flags (tools/images/json) are still read fresh from each request body,
|
||||
fallback classifications are never cached, and there is no persistence
|
||||
|
||||
26
src/admin.py
26
src/admin.py
@@ -50,8 +50,8 @@ from config import (
|
||||
FlexPreference,
|
||||
RouterConfig,
|
||||
RoutingProfile,
|
||||
STALENESS_MINUTES_MAX,
|
||||
STALENESS_MINUTES_MIN,
|
||||
STALENESS_SECONDS_MAX,
|
||||
STALENESS_SECONDS_MIN,
|
||||
load_config,
|
||||
)
|
||||
|
||||
@@ -398,7 +398,7 @@ _FLOAT_KNOBS: dict[str, tuple[tuple[str, ...], float, float, tuple[str, ...]]] =
|
||||
# _FLOAT_KNOBS rather than a widening of it, for two reasons that both come
|
||||
# down to the declared type having to survive the write:
|
||||
#
|
||||
# * _FLOAT_KNOBS coerces with ``float(value)``. ``session_cache.staleness_minutes``
|
||||
# * _FLOAT_KNOBS coerces with ``float(value)``. ``session_cache.staleness_seconds``
|
||||
# is declared ``int``, so a float write would leave the running cfg holding
|
||||
# a value load_config could never produce, and a fractional one would 422
|
||||
# the persisted twin at load — the runtime and persisted halves of one
|
||||
@@ -422,10 +422,10 @@ _FLOAT_KNOBS: dict[str, tuple[tuple[str, ...], float, float, tuple[str, ...]]] =
|
||||
#
|
||||
# knob -> (cfg path, low, high)
|
||||
_INT_KNOBS: dict[str, tuple[tuple[str, ...], int, int]] = {
|
||||
"session_cache_staleness_minutes": (
|
||||
("session_cache", "staleness_minutes"),
|
||||
STALENESS_MINUTES_MIN,
|
||||
STALENESS_MINUTES_MAX,
|
||||
"session_cache_staleness_seconds": (
|
||||
("session_cache", "staleness_seconds"),
|
||||
STALENESS_SECONDS_MIN,
|
||||
STALENESS_SECONDS_MAX,
|
||||
),
|
||||
}
|
||||
|
||||
@@ -496,7 +496,7 @@ _CONFIG_ALLOWLIST: dict[str, tuple[str, ...]] = {
|
||||
),
|
||||
"circuit_breaker.enabled": ("circuit_breaker", "enabled"),
|
||||
"session_cache.enabled": ("session_cache", "enabled"),
|
||||
"session_cache.staleness_minutes": ("session_cache", "staleness_minutes"),
|
||||
"session_cache.staleness_seconds": ("session_cache", "staleness_seconds"),
|
||||
"local_compute.enabled": ("local_compute", "enabled"),
|
||||
"verification.local_llm_enabled": ("verification", "local_llm_enabled"),
|
||||
"pinch.enabled": ("pinch", "enabled"),
|
||||
@@ -516,7 +516,7 @@ _CONFIG_GET_ORDER: list[str] = [
|
||||
"objective.incumbent_challenger_cache_rate",
|
||||
"circuit_breaker.enabled",
|
||||
"session_cache.enabled",
|
||||
"session_cache.staleness_minutes",
|
||||
"session_cache.staleness_seconds",
|
||||
"local_compute.enabled",
|
||||
"verification.local_llm_enabled",
|
||||
"pinch.enabled",
|
||||
@@ -948,8 +948,8 @@ def _runtime_state(cfg: Any) -> dict:
|
||||
# Adjacent and in this order for the same reason the incumbent pair is:
|
||||
# the switch reads above the window it gates, so the UI note can say so
|
||||
# in four words instead of a paragraph.
|
||||
"session_cache_staleness_minutes": _get_at(
|
||||
cfg, _INT_KNOBS["session_cache_staleness_minutes"][0]
|
||||
"session_cache_staleness_seconds": _get_at(
|
||||
cfg, _INT_KNOBS["session_cache_staleness_seconds"][0]
|
||||
),
|
||||
"pinch_enabled": _get_at(cfg, _BOOL_KNOBS["pinch_enabled"]),
|
||||
"pinch_prefix_probe": _get_at(cfg, _BOOL_KNOBS["pinch_prefix_probe"]),
|
||||
@@ -2433,7 +2433,7 @@ def build_router(
|
||||
),
|
||||
)
|
||||
# ``True`` is an ``int`` in Python, so without this a checkbox body
|
||||
# would land as a real 1-minute window. Floats are refused rather
|
||||
# would land as a real 1-second window. Floats are refused rather
|
||||
# than truncated: the field is declared ``int``, and 20.5 silently
|
||||
# becoming 20 is a value the operator never chose. 20.0 is refused
|
||||
# with it, which is the cost of one unambiguous rule.
|
||||
@@ -2441,7 +2441,7 @@ def build_router(
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail=(
|
||||
f"{knob} expects a whole number of minutes in "
|
||||
f"{knob} expects a whole number of seconds in "
|
||||
f"[{low}, {high}], got {value!r}"
|
||||
),
|
||||
)
|
||||
|
||||
@@ -783,36 +783,36 @@ class PinchConfig(StrictModel):
|
||||
return v
|
||||
|
||||
|
||||
# The legal range for ``session_cache.staleness_minutes``, kept as module
|
||||
# The legal range for ``session_cache.staleness_seconds``, kept as module
|
||||
# constants because three surfaces have to agree on it: this validator (file
|
||||
# and overlay edits), ``admin._INT_KNOBS`` (the runtime write, which bypasses
|
||||
# every Pydantic validator), and the number field's min/max in
|
||||
# ``admin/frontend/controls.html``. Two of the three import these; the third is
|
||||
# pinned by a test.
|
||||
#
|
||||
# The floor is 1, not 0, and that is the pre-existing ``> 0`` rule rather than
|
||||
# The floor is 5, not 0, and that is the pre-existing ``> 0`` rule rather than
|
||||
# a new opinion. It matters more than it looks: 0 would make ``session_cache.get``
|
||||
# miss on every turn, which LOOKS like "stop reusing labels" but is not —
|
||||
# ``session_cache.put`` still writes, and the classifier-failure cascade's
|
||||
# ``stale_read`` ignores staleness entirely, so a 0-minute window still replays
|
||||
# ``stale_read`` ignores staleness entirely, so a 0-second window still replays
|
||||
# a session's label whenever the classifier is down. The knob that actually
|
||||
# stops reuse is ``session_cache.enabled``, and it has its own control.
|
||||
#
|
||||
# The ceiling is a judgement, and the judgement is that an unbounded window is
|
||||
# a cache that never expires. 120 is 6x the shipped default and ~8.5x the
|
||||
# a cache that never expires. 7200 is 6x the shipped default and ~8.5x the
|
||||
# longest single-classification run measured on live traffic (107 consecutive
|
||||
# turns across 14 minutes, 2026-09-16), so it is far above any value there is
|
||||
# turns across 840 seconds), so it is far above any value there is
|
||||
# a reason to try, while still guaranteeing a label cannot outlive the working
|
||||
# session that produced it. See CLAUDE.md, north star rule 2.
|
||||
STALENESS_MINUTES_MIN = 1
|
||||
STALENESS_MINUTES_MAX = 120
|
||||
STALENESS_SECONDS_MIN = 5
|
||||
STALENESS_SECONDS_MAX = 7200
|
||||
|
||||
|
||||
class SessionCacheConfig(StrictModel):
|
||||
"""Per-session classification cache (in-memory, process lifetime).
|
||||
|
||||
Remembers the last task_category/task_tier decision for each session for
|
||||
``staleness_minutes``, so a long agent session skips the classifier
|
||||
``staleness_seconds``, so a long agent session skips the classifier
|
||||
round-trip on every turn. Capability flags (tools/images/json) are NEVER
|
||||
cached — they are read fresh from each request body. Fallback
|
||||
classifications are NEVER cached. No persistence: the cache lives only in
|
||||
@@ -820,17 +820,17 @@ class SessionCacheConfig(StrictModel):
|
||||
"""
|
||||
|
||||
enabled: bool = False
|
||||
# Minutes since a cached classification was written before it is treated
|
||||
# Seconds since a cached classification was written before it is treated
|
||||
# as expired and the next turn reclassifies from scratch.
|
||||
staleness_minutes: int = 20
|
||||
staleness_seconds: int = 1200
|
||||
|
||||
@field_validator("staleness_minutes")
|
||||
@field_validator("staleness_seconds")
|
||||
@classmethod
|
||||
def staleness_in_range(cls, v: int) -> int:
|
||||
if not (STALENESS_MINUTES_MIN <= v <= STALENESS_MINUTES_MAX):
|
||||
if not (STALENESS_SECONDS_MIN <= v <= STALENESS_SECONDS_MAX):
|
||||
raise ValueError(
|
||||
f"session_cache.staleness_minutes must be between "
|
||||
f"{STALENESS_MINUTES_MIN} and {STALENESS_MINUTES_MAX} "
|
||||
f"session_cache.staleness_seconds must be between "
|
||||
f"{STALENESS_SECONDS_MIN} and {STALENESS_SECONDS_MAX} "
|
||||
f"(to stop reusing classifications set session_cache.enabled "
|
||||
f"to false; 0 is not that, it still caches and the "
|
||||
f"classifier-failure cascade still replays the entry)"
|
||||
|
||||
@@ -4266,7 +4266,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks):
|
||||
if cfg.session_cache.enabled and session_key is not None:
|
||||
cached = session_cache.get(
|
||||
session_key,
|
||||
staleness_seconds=cfg.session_cache.staleness_minutes * 60,
|
||||
staleness_seconds=cfg.session_cache.staleness_seconds,
|
||||
)
|
||||
if cached is not None:
|
||||
# Cache hit: skip the classifier. Capability flags (tools/images/
|
||||
|
||||
@@ -95,7 +95,7 @@ def test_config_GET_returns_allowlisted_values(client):
|
||||
"objective.plan_kwh_per_period",
|
||||
"circuit_breaker.enabled",
|
||||
"session_cache.enabled",
|
||||
"session_cache.staleness_minutes",
|
||||
"session_cache.staleness_seconds",
|
||||
"verification.local_llm_enabled",
|
||||
"objective.incumbent_cache_pricing",
|
||||
"objective.incumbent_challenger_cache_rate",
|
||||
@@ -337,46 +337,46 @@ def test_config_staleness_window_round_trips_as_a_real_int(client):
|
||||
The type assertion is the point, and it is the ``classifier.confidence_threshold``
|
||||
lesson in its other form: that incident wrote a raw ``80`` meaning 80% for a
|
||||
field the loader wanted as 0.0-1.0, and nothing caught it until someone
|
||||
looked. Here the units are minutes and the field is declared ``int``, so
|
||||
the UI must not scale it and the write must not float it — a ``20.0`` on
|
||||
looked. Here the units are seconds and the field is declared ``int``, so
|
||||
the UI must not scale it and the write must not float it — a ``1200.0`` on
|
||||
disk is a value ``load_config`` refuses, and a quietly-truncated ``20.5``
|
||||
is a window the operator never chose.
|
||||
"""
|
||||
tc, config_yaml = client
|
||||
|
||||
before = tc.get("/admin/api/config").json()
|
||||
assert before["session_cache.staleness_minutes"] == {"value": 20, "source": "base"}
|
||||
assert before["session_cache.staleness_seconds"] == {"value": 1200, "source": "base"}
|
||||
|
||||
assert (
|
||||
tc.post(
|
||||
"/admin/api/config/session_cache.staleness_minutes",
|
||||
"/admin/api/config/session_cache.staleness_seconds",
|
||||
json={"value": 5},
|
||||
).status_code
|
||||
== 200
|
||||
)
|
||||
|
||||
after = tc.get("/admin/api/config").json()
|
||||
assert after["session_cache.staleness_minutes"] == {"value": 5, "source": "overlay"}
|
||||
assert after["session_cache.staleness_seconds"] == {"value": 5, "source": "overlay"}
|
||||
|
||||
# The shipped default never moves; the admin write lands in the overlay.
|
||||
base = yaml.safe_load(config_yaml.read_text())["session_cache"]
|
||||
assert base["staleness_minutes"] == 20
|
||||
assert base["staleness_seconds"] == 1200
|
||||
|
||||
loaded = load_config(str(config_yaml), include_overlay=True)
|
||||
minutes = loaded.session_cache.staleness_minutes
|
||||
assert isinstance(minutes, int) and not isinstance(minutes, bool)
|
||||
assert minutes == 5
|
||||
seconds = loaded.session_cache.staleness_seconds
|
||||
assert isinstance(seconds, int) and not isinstance(seconds, bool)
|
||||
assert seconds == 5
|
||||
# The switch beside it did not move on the overlay merge.
|
||||
assert loaded.session_cache.enabled is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bad", [0, -1, 121, 1440, None])
|
||||
@pytest.mark.parametrize("bad", [0, -1, 4, 7201, 10000, None])
|
||||
def test_config_staleness_window_refuses_out_of_range_and_blank(client, bad):
|
||||
"""Whole-config validation refuses before a byte reaches the overlay.
|
||||
|
||||
``0`` is in here on purpose. It looks like "stop reusing classifications"
|
||||
and is not: ``session_cache.put`` still writes and the classifier-failure
|
||||
cascade's ``stale_read`` ignores staleness entirely, so a 0-minute window
|
||||
cascade's ``stale_read`` ignores staleness entirely, so a 0-second window
|
||||
still replays a session's label whenever the classifier is down. The knob
|
||||
that actually stops reuse is ``session_cache.enabled``, which has its own
|
||||
control, so 0 is refused rather than quietly honoured — and ``None``
|
||||
@@ -386,21 +386,21 @@ def test_config_staleness_window_refuses_out_of_range_and_blank(client, bad):
|
||||
local_yaml = config_yaml.with_name("config.local.yaml")
|
||||
|
||||
resp = tc.post(
|
||||
"/admin/api/config/session_cache.staleness_minutes", json={"value": bad}
|
||||
"/admin/api/config/session_cache.staleness_seconds", json={"value": bad}
|
||||
)
|
||||
assert resp.status_code == 422
|
||||
assert not local_yaml.exists()
|
||||
assert (
|
||||
tc.get("/admin/api/config").json()["session_cache.staleness_minutes"]["value"]
|
||||
== 20
|
||||
tc.get("/admin/api/config").json()["session_cache.staleness_seconds"]["value"]
|
||||
== 1200
|
||||
)
|
||||
|
||||
|
||||
def test_config_staleness_window_is_allowlisted_at_the_right_path():
|
||||
"""A wrong path would write a key StrictModel forbids, 422ing every save."""
|
||||
assert _CONFIG_ALLOWLIST["session_cache.staleness_minutes"] == (
|
||||
assert _CONFIG_ALLOWLIST["session_cache.staleness_seconds"] == (
|
||||
"session_cache",
|
||||
"staleness_minutes",
|
||||
"staleness_seconds",
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -376,22 +376,22 @@ def test_controls_page_pairs_the_session_cache_switch_with_its_window(admin_clie
|
||||
for key in (
|
||||
"session_cache_enabled:",
|
||||
"'session_cache.enabled':",
|
||||
"session_cache_staleness_minutes:",
|
||||
"'session_cache.staleness_minutes':",
|
||||
"session_cache_staleness_seconds:",
|
||||
"'session_cache.staleness_seconds':",
|
||||
):
|
||||
assert key in text
|
||||
|
||||
|
||||
def test_controls_page_staleness_window_is_bounded_and_has_no_blank(admin_client):
|
||||
"""1 to 120, and no placeholder — blank means nothing on this field.
|
||||
"""5 to 7200, and no placeholder — blank means nothing on this field.
|
||||
|
||||
The dial above uses `placeholder` to say what an empty field means, so its
|
||||
absence here is the signal that emptiness is not a setting. The floor is 1
|
||||
absence here is the signal that emptiness is not a setting. The floor is 5
|
||||
rather than 0 because 0 reads as "off" and is not: the cache still writes
|
||||
and the classifier-failure cascade still replays the entry.
|
||||
"""
|
||||
text = admin_client.get("/admin/controls").text
|
||||
assert "session_cache_staleness_minutes: { min: 1, max: 120, step: 1 }" in text
|
||||
assert "session_cache_staleness_seconds: { min: 5, max: 7200, step: 1 }" in text
|
||||
assert "this.value.trim() === '' ? null : Number(this.value)" in text
|
||||
|
||||
|
||||
|
||||
@@ -152,12 +152,12 @@ DELIBERATELY_NOT_IN_ADMIN: dict[str, str] = {
|
||||
"only the default for requests that omit the field; every agent client "
|
||||
"sends it per request, so the config value decides almost nothing."
|
||||
),
|
||||
# `session_cache.staleness_minutes` was listed here as "TTL internal to a
|
||||
# `session_cache.staleness_seconds` was listed here as "TTL internal to a
|
||||
# covered master switch". That excuse did not survive north star rule 2:
|
||||
# the window is not an implementation detail of the switch, it is the size
|
||||
# of the concession -- 96.6% of classifications are replays, and one
|
||||
# classification drove 107 consecutive turns. The only choices the portal
|
||||
# offered were a 20-minute window or no cache at all, and the answer is
|
||||
# offered were a 1200-second window or no cache at all, and the answer is
|
||||
# almost certainly in between. It has both controls now, so it is gone.
|
||||
"circuit_breaker.initial_cooldown_seconds": (
|
||||
"backoff shape internal to a covered master switch "
|
||||
|
||||
@@ -19,7 +19,7 @@ from starlette.testclient import TestClient
|
||||
|
||||
import dispatcher
|
||||
from admin import _INT_KNOBS
|
||||
from config import STALENESS_MINUTES_MAX, STALENESS_MINUTES_MIN, load_config
|
||||
from config import STALENESS_SECONDS_MAX, STALENESS_SECONDS_MIN, load_config
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
|
||||
@@ -89,7 +89,7 @@ def test_runtime_GET_reports_every_knob(seeded_client):
|
||||
"pinch_relevance_enabled",
|
||||
"incumbent_cache_pricing",
|
||||
"incumbent_challenger_cache_rate",
|
||||
"session_cache_staleness_minutes",
|
||||
"session_cache_staleness_seconds",
|
||||
"default_flex_preference",
|
||||
):
|
||||
assert knob in body
|
||||
@@ -328,7 +328,7 @@ def test_challenger_dial_refuses_a_non_number(
|
||||
def staleness_cfg_guard(monkeypatch):
|
||||
"""Restore the session-cache window on cfg after a test moves it."""
|
||||
sc = dispatcher.cfg.session_cache
|
||||
monkeypatch.setattr(sc, "staleness_minutes", sc.staleness_minutes)
|
||||
monkeypatch.setattr(sc, "staleness_seconds", sc.staleness_seconds)
|
||||
return sc
|
||||
|
||||
|
||||
@@ -340,16 +340,16 @@ def test_staleness_window_sets_a_real_int(seeded_client, staleness_cfg_guard):
|
||||
cache at all" (``session_cache_enabled``). Narrowing it and re-measuring
|
||||
the replay share is a loop nobody walks through a tracked-file edit and a
|
||||
``systemctl --user restart``; ``dispatcher`` reads
|
||||
``cfg.session_cache.staleness_minutes`` per request, so a POST is live on
|
||||
``cfg.session_cache.staleness_seconds`` per request, so a POST is live on
|
||||
the next one.
|
||||
"""
|
||||
resp = seeded_client.post(
|
||||
"/admin/api/runtime/session_cache_staleness_minutes", json={"value": 5}
|
||||
"/admin/api/runtime/session_cache_staleness_seconds", json={"value": 5}
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert resp.json() == {"ok": True, "session_cache_staleness_minutes": 5}
|
||||
assert resp.json() == {"ok": True, "session_cache_staleness_seconds": 5}
|
||||
|
||||
stored = dispatcher.cfg.session_cache.staleness_minutes
|
||||
stored = dispatcher.cfg.session_cache.staleness_seconds
|
||||
# A real int, not a float and not a bool: the field is declared ``int``,
|
||||
# and a float here would be a value load_config could never produce, so
|
||||
# the runtime and persisted halves of one control would disagree about
|
||||
@@ -358,11 +358,11 @@ def test_staleness_window_sets_a_real_int(seeded_client, staleness_cfg_guard):
|
||||
assert stored == 5
|
||||
|
||||
after = seeded_client.get("/admin/api/runtime").json()
|
||||
assert after["session_cache_staleness_minutes"]["runtime"] == 5
|
||||
assert after["session_cache_staleness_seconds"]["runtime"] == 5
|
||||
# config.yaml is untouched by a runtime write.
|
||||
assert (
|
||||
after["session_cache_staleness_minutes"]["persisted"]
|
||||
== CFG.session_cache.staleness_minutes
|
||||
after["session_cache_staleness_seconds"]["persisted"]
|
||||
== CFG.session_cache.staleness_seconds
|
||||
)
|
||||
|
||||
|
||||
@@ -378,36 +378,36 @@ def test_blank_staleness_window_is_refused_not_read_as_zero(
|
||||
is down. The switch that actually stops reuse is ``session_cache_enabled``,
|
||||
so the 422 names it rather than leaving the operator to guess.
|
||||
"""
|
||||
before = dispatcher.cfg.session_cache.staleness_minutes
|
||||
before = dispatcher.cfg.session_cache.staleness_seconds
|
||||
resp = seeded_client.post(
|
||||
"/admin/api/runtime/session_cache_staleness_minutes", json={"value": None}
|
||||
"/admin/api/runtime/session_cache_staleness_seconds", json={"value": None}
|
||||
)
|
||||
assert resp.status_code == 422
|
||||
detail = resp.json()["detail"]
|
||||
assert "no blank setting" in detail
|
||||
assert "session_cache_enabled" in detail
|
||||
assert dispatcher.cfg.session_cache.staleness_minutes == before
|
||||
assert dispatcher.cfg.session_cache.staleness_seconds == before
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bad", [0, -1, 121, 1440])
|
||||
@pytest.mark.parametrize("bad", [0, -1, 4, 7201, 10000])
|
||||
def test_staleness_window_refuses_values_outside_its_bounds(
|
||||
seeded_client, staleness_cfg_guard, bad
|
||||
):
|
||||
"""0 is not "off" and there is no unbounded window.
|
||||
|
||||
Both ends are decisions. The floor is 1 because 0 reads as off and is not
|
||||
(see the blank test). The ceiling is 120 because an unbounded window is a
|
||||
Both ends are decisions. The floor is 5 because 0 reads as off and is not
|
||||
(see the blank test). The ceiling is 7200 because an unbounded window is a
|
||||
cache that never expires — 6x the shipped default and ~8.5x the longest
|
||||
single-classification run measured on live traffic (107 consecutive turns
|
||||
across 14 minutes), so it is above every value there is a reason to try.
|
||||
across 840 seconds), so it is above every value there is a reason to try.
|
||||
"""
|
||||
before = dispatcher.cfg.session_cache.staleness_minutes
|
||||
before = dispatcher.cfg.session_cache.staleness_seconds
|
||||
resp = seeded_client.post(
|
||||
"/admin/api/runtime/session_cache_staleness_minutes", json={"value": bad}
|
||||
"/admin/api/runtime/session_cache_staleness_seconds", json={"value": bad}
|
||||
)
|
||||
assert resp.status_code == 422
|
||||
assert "must be in [1, 120]" in resp.json()["detail"]
|
||||
assert dispatcher.cfg.session_cache.staleness_minutes == before
|
||||
assert "must be in [5, 7200]" in resp.json()["detail"]
|
||||
assert dispatcher.cfg.session_cache.staleness_seconds == before
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bad", ["20", True, False, 20.5, 20.0, [20]])
|
||||
@@ -416,18 +416,18 @@ def test_staleness_window_refuses_a_non_integer(
|
||||
):
|
||||
"""``True`` is the sharp one: it is an ``int`` in Python and reads as 1.
|
||||
|
||||
A checkbox body landing on this field would silently set a 1-minute
|
||||
A checkbox body landing on this field would silently set a 1-second
|
||||
window. Floats are refused rather than truncated for the same reason a
|
||||
string is: 20.5 quietly becoming 20 is a value the operator never chose,
|
||||
and 20.0 pays the same price for one unambiguous rule.
|
||||
"""
|
||||
before = dispatcher.cfg.session_cache.staleness_minutes
|
||||
before = dispatcher.cfg.session_cache.staleness_seconds
|
||||
resp = seeded_client.post(
|
||||
"/admin/api/runtime/session_cache_staleness_minutes", json={"value": bad}
|
||||
"/admin/api/runtime/session_cache_staleness_seconds", json={"value": bad}
|
||||
)
|
||||
assert resp.status_code == 422
|
||||
assert "expects a whole number of minutes" in resp.json()["detail"]
|
||||
assert dispatcher.cfg.session_cache.staleness_minutes == before
|
||||
assert "expects a whole number of seconds" in resp.json()["detail"]
|
||||
assert dispatcher.cfg.session_cache.staleness_seconds == before
|
||||
|
||||
|
||||
def test_staleness_runtime_bounds_match_the_config_validator():
|
||||
@@ -439,9 +439,9 @@ def test_staleness_runtime_bounds_match_the_config_validator():
|
||||
validator would let an operator set at runtime a value the service refuses
|
||||
to boot with.
|
||||
"""
|
||||
path, low, high = _INT_KNOBS["session_cache_staleness_minutes"]
|
||||
assert path == ("session_cache", "staleness_minutes")
|
||||
assert (low, high) == (STALENESS_MINUTES_MIN, STALENESS_MINUTES_MAX)
|
||||
path, low, high = _INT_KNOBS["session_cache_staleness_seconds"]
|
||||
assert path == ("session_cache", "staleness_seconds")
|
||||
assert (low, high) == (STALENESS_SECONDS_MIN, STALENESS_SECONDS_MAX)
|
||||
|
||||
|
||||
def test_post_invalid_flex_value_returns_422(seeded_client):
|
||||
|
||||
@@ -1435,7 +1435,7 @@ def test_cache_expiry_reclassifies(router, logbuf, session_cache_on, monkeypatch
|
||||
json={"model": "auto", "messages": _session_messages()})
|
||||
assert counts["calls"] == 1
|
||||
|
||||
# Advance beyond the 20-minute TTL and send another turn.
|
||||
# Advance beyond the 1200-second TTL and send another turn.
|
||||
now = [session_cache.time.time()]
|
||||
monkeypatch.setattr(session_cache.time, "time", lambda: now[0] + 20 * 60 + 1)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user