Files
6krrt/tests/test_admin_allowlist_repoll.py
adlee-was-taken 54f31501b4 feat(admin): re-poll the catalog after an allowlist edit
Allowlisting a model does nothing until a poll ingests it, and the poller
timer runs every 2 hours. Observed live today: two models were
allowlisted at 18:59, the poller had last run at 17:46, and their catalog
rows still read `deprecated` with a `last_updated` three days old. That
reads as "the allowlist did not work", not "the catalog has not caught up
yet" -- and there is nothing in the portal that says which.

Removal matters as much as addition, and is the less obvious half: the
poller is what marks a dropped row `deprecated`, so without this a
de-allowlisted model keeps serving traffic until the next tick.

**Debounced, not fire-per-edit.** The allowlist editor adds entries one
at a time, so a five-model session would otherwise mean five full
multi-provider catalog fetches. Each edit cancels the pending timer and
starts a new one, so a burst costs exactly one poll --
`test_a_burst_of_edits_costs_exactly_one_poll` pins that.

Three things it deliberately does not do:

- It does not run inside the endpoint's transaction. `poller.main` opens
  its OWN connection, so scheduling before the commit would race it
  against an uncommitted write -- and the poll would then filter the
  catalog against an allowlist missing the row that triggered it,
  deprecating the very model just added.
- It does not fail the edit. The edit is already committed; surfacing a
  network error from the re-poll would report the wrong thing as broken,
  and the timer retries on its own schedule regardless.
- It does not live at module level. State is in `build_router`'s closure,
  keeping the module's no-globals contract.

`freshness.repoll_after_allowlist_change_seconds` (5.0) tunes it; 0
disables the trigger and leaves the timer as the only path.

tests/conftest.py disables the trigger suite-wide -- `poller.main` reads
its own config, opens its own DB and talks to the network, so the
existing allowlist tests would have started firing real polls the moment
this shipped. The new tests clear that guard and substitute the runner,
so the debounce is covered without the side effects.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
2026-09-09 19:16:08 -04:00

187 lines
6.1 KiB
Python

"""An allowlist edit should make the catalog catch up by itself.
Allowlisting a model does nothing until a poll ingests it, and the poller timer
runs every 2 hours. Observed live on 2026-09-09: two models were allowlisted at
18:59, the poller had last run at 17:46, and their catalog rows still read
``deprecated`` with a ``last_updated`` three days old. That reads as "the
allowlist did not work" rather than "the catalog has not caught up yet".
Debounce rather than fire-per-edit is the load-bearing choice: the allowlist
editor adds entries one at a time, so a five-model session would otherwise mean
five full multi-provider catalog fetches.
``poller.main`` is substituted in every test here. It reads its own config,
opens its own DB and talks to the network -- ``tests/conftest.py`` disables the
trigger suite-wide for exactly that reason, and these tests clear that guard
deliberately.
"""
from __future__ import annotations
import shutil
import sqlite3
import time
from pathlib import Path
import pytest
from fastapi import FastAPI
from starlette.testclient import TestClient
import admin
from admin import build_router
from config import load_config
ROOT = Path(__file__).resolve().parent.parent
SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text()
def _wait_for(predicate, timeout: float = 3.0) -> None:
"""Poll instead of sleeping a fixed span, so a slow machine does not flake."""
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
if predicate():
return
time.sleep(0.01)
raise AssertionError(f"condition not met within {timeout}s")
@pytest.fixture
def repoll(tmp_path, monkeypatch):
"""(client, calls, set_delay) with the trigger live and the runner faked."""
monkeypatch.delenv("ROUTER_DISABLE_BACKGROUND_POLL", raising=False)
(tmp_path / "config").mkdir(parents=True, exist_ok=True)
config_yaml = tmp_path / "config" / "config.yaml"
shutil.copyfile(ROOT / "config" / "config.yaml", config_yaml)
conn = sqlite3.connect(str(tmp_path / "test.db"))
conn.row_factory = sqlite3.Row
conn.executescript(SCHEMA_SQL)
conn.close()
cfg = load_config(str(config_yaml))
def _db_factory() -> sqlite3.Connection:
c = sqlite3.connect(str(tmp_path / "test.db"))
c.row_factory = sqlite3.Row
return c
app = FastAPI()
app.include_router(
build_router(cfg, _db_factory, base_dir=str(tmp_path)), prefix="/admin"
)
calls: list[int] = []
monkeypatch.setattr(admin.poller, "main", lambda: calls.append(1))
def set_delay(seconds: float) -> None:
# The scheduler reads this at call time, so mutating the cfg the
# router closed over is enough -- no need to rebuild it.
cfg.freshness.repoll_after_allowlist_change_seconds = seconds
set_delay(0.02)
with TestClient(app) as client:
yield client, calls, set_delay
def _add(client, model_id: str):
return client.post(
"/admin/api/providers/openrouter/allowlist", json={"model_id": model_id}
)
def test_adding_to_the_allowlist_triggers_a_poll(repoll):
client, calls, _ = repoll
assert _add(client, "vendor/new-model").status_code == 200
_wait_for(lambda: calls == [1])
def test_removing_from_the_allowlist_triggers_a_poll(repoll):
"""Removal matters as much as addition: the poller is what marks a dropped
row ``deprecated``, so without this the model keeps serving traffic after
being de-allowlisted."""
client, calls, _ = repoll
_add(client, "vendor/new-model")
_wait_for(lambda: calls == [1])
calls.clear()
resp = client.delete("/admin/api/providers/openrouter/allowlist/vendor/new-model")
assert resp.status_code == 200
_wait_for(lambda: calls == [1])
def test_a_burst_of_edits_costs_exactly_one_poll(repoll):
"""The whole point of the debounce. Five models added one at a time is the
normal way the allowlist editor is used."""
client, calls, set_delay = repoll
set_delay(0.4)
for i in range(5):
assert _add(client, f"vendor/model-{i}").status_code == 200
_wait_for(lambda: len(calls) >= 1)
time.sleep(0.4)
assert calls == [1], "five edits inside the window must coalesce to one poll"
def test_a_failed_poll_does_not_fail_the_edit(repoll, monkeypatch):
"""The edit is already committed before the poll is even scheduled.
Surfacing a network error from the re-poll would report the wrong thing as
broken, and the scheduled timer retries on its own anyway."""
client, _calls, _ = repoll
def boom():
raise RuntimeError("provider unreachable")
monkeypatch.setattr(admin.poller, "main", boom)
resp = _add(client, "vendor/new-model")
assert resp.status_code == 200
assert resp.json()["model_id"] == "vendor/new-model"
# Let the timer fire so the exception is raised inside the test, not after.
time.sleep(0.15)
def test_zero_disables_the_trigger_entirely(repoll):
"""0 leaves the scheduled timer as the only path, for an operator who does
not want an admin edit reaching the network at all."""
client, calls, set_delay = repoll
set_delay(0.0)
assert _add(client, "vendor/new-model").status_code == 200
time.sleep(0.15)
assert calls == []
def test_the_poll_sees_the_committed_row(repoll, tmp_path, monkeypatch):
"""Ordering, and it is not cosmetic: ``poller.main`` opens its OWN
connection. Scheduling from inside the endpoint's still-open transaction
would race the poll against an uncommitted write, and the poll would then
filter the catalog against an allowlist missing the row that triggered it
-- deprecating the very model just added."""
client, _calls, _ = repoll
seen: list[list[str]] = []
def read_allowlist():
conn = sqlite3.connect(str(tmp_path / "test.db"))
try:
seen.append([
r[0] for r in conn.execute(
"SELECT model_id FROM provider_model_allowlist"
)
])
finally:
conn.close()
monkeypatch.setattr(admin.poller, "main", read_allowlist)
_add(client, "vendor/new-model")
_wait_for(lambda: len(seen) == 1)
assert seen == [["vendor/new-model"]]