From 162cc6e1131ed161a4f01350cfd44fa3dd0c62eb Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 29 Aug 2026 13:40:33 -0400 Subject: [PATCH 01/18] feat(admin): wire admin router into the service dispatch path Mount the admin portal at /admin via build_router(cfg, _db, root), extend the startup migration to ensure admin tables (renaming _ensure_route_decisions_table -> _ensure_tables), and feed model deprecation overrides into the routing hard filters through _admin_deprecated_models(). Pin ruamel.yaml==0.18.10 (comment-preserving writes) for the /admin/api/config persisted-config editor. Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus --- dispatcher.py | 51 +++++++++++++++++++++++++++++++++++++++++------- requirements.txt | 4 ++++ 2 files changed, 48 insertions(+), 7 deletions(-) diff --git a/dispatcher.py b/dispatcher.py index a198f05..f82ef0f 100644 --- a/dispatcher.py +++ b/dispatcher.py @@ -47,6 +47,7 @@ from typing import Any, Literal, Optional import requests import secrets +from pathlib import Path from dotenv import load_dotenv from fastapi import BackgroundTasks, FastAPI, HTTPException from fastapi.responses import JSONResponse, Response, StreamingResponse @@ -66,6 +67,7 @@ from context_prune import ( import events import session_cache import circuit_breaker +import admin from routing import ( BATCH, INTERACTIVE, @@ -347,21 +349,20 @@ def ensure_route_decisions(conn: sqlite3.Connection) -> None: conn.commit() -def _ensure_route_decisions_table() -> None: - """Create route_decisions on the live DB if it predates this feature.""" +def _ensure_tables() -> None: + """Create pending tables on the live DB if a live router.db predates them.""" try: conn = sqlite3.connect(cfg.database.path) try: ensure_route_decisions(conn) + admin.ensure_admin_tables(conn) finally: conn.close() except Exception as exc: # noqa: BLE001 - logs.warning( - "startup_route_decisions_migration_failed", error=str(exc) - ) + logs.warning("startup_tables_migration_failed", error=str(exc)) -_ensure_route_decisions_table() +_ensure_tables() def _classifier_client() -> OpenAI: @@ -759,6 +760,7 @@ def route(req: TaskRequest) -> RouteResponse: conn = _db() try: rows = load_candidates(conn, classification.task_category) + admin_deprecated = _admin_deprecated_models(conn) finally: conn.close() @@ -769,7 +771,7 @@ def route(req: TaskRequest) -> RouteResponse: allowed_access_levels=cfg.routing.allowed_access_levels, exclude_stale=cfg.freshness.exclude_stale, exclude_deprecated=cfg.freshness.exclude_deprecated, - exclude_models=_open_circuits(rows, cfg), + exclude_models=_open_circuits(rows, cfg) | admin_deprecated, min_tool_proficiency=( cfg.routing.min_tool_proficiency if req.tools_present else None ), @@ -1852,6 +1854,29 @@ def _open_circuits(rows: list[dict], cfg) -> set[str]: } +def _admin_deprecated_models(conn: sqlite3.Connection) -> set[str]: + """Model ids the operator has deprecated through admin overrides. + + Reads ``admin_model_overrides`` once per route; deprecation is the only + availability state that drops a model from live ``/route`` candidates. + "stale" and "active" overrides are surfaced in the admin UI but do not + force exclusion here (stale already has its own filter, active never does). + Safe to call repeatedly; the query is small and the table is keyed by + (model_id, provider). If the table doesn't exist yet (fresh DB before the + startup migration), returns an empty set. + """ + try: + return { + row["model_id"] + for row in conn.execute( + "SELECT model_id FROM admin_model_overrides WHERE availability = 'deprecated'" + ) + } + except sqlite3.OperationalError: + # Table does not exist yet; migration will create it on next startup. + return set() + + def _open_upstream( model_id: str, url: str, @@ -2931,3 +2956,15 @@ def dispatch_endpoint(req: TaskRequest): completion_tokens=completion_tokens, telemetry=telemetry, ) + + +# --- admin API ---------------------------------------------------------------- +# The admin reads (health + snapshot) live in admin.py, which — like metrics.py +# — MUST NOT import dispatcher. Mount it here so the admin routes share the +# service's config and DB factory without admin.py ever importing us. +from admin import build_router + +app.include_router( + build_router(cfg, _db, str(Path(__file__).resolve().parent)), + prefix="/admin", +) diff --git a/requirements.txt b/requirements.txt index 3468c4a..839092c 100644 --- a/requirements.txt +++ b/requirements.txt @@ -2,6 +2,10 @@ # 2.53 -> 3.0 and httpx -> httpx2; a service that restarts on boot should not # change its dependency tree underneath itself. Bump deliberately. pyyaml==6.0.3 +# ruamel.yaml: comment-preserving round-trip writes for the /admin/api/config +# persisted-config editor. pyyaml cannot re-emit comments; this one can, so +# editing a tunable value doesn't strip the operator's config.yaml annotations. +ruamel.yaml==0.18.10 pydantic==2.13.4 requests==2.34.2 fastapi==0.141.1 -- 2.49.1 From 71939fdea2f8612bc295d27c260f057b1a780c16 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 29 Aug 2026 13:40:36 -0400 Subject: [PATCH 02/18] docs(admin): document admin portal and refresh test count Add the admin web portal section to README (dashboards, triggers, runtime toggles, config edits, availability overrides) and document admin.py/admin_schema.sql/admin/frontend in CLAUDE.md. Update the test count to the verified 731 tests across 40 files. Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus --- CLAUDE.md | 10 +++++++++- README.md | 46 ++++++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 53 insertions(+), 3 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index fc3b62d..10b9506 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -270,7 +270,15 @@ rather than from months of history. - `router_cli.py` — one-shot `/route` probe. Posts a task to the running router and prints the decision tree, or emits raw JSON with `--json`. Spends no quota because it only routes. -- `tests/` — 562 tests across 27 files, all passing, all offline. Verified on +- `admin.py` / `admin_schema.sql` / `admin/frontend/index.html` — integrated `/admin` + management portal served by the running router, loopback-only, no auth. Read-only + dashboards for quota burn, per-model usage, live routing decisions, verdict mix, + and scoring coverage; operational triggers (refresh catalog, seed energy, apply + feedback, restart service); runtime toggles that reset on restart; persisted + config edits limited to an allowlist; model availability overrides that feed + routing hard filters; and a bucketed history endpoint over + `energy_observations` and `route_decisions`. +- `tests/` — 731 tests across 40 files, all passing, all offline. Verified on Python 3.10 and 3.14; nothing declares `requires-python`, so 3.10 is the tested floor rather than a promised one. diff --git a/README.md b/README.md index 8578ac9..543694a 100644 --- a/README.md +++ b/README.md @@ -287,6 +287,47 @@ python tui.py `textual` is pinned in `requirements.txt` solely for the TUI modules (`tui.py`, `tui_screens.py`, `tui_sse.py`). It is imported only by these modules; the FastAPI service dispatch path never touches it, so the router itself has no UI dependency. +### Admin web portal + +`GET /admin/` serves a small management UI from the running router, on the same +loopback-only bind as the API. It has no auth layer yet, so like the other +endpoints it is reachable only from `127.0.0.1`. The portal is implemented in +`admin.py`, with `admin_schema.sql` for its tables and `admin/frontend/index.html` +for the browser UI. + +**Read-only dashboards** — `GET /admin/api/snapshot` exposes data for: + +- quota burn against `objective.plan_kwh_per_period` +- per-model usage from `energy_observations` +- live routing decisions from `route_decisions` +- verdict mix and scoring coverage +- history: `GET /admin/api/history?range=6h` (also `1h`, `24h`, `7d`, `30d`) + +**Operational triggers** — async, fire-and-forget maintenance jobs: + +- `POST /admin/api/refresh-catalog` runs the poller and tier pass +- `POST /admin/api/seed-energy?samples=N` starts a reference sweep +- `POST /admin/api/apply-feedback?dry_run=true` runs `feedback.py` +- `POST /admin/api/restart-service` restarts the running systemd unit + +**Runtime toggles** + +`GET /admin/api/runtime` shows persisted-vs-runtime values. `POST /admin/api/runtime/{knob}` +flips in-memory settings such as `log_route_decisions` and `local_llm_enabled`. +Changes take effect immediately but reset on restart. + +**Persisted config edits** + +`GET /admin/api/config` lists allowlisted keys. `POST /admin/api/config/{key}` +writes one allowlisted key back to `config.yaml` with a timestamped backup +and whole-config validation. Arbitrary keys are rejected. + +**Model availability overrides** + +`POST /admin/api/models/{model_id}/{provider}/availability` marks a model as +`active`, `deprecated`, or `stale`. `DELETE` on the same path removes the override. +Deprecation feeds into the routing hard filters. + ### Probe routing without spending `router_cli.py` is a one-shot shell probe that POSTs to `/route` once and prints the full decision tree. @@ -461,7 +502,7 @@ Key behaviors: | **Config** | `config.yaml` loaded & validated by Pydantic (`config.py`) | | **OpenAI Client** | `openai==3.0.0` (official SDK) | | **HTTP** | `requests` for poller, `httpx` (via openai/uvicorn) | -| **Testing** | `pytest` — 562 tests across 27 files, all offline | +| **Testing** | `pytest` — 731 tests across 40 files, all offline | | **Config Files** | `config.yaml`, `leaderboards.yaml`, `evals/tasks.yaml` | | **Deployment** | systemd user units (`.service` + `.timer` files in `deploy/`) | | **Integration** | `opencode.json` in the repo routes through it by default; any OpenAI-compatible client works | @@ -883,6 +924,7 @@ allowance. | `POST` | `/dispatch` | Same as `/route`, plus complete the provider call, stream response, log observation | | `GET` | `/v1/models` | OpenAI-compatible model list (router virtual models + catalog) | | `POST` | `/v1/chat/completions` | OpenAI-compatible completions — routes then proxies, **streaming supported** | +| `GET` | `/admin` | Loopback-only web management portal (read-only dashboards, operational triggers, runtime toggles, allowlisted config edits) | `/metrics` returns a single JSON object with these top-level keys: @@ -1068,7 +1110,7 @@ carries the same modality block. ## Testing ```bash -python -m pytest # 562 tests +python -m pytest # 731 tests python -m pytest --cov # with coverage ``` -- 2.49.1 From 642773ab63c48019c22ccbd07e1def20d42c8b20 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 29 Aug 2026 13:40:41 -0400 Subject: [PATCH 03/18] chore(admin): normalize config.yaml formatting from admin config round-trip Cosmetic only: 0.10 -> 0.1, null -> blank scalar, and block-list re-indentation produced by the ruamel.yaml round-trip write path (no semantic value changes). Harmless, and shows what an allowlisted /admin/api/config write produces. Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus --- config.yaml | 31 ++++++++++++++++--------------- 1 file changed, 16 insertions(+), 15 deletions(-) diff --git a/config.yaml b/config.yaml index c536550..27e8e16 100644 --- a/config.yaml +++ b/config.yaml @@ -18,7 +18,7 @@ objective: # currently rest on 2-3 samples per category, so a 0.05 gap is # indistinguishable from sampling variation and paying for it buys noise. # Narrow it as samples accumulate. - quality_tolerance: 0.10 + quality_tolerance: 0.1 # Cost is priced per-request from catalog prices, NOT from a benchmark # sweep. A fixed 400-token reference task ranked glm-5.2-fast 3.2x cheaper @@ -52,7 +52,7 @@ objective: # wall you hit mid-task. So this is the cost mandate stated as a guarantee. # For scale: the reference task runs ~5e-06 kWh on the cheapest model and # ~2.2e-04 on the most expensive. - max_energy_per_request: null + max_energy_per_request: # The subscription's kWh allowance per billing period, for reporting burn in # /health. Set to match your plan; null disables the report. NeuralWatt also @@ -115,15 +115,15 @@ proficiency: leaderboard_weight: 0.3 self_eval_weight: 0.7 categories: - - coding_general - - coding_refactor - - debugging - - docs_writing - - summarization - - translation - - reasoning_math - - tool_use_agentic - - general_chat + - coding_general + - coding_refactor + - debugging + - docs_writing + - summarization + - translation + - reasoning_math + - tool_use_agentic + - general_chat escalation: enabled: true @@ -224,7 +224,7 @@ routing: # routing excludes anything not listed here. Add 'preview'/'canary' only if # the account actually holds the grant — otherwise dispatch earns a 403. allowed_access_levels: - - public + - public # '-flex' rows are held server-side during peak until a capacity gap opens. # That's correct for overnight/batch agent work and wrong for anything @@ -274,7 +274,7 @@ routing: # off, let POST /outcome report real pass/fail, and compare # tool_use_agentic proficiency for deepseek before and after. That is the # one signal here that knows whether the work actually worked. - min_tool_proficiency: null + min_tool_proficiency: tool_use_category: tool_use_agentic # Request-side capability gates. These read the request body (image parts, @@ -299,7 +299,7 @@ local_vision: # and the model must be pulled (`ollama pull qwen3-vl:4b`) on that host. enabled: true base_url: "http://localhost:11434/v1" - api_key_env: null + api_key_env: # A Modelfile-tagged variant of qwen3-vl:4b, not the base library tag. # Measured live: the base tag comes up at Ollama's own default num_ctx # (32768) and costs 9.4GB loaded — resident alongside the classifier's @@ -398,7 +398,7 @@ classifier: base_url: "http://localhost:11434/v1" # Unset means unauthenticated, which is the Ollama case. Name the env var # holding the key when the endpoint actually checks one. - api_key_env: null + api_key_env: # A Modelfile-tagged variant of mistral-nemo:12b, not the base library tag # — must match a model `ollama list` reports. Ollama loads a model at its # library Modelfile's default context unless told otherwise, and the base @@ -517,3 +517,4 @@ logging: # systemctl --user edit llm-router # Environment="LLM_ROUTER_LOG_LEVEL=debug" # systemctl --user restart llm-router level: info + -- 2.49.1 From 08eb4369cbee7c576759881fc2d6dd9755203ace Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 29 Aug 2026 13:40:45 -0400 Subject: [PATCH 04/18] chore: ignore admin config backups and playwright node artifacts Ignore config.yaml.bak.* (timestamped config-write backups), node_modules/, package.json, package-lock.json (Playwright browser-QA noise), and .playwright-mcp/. Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus --- .gitignore | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/.gitignore b/.gitignore index d23c91e..c1f4ec1 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,8 @@ router.log .venv/ venv/ .omo/ +config.yaml.bak.* +node_modules/ +package.json +package-lock.json +.playwright-mcp/ -- 2.49.1 From 906c9e6316bcdd4e5abb363daf66262e582f7405 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 29 Aug 2026 14:49:58 -0400 Subject: [PATCH 05/18] fix(admin): replace stretched verdict doughnut with partition bar and split history into per-metric mini charts --- admin/frontend/index.html | 125 +++++++++++++++++++++++++++----------- 1 file changed, 88 insertions(+), 37 deletions(-) diff --git a/admin/frontend/index.html b/admin/frontend/index.html index fd91224..15ede8c 100644 --- a/admin/frontend/index.html +++ b/admin/frontend/index.html @@ -90,9 +90,15 @@ a:hover{text-decoration:underline} .bar-val{color:var(--text-dim);flex:0 0 100px;text-align:right} /* ── Chart ── */ -.chart-container{position:relative;width:100%;aspect-ratio:1/1;max-height:260px} +.chart-container{position:relative;width:100%;max-height:260px} .chart-container canvas{width:100%!important;height:100%!important} +/* ── History mini-charts grid ── */ +.history-grid{display:grid;grid-template-columns:repeat(auto-fit,minmax(240px,1fr));gap:10px;margin-top:8px} +.history-mini{display:flex;flex-direction:column;gap:2px} +.history-mini .chart-title{font-size:0.72rem;font-weight:600;color:var(--text);margin:0 2px} +.history-mini .chart-wrap{position:relative;width:100%;height:150px} + /* ── Warnings ── */ .warnings{display:flex;flex-direction:column;gap:6px} .warning-item{padding:6px 8px;background:var(--yellow-dim);border-left:3px solid var(--yellow);border-radius:0 var(--radius) var(--radius) 0;font-size:0.78rem;line-height:1.4} @@ -217,7 +223,7 @@ a:hover{text-decoration:underline}

◉ Verdict Mix

-
+
@@ -247,7 +253,13 @@ a:hover{text-decoration:underline} -
+
+
Decisions
+
Requests
+
Cost (USD)
+
Energy (kWh)
+
Carbon (g)
+
@@ -416,7 +428,7 @@ function addDecision(dec) { ═══════════════════════════════════════ */ let verdictChart = null; -let historyChart = null; +const historyCharts = {}; // metric-key -> Chart instance for the mini history charts function formatTs(ts) { if (!ts) return '—'; @@ -517,9 +529,9 @@ function renderVerdict(mix) { const colors = []; const labelsLC = labels.map(l => l.toLowerCase()); const verdictColors = { - pass: 'var(--green)', approve: 'var(--green)', verified: 'var(--green)', - fail: 'var(--red)', reject: 'var(--red)', rejected: 'var(--red)', - ambiguous: 'var(--yellow)', unknown: 'var(--text-dim)' + pass: '#2ecc71', approve: '#2ecc71', verified: '#2ecc71', + fail: '#e74c3c', reject: '#e74c3c', rejected: '#e74c3c', + ambiguous: '#f1c40f', unknown: '#5a6a7f' }; for (const l of labelsLC) { let found = false; @@ -528,8 +540,8 @@ function renderVerdict(mix) { } if (!found) colors.push(`hsl(${Math.abs(hashCode(l)) % 360}, 60%, 60%)`); } - if (verdictChart) { updateChart('verdict', labels, values, 'doughnut', colors); } - else { verdictChart = renderChart('verdict-chart', labels, values, 'doughnut', colors); } + if (verdictChart) { updateChart('verdict', labels, values, 'verdict-bar', colors); } + else { verdictChart = renderChart('verdict-chart', labels, values, 'verdict-bar', colors); } el.innerHTML = labels.map((l, i) => `${escapeHtml(String(l))}: ${values[i]}` ).join(''); @@ -565,12 +577,6 @@ function renderCategoryBreakdown(decisions) { } function renderHistory(data) { - const seriesNames = Object.keys(data); - if (!seriesNames.length || seriesNames.every(s => !data[s] || !data[s].length)) { - if (historyChart) { historyChart.destroy(); historyChart = null; return; } - return; - } - const seriesColors = { decisions_per_bucket: { border: '#3498db', bg: 'rgba(52,152,219,0.1)' }, requests_per_bucket: { border: '#9b59b6', bg: 'rgba(155,89,182,0.1)' }, @@ -579,19 +585,28 @@ function renderHistory(data) { carbon_per_bucket: { border: '#e74c3c', bg: 'rgba(231,76,60,0.1)' }, }; - const seriesLabels = { - decisions_per_bucket: 'Decisions', - requests_per_bucket: 'Requests', - cost_per_bucket: 'Cost (USD)', - energy_per_bucket: 'Energy (kWh)', - carbon_per_bucket: 'Carbon (g)', - }; + // Metric key -> (canvas id, label). Kept in a fixed order so the five + // mini-charts always line up with the five coloured titles above them. + const seriesMeta = [ + { key: 'decisions_per_bucket', id: 'history-decisions-chart', label: 'Decisions' }, + { key: 'requests_per_bucket', id: 'history-requests-chart', label: 'Requests' }, + { key: 'cost_per_bucket', id: 'history-cost-chart', label: 'Cost (USD)' }, + { key: 'energy_per_bucket', id: 'history-energy-chart', label: 'Energy (kWh)' }, + { key: 'carbon_per_bucket', id: 'history-carbon-chart', label: 'Carbon (g)' }, + ]; - const datasets = seriesNames.map(s => { - const pts = data[s] || []; - const c = seriesColors[s] || { border: 'var(--text-dim)', bg: 'rgba(255,255,255,0.05)' }; - return { - label: seriesLabels[s] || s, + for (const meta of seriesMeta) { + const pts = data[meta.key] || []; + const c = seriesColors[meta.key] || { border: 'var(--text-dim)', bg: 'rgba(255,255,255,0.05)' }; + + // Destroy the chart if this metric has faded out of the returned window. + if (!pts.length) { + if (historyCharts[meta.key]) { historyCharts[meta.key].destroy(); delete historyCharts[meta.key]; } + continue; + } + + const dataset = { + label: meta.label, data: pts.map(([ts, v]) => ({ x: ts * 1000, y: v })), borderColor: c.border, backgroundColor: c.bg, @@ -601,13 +616,16 @@ function renderHistory(data) { tension: 0.2, yAxisID: 'y', }; - }); + const datasets = [dataset]; - if (historyChart) { - historyChart.data.datasets = datasets; - historyChart.update('none'); - } else { - historyChart = renderChart('history-chart', [], datasets, 'line'); + if (historyCharts[meta.key]) { + historyCharts[meta.key].data.datasets = datasets; + // Re-clear any stale unit tick labels after a range change. + historyCharts[meta.key].options.scales.y.title.text = meta.label; + historyCharts[meta.key].update('none'); + } else { + historyCharts[meta.key] = renderChart(meta.id, meta.label, datasets, 'line'); + } } } @@ -817,10 +835,28 @@ function renderChart(canvasId, labels, valuesOrDatasets, type, colors) { let opts = { responsive: true, - maintainAspectRatio: true, + maintainAspectRatio: false, plugins: { legend: { display: false } }, }; + // Single-bar horizontal "partition table" — one stacked bar split into + // segments representing the composition of the mix. + if (type === 'verdict-bar') { + const datasets = [{ + data: Array.isArray(valuesOrDatasets) ? valuesOrDatasets : [valuesOrDatasets], + backgroundColor: colors || ['#3498db','#2ecc71','#e74c3c','#f1c40f','#9b59b6','#1abc9c'], + borderWidth: 1, + borderColor: 'var(--surface)', + barThickness: 18 + }]; + opts.indexAxis = 'y'; + opts.scales = { + x: { stacked: true, display: false, grid: { display: false }, border: { display: false }, ticks: { display: false } }, + y: { stacked: true, display: false, grid: { display: false }, border: { display: false }, ticks: { display: false } } + }; + return new Chart(canvas, { type: 'bar', data: { labels: [''], datasets }, options: opts }); + } + if (type === 'doughnut') { const datasets = [{ data: valuesOrDatasets, @@ -835,10 +871,25 @@ function renderChart(canvasId, labels, valuesOrDatasets, type, colors) { if (type === 'line') { opts.scales = { x: { display: true, type: 'time', time: { tooltipFormat: 'HH:mm', unit: 'minute' }, ticks: { maxTicksLimit: 20 } }, - y: { display: true, beginAtZero: true, grid: { color: 'rgba(255,255,255,0.04)' } } + y: { + display: true, + beginAtZero: true, + grid: { color: 'rgba(255,255,255,0.04)' }, + title: { + display: true, + text: (typeof labels === 'string' ? labels : ''), + color: '#5a6a7f', + font: { family: 'JetBrains Mono, monospace', size: 9 } + } + } }; - opts.plugins.legend = { display: true, labels: { color: 'var(--text)', font: { family: 'JetBrains Mono, monospace', size: 10 } } }; - return new Chart(canvas, { type: 'line', data: { labels, datasets: valuesOrDatasets }, options: opts }); + // Each mini chart is a single series with a coloured title above it, so + // the in-chart legend would only duplicate the label. + opts.plugins.legend = { display: false }; + // `labels` carries the y-axis unit title (a string) when provided; the + // time axis derives its x values from the data points, so a real label + // array is unnecessary here. + return new Chart(canvas, { type: 'line', data: { labels: [], datasets: valuesOrDatasets }, options: opts }); } return new Chart(canvas, { type, data: { labels, datasets: [{ data: valuesOrDatasets, backgroundColor: colors || ['#3498db','#2ecc71','#e74c3c','#f1c40f','#9b59b6'] }] }, options: opts }); -- 2.49.1 From cc8231367b52c8a86b70055c1c8917040af3750e Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sat, 29 Aug 2026 22:13:43 -0400 Subject: [PATCH 06/18] feat(admin): rebuild frontend on a true Tabler shell with liquid-glass theme Standalone pages for Models and the new Decisions log (filter/search over route_decisions, live via SSE), full page/page-wrapper/page-header/page-body shell on Dashboard/Controls/Models/Decisions, and a glass-effect pass (translucent blurred navbar/cards/buttons/badges/progress bars) over the previously flat Tabler skin. Dashboard reshuffle: Quota Meter and Warnings cards replaced by a header capsule chip (click -> popup, also links out to the NeuralWatt dashboard) and a navbar bell with a dismissible dropdown; Category Breakdown gets real bars instead of bare dots; Per-Model Usage is a responsive grid with a Calls/Cost/Energy sort toggle; Recent Decisions moved to the bottom with a link to the full log. Fixes bundled in along the way: - objective.max_energy_per_request: Persisted Config was sending the literal string "null" instead of JSON null, so every save 422'd. Null fields now render empty with a placeholder and save as real null. - Category Breakdown's counts were unreadable: bg-secondary badge text color was too close to its own background. - Logo lost its color/face: the filter:none override lost the cascade to Tabler's [data-bs-theme=dark] .navbar-brand-autodark rule (higher specificity); needed !important. controls.html was missing the fix, the favicon link, and the 36px (now 45px) sizing entirely. - Persistent left-edge gap at wide viewports: this Chrome/Linux build reports a phantom left margin on equal to the scrollbar's width whenever itself is the scrolling element. Moved the scroll context down to . - Warnings dropdown rendered behind the quota chip: backdrop-filter on header.navbar creates a stacking context, and the later-in-DOM .page element (whose quota chip also has backdrop-filter) painted on top of it regardless of the dropdown's own z-index. Pinned the navbar's stacking context above it explicitly. - Admin pages had no Cache-Control, so browsers served stale pages after every edit during iteration; now explicitly no-cache. AGENTS.md: manual/dev instances now bind :8081 instead of a bare 8080, with an explicit warning not to pkill/kill anything matching uvicorn/dispatcher/8080 -- that's the systemd-managed production instance (opencode's own model traffic runs through it), and killing it by pattern match was today's actual cause of the router's "flaky restarts" incident, documented in CLAUDE.md. config.yaml: circuit_breaker, session_cache, pinch, and pinch.relevance flipped to enabled -- deliberate, to start collecting real-traffic signal on knobs that shipped off by design. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U --- AGENTS.md | 19 +- CLAUDE.md | 51 +- admin.py | 37 +- admin/frontend/controls.html | 474 +++++++++++++ admin/frontend/decisions.html | 398 +++++++++++ admin/frontend/index.html | 1244 +++++++++++++++++---------------- admin/frontend/models.html | 318 +++++++++ config.yaml | 8 +- tests/test_admin_frontend.py | 19 + 9 files changed, 1935 insertions(+), 633 deletions(-) create mode 100644 admin/frontend/controls.html create mode 100644 admin/frontend/decisions.html create mode 100644 admin/frontend/models.html diff --git a/AGENTS.md b/AGENTS.md index 568b657..7523f99 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -119,10 +119,21 @@ pip install -r requirements.txt sqlite3 router.db < schema.sql cp .env.example .env # fill in NEURALWATT_API_KEY -# Start the service (binds 127.0.0.1:8080) -python -m uvicorn dispatcher:app --reload -# Or via systemd: -systemctl --user start llm-router.service +# Port 8080 is the systemd-managed PRODUCTION instance. opencode's own +# model traffic goes through it (see opencode.json's baseURL) and +# `Restart=always` resurrects it ~5s after any kill — so never +# `pkill`/`kill` anything matching uvicorn/dispatcher/8080 to "free the +# port". That fights the supervisor, and on this repo it can cut off your +# own inference mid-task. See CLAUDE.md's "A bare SIGTERM could hang the +# process forever" section for the incident this note comes from. +# +# To pick up a dispatcher.py change on the real instance: +systemctl --user restart llm-router.service + +# For an ad hoc/manual run — iterating with --reload, a throwaway instance +# for Playwright smoke tests against the admin frontend, anything that +# isn't "use the real router" — bind a different port so it can't collide: +python -m uvicorn dispatcher:app --reload --port 8081 # Run the TUI (service must be running) python tui.py diff --git a/CLAUDE.md b/CLAUDE.md index 10b9506..f283b69 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1200,12 +1200,51 @@ are still honored correctly regardless of that policy (systemd tracks deliberate stops separately from the Restart= decision), so this only adds self-healing for the unexplained case. -The signal's actual source is still open — live investigation and audit -trail in `code_plans/router-unreachable-signal-investigation.md`. Confirmed -so far: it is not `systemctl`, not the admin portal's restart trigger, not -suspend/resume, not the OOM killer, and — via `auditd` — not delivered -through the `kill` or `tgkill` syscalls either, which is why the watch was -extended to `pidfd_send_signal` and the `rt_*sigqueueinfo` syscalls. +**Resolved 2026-08-29 — the source was opencode, killing its own supply +line.** Full audit trail in +`code_plans/router-unreachable-signal-investigation.md`; the syscall-level +extension to `pidfd_send_signal` (past `kill`/`tgkill`, both cleared by +`auditd`) is what finally caught it. `sudo ausearch -k routerkill` matched +five incidents in one afternoon to `pidfd_send_signal(..., SIGTERM)` / +`SIGKILL`-after-escalation calls from a non-interactive `zsh -c "pkill ..."` +(never in `~/.zsh_history`, since it's not a login shell), and +`~/.local/share/opencode/log/opencode.log` matched every one of those +timestamps, to the millisecond, to one opencode run testing the admin +frontend: `pkill -f "uvicorn dispatcher:app"` (or `.*dispatcher`, or plain +`"8080"`), then `python -m uvicorn dispatcher:app --host 127.0.0.1 --port +8080` for its own throwaway instance. When the port came back occupied 5s +later (`Restart=always` resurrecting the real service), the agent read that +as "the kill didn't work" and escalated to `pkill -9` — the one case +(16:01:00) that arrived as a bare `SIGKILL`, `status=9/KILL`, rather than a +caught `SIGTERM`. + +This was self-inflicted in a sharper way than it looks: `opencode.json` +points opencode's *own* model traffic at `http://127.0.0.1:8080/v1` — the +same production instance it was killing to test against. Every kill briefly +cut off the agent's own inference supply. + +The actual bug was in `AGENTS.md`, not in this service: its "How to run +things" section told an agent to bring the router up with a bare +`python -m uvicorn dispatcher:app --reload` (implicitly on 8080, no port +flag) as an equally-valid alternative to `systemctl --user start`, with no +warning that 8080 is normally already held by the supervised instance. An +agent following that instruction and finding the port taken has no way to +know the right move is `systemctl --user restart` (which the doc *does* say +two sections later, for the "changed dispatcher.py" case, but not for "I +want to smoke-test against a running instance"). Fixed there: manual/ad hoc +runs now bind `--port 8081` explicitly, and the section says outright not to +`pkill`/`kill` anything matching `uvicorn`/`dispatcher`/`8080` — that's the +systemd-managed instance, `Restart=always` will fight you, and on this repo +it may be your own model access. + +**Convention going forward: 8080 is production, always.** It's the port +baked into `opencode.json`, every curl example in this file, the systemd +unit, and the admin frontend's own fetches — moving it would touch more +surface than the problem is worth. A throwaway instance (manual iteration, +Playwright smoke tests against the admin frontend, anything that isn't "use +the real router") binds **8081** instead, and nothing should ever send a +kill signal to a process matched by name/port rather than by a PID it +started itself. ## Pointing a coding agent at it diff --git a/admin.py b/admin.py index 4c9fc2b..e763d33 100644 --- a/admin.py +++ b/admin.py @@ -442,10 +442,31 @@ def build_router( router = APIRouter() _admin_frontend = _module_dir / "admin" / "frontend" / "index.html" + _admin_controls = _module_dir / "admin" / "frontend" / "controls.html" + _admin_models = _module_dir / "admin" / "frontend" / "models.html" + _admin_decisions = _module_dir / "admin" / "frontend" / "decisions.html" + + # No-cache: these pages get hand-edited and reloaded constantly during + # frontend iteration, and FileResponse sets no Cache-Control of its own — + # browsers fall back to heuristic freshness, which was serving a stale + # page after edits until a manual hard-refresh forced revalidation. + _NO_CACHE_HEADERS = {"Cache-Control": "no-cache"} @router.get("/") def admin_index() -> FileResponse: - return FileResponse(_admin_frontend, media_type="text/html") + return FileResponse(_admin_frontend, media_type="text/html", headers=_NO_CACHE_HEADERS) + + @router.get("/controls") + def admin_controls() -> FileResponse: + return FileResponse(_admin_controls, media_type="text/html", headers=_NO_CACHE_HEADERS) + + @router.get("/models") + def admin_models_page() -> FileResponse: + return FileResponse(_admin_models, media_type="text/html", headers=_NO_CACHE_HEADERS) + + @router.get("/decisions") + def admin_decisions_page() -> FileResponse: + return FileResponse(_admin_decisions, media_type="text/html", headers=_NO_CACHE_HEADERS) @router.get("/api/health") def admin_health() -> dict: @@ -586,6 +607,20 @@ def build_router( finally: conn.close() + @router.get("/api/decisions") + def admin_decisions_list(limit: int = 500) -> list: + """Larger decision batch for the standalone /decisions page. + + Filtering/search happens client-side over this batch — the operator + scale here (hundreds, not millions, of rows) doesn't justify server- + side query params yet. + """ + conn = _db_callable() + try: + return metrics.recent_decisions(conn, limit=min(limit, 2000)) + finally: + conn.close() + @router.get("/api/history") def admin_history(range: str = "24h") -> dict: """Bucketed time-series over energy/decision tables for the admin UI. diff --git a/admin/frontend/controls.html b/admin/frontend/controls.html new file mode 100644 index 0000000..196909d --- /dev/null +++ b/admin/frontend/controls.html @@ -0,0 +1,474 @@ + + + + + +Controls · LLM Router Admin + + + + + + + + + + + + + + +
+
+ +
+
+
+ + +
+
+

Operational Triggers

+
+

Fire-and-forget maintenance jobs.

+
+ + + + +
+
+
+
+
+
+
+

Runtime Knobs

+
+

Toggle in-memory only; no config.yaml write.

+
+
+
+
+ + +
+
+

Persisted Config

+
+

Allowlisted keys only; persisted to config.yaml.

+
+ + +
Loading config…
+
+
+ + +
+
+
+
+ +
+
+
+
+
+ © 2026 adLee · 6krrt, local LLM model router · admin +
+
+
+
+ + + + + + + diff --git a/admin/frontend/decisions.html b/admin/frontend/decisions.html new file mode 100644 index 0000000..354746e --- /dev/null +++ b/admin/frontend/decisions.html @@ -0,0 +1,398 @@ + + + + + +Decisions · LLM Router Admin + + + + + + + + + + + + + + +
+
+ +
+
+
+
+
+
+
+ + + + + +
+ +
+
+ + + + + + + + + + +
TimeKindCategoryTierSourceModelCostProfRejected
Loading…
+
+
+
+
+
+
+
+
+ © 2026 adLee · 6krrt, local LLM model router · admin +
+
+
+
+ + + + diff --git a/admin/frontend/index.html b/admin/frontend/index.html index 15ede8c..3a6983d 100644 --- a/admin/frontend/index.html +++ b/admin/frontend/index.html @@ -1,304 +1,400 @@ - + -Admin Dashboard — LLM Router +Dashboard · LLM Router Admin + + + + + + -
-
-

admin@router ▸ dashboard

- - connecting… + + + + + +
+
+ +
+
+
+ + +
+
+
+

Per-Model Usage

+
+ + + +
+
+
+
+
Loading…
+
+
+
+
+ + +
+
+

Verdict Mix

+
+
+
+
+
+

Category Breakdown

+
+
+
+ + +
+
+
+

History

+
+ + + + + +
+
+
+
+
+
Decisions
+
+
+
+
Requests
+
+
+
+
Cost (USD)
+
+
+
+
Energy (kWh)
+
+
+
+
Carbon (g)
+
+
+
+
+
+
+ + +
+
+
+

Recent Decisions

+ View all → +
+
+
+ + + +
timekindcategorytiermodelcostprof
+
+
+
+
+ +
+
+
+
+
+ © 2026 adLee · 6krrt, local LLM model router · admin +
+
-
—
-
- -
-
-

⚡ Quota Meter

-
Loading…
-
-
-

◈ Model Availability

-
- - -
modelprovidertierstatusoverride
Loading…
+ +