feat(config): gitignored config.local.yaml overlay, incident #5, backup tooling #26

Merged
alee merged 9 commits from feat/config-local-overlay into main 2026-09-05 00:28:38 +00:00
23 changed files with 1720 additions and 69 deletions

2
.gitignore vendored
View File

@@ -8,6 +8,8 @@ router.log
venv/
.omo/
config.yaml.bak.*
config/config.local.yaml
config/config.local.yaml.bak.*
node_modules/
package.json
package-lock.json

View File

@@ -217,7 +217,7 @@ rather than from months of history.
- `tui.py` — Textual dashboard over `/metrics` + `/events/decisions`; live feed, category→model panel, detail popup; data layer split into `tui_model.py`. [architecture](docs/architecture.md).
- `router_cli.py` — one-shot `/route` probe (no spend), raw JSON with `--json`. [api](docs/api.md).
- `admin.py` / `config/admin_schema.sql` / `admin/frontend/*.html` — loopback `/admin` portal: dashboard, models overrides, decisions log, controls. [admin-portal](docs/admin-portal.md).
- `tests/` — 1020 tests across 41 files, offline, verified on Python 3.10 and 3.14. [README](README.md).
- `tests/` — 1155 tests across 40+ files, offline, verified on Python 3.10 and 3.14. [README](README.md).
## Proficiency: category now changes routing
@@ -744,7 +744,7 @@ in the same pass — it was declared, never read, and shadowed the
## Setup
Full install steps (venv, deps, config, first run) in [README ## Installation](README.md#installation). Set `classifier.model`/`base_url` and `objective.plan_kwh_per_period` in `config/config.yaml` (README config table). Model tags (`num_ctx`) + `verification.model` same-tag note in [docs/local-models.md](docs/local-models.md). Requirements are pinned — bump deliberately (README).
Full install steps (venv, deps, config, first run) in [README ## Installation](README.md#installation). Host-local deployment values go in `config/config.local.yaml` (gitignored overlay) — `classifier.model`/`base_url`, `local_energy.*`, host-specific URLs. General defaults in `config/config.yaml` stay shareable. `objective.plan_kwh_per_period` in the README config table. Model tags (`num_ctx`) + `verification.model` same-tag note in [docs/local-models.md](docs/local-models.md). Requirements are pinned — bump deliberately (README).
## Run as a service

View File

@@ -71,6 +71,7 @@ python -m venv .venv && source .venv/bin/activate
pip install -r requirements.txt
sqlite3 router.db < config/schema.sql
cp .env.example .env # fill in NEURALWATT_API_KEY (.env stays at repo root)
cp config/config.local.yaml.example config/config.local.yaml # optional: deployment-specific overlay (gitignored)
PYTHONPATH=src python -m poller # populate the catalog
PYTHONPATH=src python -m tier # resolve tiers
PYTHONPATH=src python -m config # sanity-check config loads

View File

@@ -261,7 +261,7 @@ header.navbar{padding-top:2px!important;padding-bottom:2px!important}
</div>
</div>
<div class="card-body settings-body">
<p class="text-muted settings-hint">Allowlisted keys, written to <code>config.yaml</code>.</p>
<p class="text-muted settings-hint">Allowlisted keys, written to <code>config/config.local.yaml</code> (machine-local overlay; <code>config.yaml</code> stays clean).</p>
<div class="settings-list" id="config-list">
<div class="text-muted fst-italic small px-2 py-1">Loading config&hellip;</div>
</div>
@@ -516,9 +516,13 @@ function renderConfig(config) {
list.innerHTML = '<div class="text-muted fst-italic small px-2 py-1">No config data</div>';
return;
}
const html = Object.entries(config).map(([key, val]) => {
const html = Object.entries(config).map(([key, raw]) => {
const entry = raw || {};
const val = entry.value;
const source = entry.source;
const isBool = typeof val === 'boolean';
let control;
if (typeof val === 'boolean') {
if (isBool) {
// A switch, matching Runtime Knobs — the bare checkbox here was the one
// control on the page that didn't look like the others.
control = `<div class="form-check form-switch">
@@ -531,9 +535,19 @@ function renderConfig(config) {
} else {
control = `<input class="form-control form-control-sm config-input toggle-input" type="text" value="${val === null ? '' : escapeHtml(String(val))}" placeholder="${val === null ? 'null' : ''}" data-config-input>`;
}
// Provenance badge is legitimate meta: it explains where the effective
// value came from without restating the value the control already shows.
let meta = UNITS[key] ? escapeHtml(UNITS[key]) : '';
if (source === 'overlay') {
meta += `${meta ? ' ' : ''}<span class="badge bg-purple" title="set in config.local.yaml">overlay</span>`;
} else if (source === 'base') {
meta += `${meta ? ' ' : ''}<span class="badge bg-secondary" title="from config.yaml">base</span>`;
}
return settingRow({
key,
meta: UNITS[key] ? escapeHtml(UNITS[key]) : '',
meta,
control,
dirtyAttrs: ` data-key="${escapeHtml(key)}" data-orig="${escapeHtml(String(val === null ? '' : val))}"`,
});

View File

@@ -0,0 +1,16 @@
# config.local.yaml — machine-local overrides
#
# This file is gitignored. It takes precedence over config/config.yaml when
# load_config() merges the overlay, so place deployment-specific or personal
# secrets and knobs here instead of editing the tracked config.yaml:
# - local_energy.enabled + local_energy.tariff_usd_per_kwh
# - host-specific classifier.base_url / verification.base_url
# - any value you change between machines but want to keep in the repo default
#
# Only the keys you want to override need to be present — the rest come from
# config/config.yaml.
# Example: enable local energy metering on your dev machine.
# local_energy:
# enabled: true
# tariff_usd_per_kwh: 0.159

View File

@@ -0,0 +1,12 @@
# Snapshots router.db, .env and config/config.local.yaml to a directory OUTSIDE
# the repository. See docs/incidents.md #5: `git clean -fdx` destroyed all three
# in place, and a backup kept inside the repo would have gone with them.
[Unit]
Description=Snapshot the LLM router database and operator config
Documentation=file:%h/Sources/6krrt/docs/incidents.md
[Service]
Type=oneshot
WorkingDirectory=%h/Sources/6krrt
Environment=REPO=%h/Sources/6krrt
ExecStart=%h/Sources/6krrt/deploy/llm-router-backup.sh

55
deploy/llm-router-backup.sh Executable file
View File

@@ -0,0 +1,55 @@
#!/usr/bin/env bash
# Snapshot the router's irreplaceable state.
#
# WHY THIS EXISTS: on 2026-09-04 an agent ran `git clean -fdx` in the repo.
# The -x flag removes IGNORED files, and everything this deployment needs is
# ignored by design -- router.db went to 0 bytes, taking 22,776 energy
# observations, 17,321 routing decisions and 148 proficiency scores, plus .env
# and the whole virtualenv. Recovery was luck: a QA copy happened to exist in
# /tmp from 25 seconds earlier. See docs/incidents.md #5.
#
# Backups therefore live OUTSIDE the repository. A backup inside it -- even
# gitignored -- would have been destroyed by the same command.
#
# What is NOT recoverable without this:
# - energy_observations : historical measurements, cannot be regenerated
# - route_decisions : same
# - proficiency : rebuildable only by re-running evals, which costs
# real provider credits
set -euo pipefail
REPO="${REPO:-$HOME/Sources/6krrt}"
DEST="${LLM_ROUTER_BACKUP_DIR:-$HOME/.local/share/6krrt-backups}"
KEEP="${LLM_ROUTER_BACKUP_KEEP:-24}"
mkdir -p "$DEST"
stamp=$(date +%Y%m%d-%H%M%S)
# .backup is the ONLY safe way to copy a live SQLite file. `cp` on a database
# with an open writer can produce a torn copy that passes a size check and
# fails integrity_check.
sqlite3 "$REPO/router.db" ".backup '$DEST/router-$stamp.db'"
gzip -f "$DEST/router-$stamp.db"
# Operator data that also lives only in ignored files.
[ -f "$REPO/.env" ] && { cp -f "$REPO/.env" "$DEST/env-$stamp.bak"; chmod 600 "$DEST/env-$stamp.bak"; }
[ -f "$REPO/config/config.local.yaml" ] && cp -f "$REPO/config/config.local.yaml" "$DEST/config.local-$stamp.yaml"
# Verify before rotating: a backup that has never been read is a guess.
tmp=$(mktemp)
zcat "$DEST/router-$stamp.db.gz" > "$tmp"
if [ "$(sqlite3 "$tmp" 'SELECT integrity_check FROM pragma_integrity_check LIMIT 1;')" != "ok" ]; then
rm -f "$tmp" "$DEST/router-$stamp.db.gz"
echo "backup FAILED integrity_check; discarded, older backups retained" >&2
exit 1
fi
rm -f "$tmp"
# Rotate only after a good backup exists, so a failing run never leaves you
# with fewer copies than you started with.
for pat in "router-*.db.gz" "env-*.bak" "config.local-*.yaml"; do
# shellcheck disable=SC2012
ls -1t "$DEST"/$pat 2>/dev/null | tail -n +$((KEEP + 1)) | xargs -r rm -f
done
echo "backup ok: $DEST/router-$stamp.db.gz ($(du -h "$DEST/router-$stamp.db.gz" | cut -f1)), keeping $KEEP"

View File

@@ -0,0 +1,14 @@
# Hourly, with 24 kept by default -- roughly a day of hourly granularity.
# Deliberately more frequent than the 2h poller: the poller refetches a remote
# catalog that can always be refetched, while energy_observations and
# route_decisions exist nowhere else.
[Unit]
Description=Hourly snapshot of the LLM router database
[Timer]
OnBootSec=5min
OnUnitActiveSec=1h
Persistent=true
[Install]
WantedBy=timers.target

View File

@@ -0,0 +1,14 @@
# Pushes irreplaceable-and-small state to a private off-site git repo.
# Local snapshots survive `git clean -fdx` (docs/incidents.md #5) because they
# live outside the repo; they do NOT survive the disk. This does.
[Unit]
Description=Push LLM router state off-site to a private git repo
Documentation=file:%h/Sources/6krrt/docs/incidents.md
After=network-online.target
Wants=network-online.target
[Service]
Type=oneshot
WorkingDirectory=%h/Sources/6krrt
Environment=REPO=%h/Sources/6krrt
ExecStart=%h/Sources/6krrt/deploy/offsite-sync.sh

View File

@@ -0,0 +1,13 @@
# Every 6h. Less frequent than the hourly local snapshot on purpose: this one
# needs the network and pushes a ~5MB binary daily, so the local timer stays
# the fine-grained safety net and this is the off-site floor.
[Unit]
Description=Periodic off-site backup of LLM router state
[Timer]
OnBootSec=15min
OnUnitActiveSec=6h
Persistent=true
[Install]
WantedBy=timers.target

85
deploy/offsite-sync.sh Executable file
View File

@@ -0,0 +1,85 @@
#!/usr/bin/env bash
# Push the irreplaceable-and-small state to a private off-site git repo.
#
# Closes the last gap from docs/incidents.md #5: local snapshots survive
# `git clean -fdx` because they live outside the repo, but they do NOT survive
# the disk. This does.
#
# WHAT IS SYNCED
# .omo/ plan artifacts + evidence ledger (13MB, text, versions well)
# systemd units tuned: graceful-shutdown fix, timers
# opencode plugins session-registry.js etc. -- hand-written, not from a package
# claude memory cross-session project memory
# config.local.yaml operator tariff/overrides (personal, NOT a credential)
# router.db gzipped, at most once per day -- see SIZE below
#
# WHAT IS DELIBERATELY NOT SYNCED
# .env — the provider API key. There is no usable secret key on this host to
# encrypt it to (only public keys), so it would land in git history in
# plaintext, and git history is forever even in a private repo. An API
# key is REPLACEABLE: regenerate it from the provider. 22,000 energy
# observations are not. It stays in the local backups only.
# .venv, node_modules — rebuildable from pinned requirements.
# ollama models — ~30GB, re-pullable; recipes in docs/local-models.md.
#
# SIZE: router.db is ~5MB gzipped and binary, so git cannot delta it. It is
# committed at most DAILY (not hourly) to keep growth near 5MB/day rather than
# 120MB/day. Set SYNC_DB=0 to skip it entirely and rely on local snapshots.
set -euo pipefail
REPO="${REPO:-$HOME/Sources/6krrt}"
REMOTE="${OFFSITE_REMOTE:-ssh://git@git.adlee.work:2222/alee/6kbackups.git}"
WORK="${OFFSITE_WORKDIR:-$HOME/.local/share/6krrt-offsite}"
SYNC_DB="${SYNC_DB:-1}"
if [ ! -d "$WORK/.git" ]; then
git clone "$REMOTE" "$WORK" 2>/dev/null || { mkdir -p "$WORK"; git -C "$WORK" init -q; git -C "$WORK" remote add origin "$REMOTE"; }
fi
cd "$WORK"
git fetch -q origin 2>/dev/null || true
git checkout -q -B main 2>/dev/null || true
git reset -q --hard origin/main 2>/dev/null || true
rsync -a --delete "$REPO/.omo/" ./omo/ 2>/dev/null || true
mkdir -p home config
rsync -a --delete "$HOME/.config/systemd/user/" ./home/systemd-user/ 2>/dev/null || true
rsync -a --delete "$HOME/.config/opencode/" ./home/opencode/ --exclude 'cache/' --exclude 'log/' --exclude '*.log' 2>/dev/null || true
rsync -a --delete "$HOME/.claude/projects/-home-alee-Sources-6krrt/memory/" ./home/claude-memory/ 2>/dev/null || true
[ -f "$REPO/config/config.local.yaml" ] && cp -f "$REPO/config/config.local.yaml" ./config/
# Daily DB snapshot, same filename so the working tree stays flat.
if [ "$SYNC_DB" = "1" ]; then
today=$(date +%Y-%m-%d)
if [ ! -f .db-stamp ] || [ "$(cat .db-stamp)" != "$today" ]; then
tmp=$(mktemp)
sqlite3 "$REPO/router.db" ".backup '$tmp'"
[ "$(sqlite3 "$tmp" 'SELECT integrity_check FROM pragma_integrity_check LIMIT 1;')" = "ok" ] \
&& { gzip -c "$tmp" > router.db.gz; echo "$today" > .db-stamp; } \
|| echo "db snapshot failed integrity_check; not synced" >&2
rm -f "$tmp"
fi
fi
cat > README.md <<'INNER'
# 6krrt workstation backup
Off-site copy of state the main repo does not carry. See
`docs/incidents.md` #5 in the main repo for why this exists.
**Not here on purpose:** `.env` (provider API key — regenerate it instead;
there is no usable secret key on the source host to encrypt it to), `.venv`,
`node_modules`, and Ollama models.
Restore: clone the main repo, rebuild the venv from pinned requirements, then
copy `omo/` back to `.omo/`, `config/config.local.yaml` into place, and
`home/*` to their `~/.config` locations. `router.db.gz` is a daily snapshot.
INNER
git add -A
if git diff --cached --quiet; then
echo "offsite: no changes"
else
git -c user.name="6krrt-backup" -c user.email="backup@localhost" \
commit -q -m "backup $(date -Iseconds)"
git push -q -u origin main && echo "offsite: pushed $(git rev-parse --short HEAD)"
fi

86
deploy/workstation-backup.sh Executable file
View File

@@ -0,0 +1,86 @@
#!/usr/bin/env bash
# Tarball everything about THIS workstation that git does not have.
#
# The repo itself is pushed to Gitea and needs no backup. What has no other
# copy is the ignored/untracked state around it -- and docs/incidents.md #5
# records what happens when that is lost: `git clean -fdx` took router.db,
# .env, .venv and config/config.local.yaml in one command, and recovery was
# luck.
#
# INCLUDED (irreplaceable or expensive to recreate):
# router.db measurement history; cannot be regenerated
# .env provider API key
# config/config.local.yaml operator tariff and local overrides
# .omo/ plans, evidence ledger, boulder state
# ~/.config/systemd/user tuned units (the graceful-shutdown fix etc.)
# ~/.config/opencode agent plugins incl. session-registry.js
# ~/.claude/.../memory cross-session memory for this project
#
# EXCLUDED on purpose:
# .venv rebuildable: python -m venv .venv && pip install -r requirements.txt
# node_modules rebuildable
# ollama models ~30GB and re-pullable; Modelfile recipes are in docs/local-models.md
# .git the remote has it
set -euo pipefail
REPO="${REPO:-$HOME/Sources/6krrt}"
DEST="${LLM_ROUTER_BACKUP_DIR:-$HOME/.local/share/6krrt-backups}"
KEEP="${WORKSTATION_BACKUP_KEEP:-7}"
mkdir -p "$DEST"
stamp=$(date +%Y%m%d-%H%M%S)
out="$DEST/workstation-$stamp.tar.gz"
staging=$(mktemp -d)
trap 'rm -rf "$staging"' EXIT
# Snapshot the DB properly rather than tarring a live file: `cp`/`tar` on an
# open SQLite database can capture a torn page set that still looks valid.
sqlite3 "$REPO/router.db" ".backup '$staging/router.db'"
mkdir -p "$staging/repo/config"
[ -f "$REPO/.env" ] && cp -f "$REPO/.env" "$staging/repo/.env"
[ -f "$REPO/config/config.local.yaml" ] && cp -f "$REPO/config/config.local.yaml" "$staging/repo/config/"
[ -d "$REPO/.omo" ] && cp -a "$REPO/.omo" "$staging/repo/.omo"
mkdir -p "$staging/home"
[ -d "$HOME/.config/systemd/user" ] && cp -a "$HOME/.config/systemd/user" "$staging/home/systemd-user"
[ -d "$HOME/.config/opencode" ] && cp -a "$HOME/.config/opencode" "$staging/home/opencode"
mem="$HOME/.claude/projects/-home-alee-Sources-6krrt/memory"
[ -d "$mem" ] && cp -a "$mem" "$staging/home/claude-memory"
cat > "$staging/RESTORE.md" <<'INNER'
# Restoring this workstation
The repo comes from Gitea; this tarball has only what git does not.
git clone ssh://git@git.adlee.work:2222/alee/6krrt.git
cd 6krrt
python3 -m venv .venv && .venv/bin/pip install -r requirements.txt
Then from this archive:
cp router.db <repo>/router.db
cp repo/.env <repo>/.env && chmod 600 <repo>/.env
cp repo/config/config.local.yaml <repo>/config/
cp -a repo/.omo <repo>/.omo
cp -a home/systemd-user/* ~/.config/systemd/user/ && systemctl --user daemon-reload
cp -a home/opencode ~/.config/opencode
Ollama models are not here. Re-pull and re-tag per docs/local-models.md:
ollama pull qwen2.5-coder:14b && ollama pull qwen3-vl:4b
(then the Modelfile num_ctx tags documented there)
Verify:
sqlite3 <repo>/router.db "select integrity_check from pragma_integrity_check limit 1;"
systemctl --user restart llm-router.service && curl -s localhost:8080/health
INNER
tar -czf "$out" -C "$staging" .
# Verify the archive reads back before rotating anything away.
tar -tzf "$out" >/dev/null || { rm -f "$out"; echo "archive FAILED to verify; discarded" >&2; exit 1; }
# shellcheck disable=SC2012
ls -1t "$DEST"/workstation-*.tar.gz 2>/dev/null | tail -n +$((KEEP + 1)) | xargs -r rm -f
echo "workstation backup ok: $out ($(du -h "$out" | cut -f1)), keeping $KEEP"

View File

@@ -0,0 +1,33 @@
# Local Overlay (`config/config.local.yaml`)
`config.local.yaml` is a gitignored, deep-merged overlay on top of the
tracked `config/config.yaml`. Anything placed in the overlay takes
precedence when `load_config()` merges the two files — the overlay only
needs the keys you want to change; everything else falls through to the
base.
## What belongs in the overlay
Values that are specific to your machine or deployment and would turn
`config/config.yaml` dirty if tracked:
- `local_energy.enabled` + `local_energy.tariff_usd_per_kwh` — metering
per-machine, tariff often differs between sites.
- `classifier.base_url` / `verification.base_url` — Ollama may live on a
different host or VPN address on each box.
- Any tuning knob you change between machines (`num_ctx` overrides,
per-host profile tweaks).
## What does NOT belong here
**Secrets.** Secrets (`NEURALWATT_API_KEY`, etc.) stay in `.env`. The
overlay is YAML loaded as configuration, not as a credential store.
## A dirty `config/config.yaml` is a smell
If your working copy of `config/config.yaml` is modified, someone edited
the shared defaults by hand. Move those edits into a new entry in
`config.local.yaml` — the base file should stay clean enough to commit
and share.
See `config/config.local.yaml.example` for concrete key examples.

View File

@@ -1,12 +1,16 @@
# Incidents: how this router has broken, and how to tell which one it is
Four times now, a change that looked local to the router has silently degraded
Five times now, a change that looked local to the router has silently degraded
the agent depending on it. They share a shape worth naming: **none of them
announce themselves as router problems.** Three of the four presented as an
announce themselves as router problems.** Four of the five presented as an
opaque client-side error — a connection refused, an "Unprocessable Content", an
"internal server error" — and diagnosing each meant knowing which log or table
to look in.
**#5 is the only one so far that destroyed data**, and the only one where
recovery depended on luck rather than design. Read it before running any
cleanup command in this repo.
This page exists so the next one takes minutes rather than hours. Start with the
symptom table, then read only the relevant section.
@@ -24,6 +28,7 @@ and why the diagnostics below are worth having to hand.
| Agent dies mid-task on an opaque 4xx | #3 candidate set empty | `sqlite3 router.db "select task_tier, required_context_tokens, rejected_reason from route_decisions where selected_model is null order by id desc limit 5;"` |
| "internal server error", dashboard blank | #4 config/db path | `journalctl --user -u llm-router --since '10 min ago' \| grep -c '" 500'` then `git diff config/config.yaml` |
| Prices/windows look wrong, nothing errors | catalog frozen | `sqlite3 router.db "select max(last_updated) from models;"` |
| 500s + `no such table`, venv/.env gone | #5 `git clean -fdx` | `ls -la router.db .env .venv` — a 0-byte db and a missing `.venv` is conclusive |
That last row is not an incident yet — it is the silent-staleness failure
described in the "Run as a service" section of `CLAUDE.md`. An unpolled catalog
@@ -151,14 +156,93 @@ live service reads, restore it in the same step and verify
`curl -s localhost:8080/health` before moving on. A QA step that leaves
production broken has not passed.
## #5 — `git clean -fdx` destroyed the database, key and venv (2026-09-04)
**Symptom.** Router returned 500 on every request. Journal showed
`sqlite3.OperationalError: no such table: energy_observations` while systemd
reported the service `active`.
**Cause.** An agent ran `git clean -fdx` in the repo. The `-x` flag removes
**ignored** files as well as untracked ones, and everything this deployment
needs to run is ignored by design:
| lost | what it was |
|---|---|
| `router.db` | truncated to 0 bytes — 22,776 energy observations, 17,321 route decisions, 148 proficiency rows |
| `.env` | the NeuralWatt API key |
| `.venv` | the virtualenv the systemd unit's `ExecStart` runs from |
| `config/config.local.yaml` | the operator's electricity tariff |
| `node_modules` | |
This is worse than it looks from the command. `git clean -fd` is a reasonable
thing for an agent to run to get a clean tree. Adding `-x` turns it from
"discard my scratch files" into "delete the deployment", and nothing in the
repo warns you.
**Recovery was luck, not design.** There is no backup of `router.db` by policy.
What saved it was that the plan running at the time had made a QA copy at
`/tmp/qa-config-local-overlay-r2/` **25 seconds before the wipe**, and that
copy happened to include `.env`. Integrity check passed and the restore was
effectively lossless. Had that plan been a different one, the entire
measurement history of the project would be gone.
**Recovery steps, in order:**
```bash
# 1. find a surviving copy — QA/scratch dirs are the likely place
find /home/alee /tmp -name "router.db" -size +0
sqlite3 <candidate> "select count(*) from energy_observations;"
sqlite3 <candidate> "select integrity_check from pragma_integrity_check limit 1;"
# 2. restore data, key, overlay
cp -f <candidate> router.db
cp -f <candidate-dir>/.env .env && chmod 600 .env
# config/config.local.yaml from your own copy
# 3. rebuild the venv (pinned, so this is deterministic)
python3 -m venv .venv && .venv/bin/pip install -r requirements.txt
# 4. restart and verify
systemctl --user restart llm-router.service
curl -s localhost:8080/health
```
**What actually limited the damage** was two unrelated decisions made minutes
earlier: the in-flight plan's 11 files had just been committed rather than left
uncommitted, and the operator's tariff had been parked *outside* the repo
instead of restored in place. Both were reactions to the same file being
clobbered repeatedly that day — the seventh time is what prompted moving it out
of reach.
**Known gap this leaves open.** `router.db` has no backup policy. It holds
every energy/cost observation, every routing decision, and all proficiency
scores — none of which can be rebuilt without re-running evals that cost real
money, and the historical observations cannot be rebuilt at all. A periodic
snapshot is cheap insurance and does not exist.
**Rules that came out of it:**
- **Never `git clean -x` in this repo.** Use targeted paths. If you need a
clean tree, `git stash` preserves; `clean` destroys.
- **Never clean untracked files you did not create.** An untracked file in this
tree is as likely to be operator data as build residue.
- Before any destructive git command, ask what is *ignored*, not just what is
untracked. Here that list is the database, the API key and the runtime.
---
## The pattern
All four are defensible local decisions — free a port, stop a process, deprecate
an expensive model, point at a test database — that silently removed capability
while the router kept reporting itself healthy. Two of the four were caused by
an agent working on the router, using the router.
All five are defensible local decisions — free a port, stop a process, deprecate
an expensive model, point at a test database, clean the working tree — that
silently removed capability while the router kept reporting itself healthy.
Three of the five were caused by an agent working on the router, using the
router.
#5 breaks the pattern in one way worth noting: it did not degrade quietly, it
failed loudly and immediately. What it removed was not capability but *state* —
and state, unlike capability, does not come back when you fix the code.
The generalisable fix is the same each time: **compute the thing that is
actually true, and surface it where the operator is already looking.** That is

View File

@@ -0,0 +1,192 @@
# Admin profile CRUD should write to the overlay
**Status: FINAL — decision-complete.** Written 2026-09-04 against
`feat/config-local-overlay` at `0570123` (PR #26). Depends on that PR landing.
## Why this exists
`plans/config-local-overlay.md` states the rule without qualification:
> **The admin portal writes to `config/config.local.yaml`. It never writes
> `config/config.yaml`.**
The implementation splits it. Allowlisted scalars go to the overlay; **profile
CRUD writes the base file directly**, documented at `src/admin.py:670`.
This is not a considered exception. It is chronological accident: profile CRUD
shipped in PR #25 (`340453e`), and the overlay decision was made afterwards
(`e5ee92b`). Plan 8 then implemented the new rule for the scalar path it
touched and left the profile path where it was.
## The reasoning is the same one the user already endorsed
Plan 8's justification for the scalar path:
> An operator changing a knob in a loopback-only admin portal is making a
> **local operational decision**, not a project decision. Someone changing a
> project default edits `config/config.yaml` in the repo and commits it,
> deliberately, through git.
Creating `onlycheaps` in the portal is that same act. Nothing about the
argument depends on the value being a scalar rather than a nested object — the
only reason the code diverges is the ordering above.
## What it costs to leave it
The tariff no longer lives in `config/config.yaml`, so the blast radius is far
smaller than the six clobberings Plan 8 was written for. The residual harm is
real but narrower:
- Using the profiles UI **re-dirties a tracked file**. `git pull --rebase`
refuses on a dirty tree — observed, and listed in Plan 8's problem statement.
- A `git switch` or `git checkout` silently discards a profile the operator
just created, with nothing indicating why.
- `commit -am` sweeps profiles into unrelated commits.
The sharpest version: **a partial guarantee is worse than none.** An operator
who has internalized "the portal writes to the overlay, `config.yaml` stays
clean" stops checking. The failure then arrives against an expectation this
project's own fix created.
## Two defects that exist TODAY, independent of the rule
Both were found by reading the code on 2026-09-04, and the first was confirmed
live. They are the reason this is a bug fix and not only a consistency tidy.
### 1. An overlay-defined profile is live in routing and invisible to the portal
`_persisted_profiles()` (`src/admin.py:~789`) reads **base only**:
```python
store = load_config_store_safe(config_path) or {}
return store.get("profiles") or {}
```
But `load_config` deep-merges the overlay, so a `profiles:` block in
`config/config.local.yaml` **is** live in `cfg.profiles`. Confirmed by writing
a temporary `ghosttest` profile into the overlay:
| reader | sees |
|---|---|
| `load_config` (what routing uses) | `['ghosttest']` |
| `_persisted_profiles()` (what the portal lists) | `[]` |
So the portal's profile list can already disagree with what the router will
actually accept as `auto:<name>`. That is the same class of failure as the
vision-ceiling incident — a real state, computed correctly somewhere, never
surfaced.
Note this is reachable **today** by hand-editing the overlay, before any of
this plan lands. It is not created by the redirect; the redirect is what fixes
it, because the portal will finally be reading the file it writes.
### 2. Delete reads merged config but writes the base file
`admin_profile_delete` deliberately reads the merged store so the
`default_profile` guard is correct regardless of which file set it:
```python
merged = _load_merged_config_store(config_path, config_local_path)
persisted_profiles = merged.get("profiles") or {}
```
then deletes from base:
```python
_persist_config_block(config_path, ("profiles", name), None, delete=True)
```
For an overlay-defined profile that is found-then-not-deletable: the existence
check passes on merged, the write raises `KeyError`, and the operator gets a
**404 for a profile the portal just confirmed exists**. The endpoint is already
half-overlay-aware, which is a good sign the split was never intentional.
## The rule
| operation | destination |
|---|---|
| create | `config/config.local.yaml` |
| update | `config/config.local.yaml` |
| delete | overlay-defined profiles only |
| base-defined profile | **read-only**, alongside built-ins |
## The deletion wrinkle, and why it resolves rather than blocks
Deep merge can add and override. It cannot express *remove*. A profile defined
in base `config/config.yaml` cannot be deleted from the overlay without a
tombstone, and tombstones are exactly the config-framework machinery Plan 8's
non-goals rule out ("resist the config-framework instinct").
**Resolve it by classification, not mechanism.** A profile someone committed to
`config/config.yaml` *is* a project artifact — the same category as a built-in.
Render it read-only, refuse edit and delete, and say why in a message that
points at the repo.
This is cheap because the refusal path already exists at three sites, each
guarding on `name in BUILTIN_PROFILES`:
- `src/admin.py:902` — create
- `src/admin.py:936` — update
- `src/admin.py:960` — delete
The change is a widened predicate and a distinct message, not new machinery.
Keep the two refusals **distinguishable**: a built-in cannot be changed at all,
while a base-defined profile can be changed by editing the repo. Collapsing
both into "read-only" would tell the operator less than the code knows.
It also gives the provenance display Plan 8 already built a real job: base vs
overlay is precisely what explains why a given profile is not editable.
## Interactions to get right
- **`routing.default_profile` may name a profile from either file.** The
existing delete guard already reads merged config for this; keep it. Deleting
the current default stays refused (422), and that check must run against the
merged view, not the overlay alone.
- **A name collision between base and overlay** must resolve to the overlay
(standard merge precedence) and be **visible as such** in the listing, not
silently deduplicated. Two definitions for one name is the ambiguity Plan 7
refused to accept for built-ins; the same reasoning applies here.
- **Built-in name collisions stay rejected at config load**, unchanged.
- **`_profile_probe` / `_probe_candidate_zero_admit` do not change.** Admission
is still computed via `routing.select_candidates`; this plan moves a write
destination and a visibility source, not any routing logic.
- **The zero-admit warning still fires** on save, unchanged.
## Non-goals
- No tombstone or delete-marker syntax in the overlay. If a base-defined
profile must go, it goes through git.
- Do not mirror profile writes into both files. Plan 8 settled that: one value,
one home.
- Do not migrate existing base-defined profiles into the overlay automatically.
Same reasoning as Plan 8's Migration section — this plan does not decide
where someone else's committed config lives.
- No changes to `RoutingProfile`'s fields, to built-in definitions, or to how
`auto:<name>` resolves.
- Do not widen `_CONFIG_ALLOWLIST`.
## Success criteria
- Creating a profile through the portal writes `config/config.local.yaml` and
leaves `config/config.yaml` **byte-identical** — asserted directly, since the
existing `test_profile_create_preserves_comments_and_does_not_create_overlay`
asserts the opposite and must be inverted rather than deleted.
- Updating an overlay-defined profile writes the overlay only.
- Deleting an overlay-defined profile removes it from the overlay only.
- A base-defined profile returns 403 on update and delete, with a message
distinct from the built-in message and naming `config/config.yaml`.
- **The portal lists overlay-defined profiles.** A test writes a profile into
the overlay and asserts `GET /admin/api/profiles/` returns it — this is the
regression test for defect #1 and must fail on current `main`.
- A base/overlay name collision resolves to the overlay and is labelled with
its source in the listing.
- Deleting the profile named by `routing.default_profile` is still refused when
that setting comes from **either** file.
- The overlay is created on first profile write when absent, with the same
header comment the scalar path writes.
- Backup-before-write and validation-of-merged-config still hold on the profile
path.
- Full suite green with `local_energy.enabled` both true and false.
- The user's `config/config.local.yaml` values are untouched: tariff `0.159`
and `enabled: true` verbatim, verified after the test run, not before.

View File

@@ -0,0 +1,258 @@
# Generalizing to multiple providers
**Status: PARKED 2026-09-04 — DRAFT, not decision-complete, and now blocked on
provider selection rather than on design.** Written 2026-09-04 against `main`
at `340453e`; coupling surface measured and three open questions closed by code
inspection on the same commit.
Parked by the user: **Z.ai is no longer a settled first target.** The GLM
overlap that made it attractive is a real argument, but fit is unconfirmed and
the candidate search is still open. Nothing below should be read as a
commitment to a specific provider.
Remaining gaps before this can go to the opencode/Prometheus pipeline:
1. **Which provider goes first** (was treated as answered; it is not).
2. The proficiency-sharing decision — a judgement call, not a lookup.
3. The eco-without-telemetry decision.
4. Success criteria / test plan / phasing, still empty.
**What does NOT need to wait.** The measured coupling surface below is
provider-agnostic: the 20 hardcoded `neuralwatt` references are literals that
are wrong regardless of which provider lands second, and three of them are
latent defects today. Re-verified on 2026-09-04 against
`feat/config-local-overlay` — still 20 references, same six-file distribution.
Those can be parameterised independently of this plan and should not be held
hostage to the provider question.
## Why this, why now
Everything downstream of the catalog — `scoring.py`, `tiering.py`,
`routing.py`, `proficiency.py`, `verification.py`, `feedback.py`, the admin
portal, the TUI — already operates on normalized DB rows with no idea which
provider produced them. `dispatch_providers: dict[str, DispatchProvider]` in
config and the `(model_id, provider)` composite key were deliberately kept
when OpenRouter was dropped (`c3484f0`) for exactly this. The coupling that's
actually left is narrow:
- **`poller.py`** — `fetch_neuralwatt()` is a bespoke function: hardcoded
URL, NeuralWatt's specific catalog JSON shape, `provider="neuralwatt"`
written as a literal.
- **`dispatcher.py`** — three things tangled together that need separating:
1. Usage/telemetry parsing (NeuralWatt reports energy via SSE **comment**
lines — a protocol quirk, not an OpenAI standard).
2. Cost semantics (the $8/kWh-billed-capped-at-3x-list model is
NeuralWatt-specific billing behavior). What `routing.estimated_cost`
actually scores on — catalog price × request shape — is already
provider-agnostic; the kWh math is validation for one provider's
billing quirk, not the load-bearing input.
3. Account-refusal handling (`degrade to local` assumes one cloud account
— see dispatcher.py ~2537). With N providers this mostly *disappears*:
circuit-break the failing provider's rows and let ranking fail over to
the next-best candidate on a different provider.
**On re-adding OpenRouter specifically:** it was dropped for a real reason,
not a bad one — the project pivoted to scoring on *measured* billing and
carbon, and OpenRouter (an aggregator) doesn't expose per-request
energy/carbon telemetry the way NeuralWatt does. That's not a reason to
avoid it now — it's the first real test case for a `has_energy_telemetry`
capability flag, since eco/energy needs to degrade gracefully per-provider
rather than assuming every row has NeuralWatt's shape. Worth remembering
before re-proposing it as if it were untried.
## Constraint: cheap, no-commitment testing
Budget is small and this is for testing the plumbing, not production spend —
prepaid credits in ~$10 increments, no subscriptions. Candidates, **pricing
and minimums need re-verification before committing to one** (this kind of
thing moves):
| provider | fit | notes |
|---|---|---|
| Z.ai | best first target | Pay-as-you-go API, OpenAI-compatible (`https://api.z.ai/api/paas/v4` or `/api/openai/v1`), no stated minimum top-up. Three free-tier models (GLM-4.7-Flash, GLM-4.5-Flash, GLM-4.6V-Flash) for zero-cost plumbing tests. **Actually the publisher of the GLM family this catalog already serves via NeuralWatt** (`glm-5.2-fast`, `glm-5.2-flex`) — a same-weights cross-provider comparison, which is a stronger generalization test than an unrelated model set, and a way to sanity-check NeuralWatt's `static_fallback` carbon figure for GLM. That is not hypothetical: `CLAUDE.md` records the GLM rows reporting `grid_id: FI` at 475 gCO2/kWh as `carbon_source: static_fallback` — a substituted constant, not a measurement — and they are currently EXCLUDED from eco scoring for that reason. A second provider serving the same weights is a free experiment on one of this project's standing open questions. No response-level energy/carbon telemetry — same capability-flag case as OpenRouter, not a differentiator there. Don't confuse this with the separate "GLM Coding Plan" subscription ($18-168/mo) — that's a different product and not what fits the budget constraint here. |
| OpenRouter | good second | prepaid, no minimum, OpenAI-compatible, many free models for zero-cost plumbing tests, broad catalog overlap (kimi/deepseek/qwen/gemma too, not just GLM), and this repo already has git history (`c3484f0` and its parent) to mine for catalog-normalization shape. No per-request energy/carbon. It's an aggregator of aggregators, so if it ever *does* report grid/energy data, treat it as less trustworthy than a direct provider's own figure. |
| DeepInfra | plausible | prepaid, no subscription, OpenAI-compatible, cheap open-weight catalog. No energy telemetry. |
| Together AI | plausible | prepaid credits, OpenAI-compatible, per-token billing (straightforward vs. NeuralWatt's kWh math). |
| Fireworks AI | plausible | same shape as Together. |
| Groq | maybe later | OpenAI-compatible, free tier + pay-as-you-go, but a much smaller catalog — more useful for a latency-tolerance test than a routing-breadth test. |
**Recommendation, downgraded 2026-09-04.** The *shape* still holds: **one
provider first**, all the way through poller → dispatch → a handful of real
routed requests, before touching a second. That is what confirms the
abstraction generalizes instead of just looking like it does on paper.
Which provider is **reopened**. Z.ai was the pick on the strength of the GLM
overlap — a same-weights cross-provider comparison, and a free experiment on
the `static_fallback` carbon question. That argument is still good and should
be re-used to score whatever candidate comes next; it was never a claim that
Z.ai's API, pricing or terms actually fit. Treat the table above as a
shortlist to re-verify, not a ranking to execute.
Worth noting for the search: the criterion that made Z.ai attractive —
**catalog overlap with what NeuralWatt already serves** — is separable from
Z.ai itself. OpenRouter carries kimi, deepseek, qwen and gemma alongside GLM,
so it satisfies the same-weights test more broadly, and this repo already has
git history (`c3484f0` and its parent) with a working catalog normalizer to
mine. It was dropped for a reason that no longer disqualifies it, now that
`has_energy_telemetry` is the planned answer to missing energy data.
## Architecture sketch
A `Provider` protocol/interface — not fleshed out yet, but the shape:
- `fetch_catalog() -> list[ModelRow]`
- `parse_usage(response) -> UsageInfo` (cost, tokens, energy if present)
- `detect_refusal(error) -> bool`
- capability flags: `has_energy_telemetry`, `has_regional_carbon`
`fetch_neuralwatt` and the SSE-comment parser move behind it as the first
concrete implementation. `dispatch_providers` entries get a `type:`
discriminator so config knows which implementation to instantiate.
## Measured coupling surface
Claims about how narrow the coupling is should be counted, not asserted.
`grep -rn neuralwatt src/*.py` on `main` at `340453e` returns **20 hardcoded
references across 6 files**:
| file | refs | what they are |
|---|---|---|
| `poller.py` | 8 | the bespoke fetch: URL, `fetch_neuralwatt()`, literal `provider="neuralwatt"`, the sanity-floor count query, log prefixes |
| `eval_proficiency.py` | 5 | `provider: str = "neuralwatt"` defaults, the judge's `dispatch_providers["neuralwatt"]` lookup, and an `identity["provider"] != "neuralwatt"` skip |
| `dispatcher.py` | 4 | passthrough default, and a capability lookup (see below) |
| `metrics.py` | 1 | catalog-staleness query |
| `leaderboard.py` | 1 | `set_leaderboard(..., "neuralwatt", ...)` |
| `seed_energy.py` | 1 | `dispatch_providers["neuralwatt"]` |
That is the actual work list. Two entries deserve calling out because they
**contradict the "needs ~zero change" section below.**
### `metrics.py:175` — catalog staleness would ignore a second provider
```sql
SELECT MAX(last_updated) AS last_updated FROM models WHERE provider='neuralwatt'
```
The staleness warning is computed over NeuralWatt rows only. Add a provider
whose poller silently stops and the warning stays green while its catalog
freezes — the exact silent-and-open failure `CLAUDE.md`'s "Run as a service"
section describes, now with a second way in. `metrics.py` is listed under
"needs ~zero change"; it does not.
### `leaderboard.py:125` — priors are written to a hardcoded provider
`set_leaderboard(conn, cfg, model_id, "neuralwatt", category, score)` pins the
provider literal, so a curated prior for a shared model (GLM, say) attaches to
the NeuralWatt row and never to the Z.ai one. `leaderboards.yaml` ships empty
today so nothing is broken yet, but this decides the answer to the
proficiency-sharing question above by accident rather than on purpose.
### `dispatcher.py:2879` — `_model_exists` silently strips a real model id
**Corrected 2026-09-04.** An earlier revision of this plan attributed this line
to `_check_pinned_capabilities`. That was wrong: `_check_pinned_capabilities`
(`dispatcher.py:2887`) already takes `provider` as a parameter and passes it to
its query. It is not a coupling site. Line 2879 belongs to `_model_exists`,
and the defect there is worse.
```sql
SELECT 1 FROM models WHERE model_id = ? AND provider = 'neuralwatt'
```
`_model_exists` decides whether a `provider/model` string a client sent is an
opencode-style alias to **strip** (`llm-router/...`) or a real id that merely
contains a slash. Scoped to NeuralWatt, a real second-provider id returns
False and gets treated as an alias.
That matters more than it looks, because **`vendor/model` is the native id
format for the two strongest candidates**: OpenRouter ids are always
`deepseek/deepseek-chat`-shaped, and several other aggregators follow suit. So
the failure is not exotic — it is the default case for a whole class of
provider.
And it fails **silently in the wrong direction**. A provider-scoped 422 would
at least be loud. This one strips the prefix and proceeds with the remainder,
so a client pinning `deepseek/deepseek-chat` gets whatever `deepseek-chat`
resolves to — a different row, on a different provider, at a different price —
with nothing in the response indicating a substitution occurred. Fail-closed
was the design intent everywhere else in this capability path; here it fails
open.
Requirement: `_model_exists` must resolve across all configured providers, and
alias-stripping must be decided by something other than "no NeuralWatt row has
this id".
### Risk not in the draft: per-provider fetch isolation
`poller.mark_stale` runs only inside `poller.main()`, and `main()` returns
early on `RequestException` — **before** `upsert` and **before** `mark_stale`.
With two providers in one run, a failure fetching provider B can skip
`mark_stale` for provider A entirely, or a partial/empty `data` array from B
can mark rows stale that were never B's. `docs/incidents.md` records the
empty-`data` path as the one that can empty the candidate set.
Requirement: **each provider's fetch, upsert and staleness marking must be
isolated.** One provider's outage must not affect another's rows in either
direction, and the sanity floor (`poller.py:420`) must be per-provider.
## What should need ~zero change
Scoring, tiering, routing, proficiency, verification, feedback, admin, TUI.
If any of these turn out to need provider-aware branching, that's a sign the
interface boundary is in the wrong place — worth treating as a red flag
during implementation, not a shrug.
**Two of them already fail that test**, per the inventory above: `metrics.py`
hardcodes the provider in its staleness query and `leaderboard.py` hardcodes it
when writing priors. Neither is a deep coupling — both are literals, not
branching — but the list should be read as "should need ~zero change *after*
those two literals are parameterised", not as a claim that they are already
clean. The red flag to watch for during implementation is provider-aware
*logic* appearing in these modules; a hardcoded string is a different and much
cheaper problem.
## Open questions
- **Eco/carbon for a provider with no telemetry.** Exclude those rows from
eco ranking entirely, or treat eco as unweighted/missing for them without
disqualifying them? Affects whether a non-NeuralWatt row can ever win on
the eco axis, or only ever competes on cost/quality.
- **Same model, two providers — per-model or per-(model, provider)?**
PARTLY RESOLVED, and the remaining half is a design decision rather than a
verification. The schema already supports divergence: `proficiency`'s
primary key is `(model_id, provider, category)`. But
`propagate_to_variants` copies scores across variants keyed on
`base_model_id`, on the stated principle that proficiency is "a property of
the weights, not the queue". Same-weights-different-serving-stack is exactly
the case that principle does not decide: quantization and serving
differences could justify separate scores, while the existing logic would
share one. **Decide this explicitly before implementation** — it is the one
question here that code inspection cannot answer.
- **RESOLVED — circuit breaker is genuinely provider-generic.** Verified in
code, not prose: `is_down`, `record_failure` and `record_success` all take
`(model_id, provider)` and `_store` is keyed on that tuple
(`src/circuit_breaker.py:29,37,57`). Per-provider breaking works today with
no change.
- **Does the account-refusal-degrades-to-local path actually get simpler or
just get an `if` added?** Stated above as an assumption; worth confirming
against the real code before it's a success criterion.
- **RESOLVED — "no serving class" is already a real state.**
`poller.parse_serving_class` returns the schema defaults
(`'standard'`, `'default'`, `'full'`) for a base id with no suffix, so a
Z.ai or OpenRouter row lands `standard` and is interactive-eligible rather
than null. No change needed.
## Non-goals (this pass)
- Don't wire up more than one new provider before the first one round-trips
end to end.
- Don't try to make eco/carbon methodology uniform across providers that
don't expose the same telemetry — graceful absence beats a fabricated
number (same principle as the empty `leaderboards.yaml`).
- Don't rebuild the cost model — catalog-price × shape already generalizes;
confirm that rather than redesigning it.
## Not yet defined
Success criteria, test plan, and phasing/milestones — fill in once the open
questions above are resolved and this graduates to FINAL.

View File

@@ -0,0 +1,161 @@
# Remove the hardcoded `neuralwatt` literals
**Status: FINAL — decision-complete.** Written 2026-09-04 against
`feat/config-local-overlay` at `0570123`.
## Scope, stated first because it is easy to over-read
This plan does **not** add a provider, and does **not** build the `Provider`
protocol sketched in `plans/multi-provider-support.md`. That plan is PARKED
pending provider selection.
This one removes 20 hardcoded string literals that are wrong regardless of
which provider ever lands second — and fixes the three that are latent defects
today. It is separable from the parked plan by construction: nothing here needs
to know what the second provider is.
`(model_id, provider)` is already the composite key throughout the schema, and
`dispatch_providers: dict[str, DispatchProvider]` (`src/config.py:779`) is
already a mapping. The literals are the gap between that design and the code.
## The inventory
`grep -rn neuralwatt src/*.py` — **20 references across 6 files**, re-verified
2026-09-04, unchanged from the count taken at `340453e`.
| file | refs | character |
|---|---|---|
| `poller.py` | 8 | the bespoke fetch — URL, function name, literal `provider=`, sanity-floor query, log prefixes |
| `eval_proficiency.py` | 5 | defaults and a `dispatch_providers[...]` lookup |
| `dispatcher.py` | 4 | one real defect, one passthrough default, two conditionals |
| `metrics.py` | 1 | staleness query — **defect** |
| `leaderboard.py` | 1 | prior write — **defect** |
| `seed_energy.py` | 1 | `dispatch_providers[...]` lookup |
## The three that are defects now
### 1. `dispatcher.py:2879` — `_model_exists` fails OPEN
```python
def _model_exists(model_id: str) -> bool:
row = conn.execute(
"SELECT 1 FROM models WHERE model_id = ? AND provider = 'neuralwatt'",
(model_id,),
).fetchone()
```
This decides whether a `provider/model` string a client sent is an
opencode-style alias to **strip** (`llm-router/...`) or a real id that merely
contains a slash. Scoped to one provider, a real second-provider id returns
False and is treated as an alias.
`vendor/model` is the **native id format** for OpenRouter and most aggregators
(`deepseek/deepseek-chat`), so this is the default case for a whole class of
provider, not an edge one.
And the failure direction is wrong. A provider-scoped 422 would be loud. This
strips the prefix and proceeds with the remainder, so the request dispatches to
whatever the remainder resolves to — a different row, potentially a different
provider, at a different price — with nothing in the response marking a
substitution. Every other capability check in this path **fails closed** by
design; this one fails open.
**Note the near miss.** `_check_pinned_capabilities` (`dispatcher.py:2887`) sits
immediately below it, already takes `provider` as a parameter, and is correct.
An earlier revision of the multi-provider draft blamed 2879 on that function;
it does not. Fix the right one.
### 2. `metrics.py:175` — catalog staleness ignores any other provider
```sql
SELECT MAX(last_updated) AS last_updated FROM models WHERE provider='neuralwatt'
```
The staleness warning is computed over one provider's rows. A second provider
whose poll silently stops leaves the warning green while its catalog freezes.
That is precisely the **silent-and-open** failure `CLAUDE.md`'s "Run as a
service" section describes, gaining a second entrance.
### 3. `leaderboard.py:125` — priors are pinned to one provider
```python
set_leaderboard(conn, cfg, model_id, "neuralwatt", category, score)
```
A curated prior for a shared model attaches to the NeuralWatt row and never to
any other. `leaderboards.yaml` ships empty, so nothing is broken yet — but this
silently *decides* the still-open proficiency-sharing question from
`multi-provider-support.md` by accident rather than on purpose.
Because that question is genuinely undecided, **do not resolve it here.**
Parameterise the call so the provider is passed in rather than assumed, and
leave the sharing policy to the parked plan. Passing the literal from one call
site is a fix; inventing a fan-out rule is a decision this plan has no mandate
to make.
## The rest
Mechanical, and worth doing in the same pass because they are what make the
three above verifiable rather than isolated patches.
- **`poller.py` (8).** Keep `fetch_neuralwatt` as the single concrete fetcher —
this plan does not introduce the protocol — but move the URL, the provider
string, the sanity-floor query and the log prefix so they derive from one
named provider value rather than being spelled out five times. The sanity
floor (`poller.py:420`) must count rows for **the provider being polled**.
- **`eval_proficiency.py` (5)** and **`seed_energy.py` (1).** Turn the
`provider: str = "neuralwatt"` defaults and `dispatch_providers["neuralwatt"]`
lookups into an explicit provider argument threaded from the caller. The
`identity["provider"] != "neuralwatt"` skip at `eval_proficiency.py:724`
becomes a comparison against that argument.
- **`dispatcher.py:3258-3259`** — the passthrough default and its conditional.
Same treatment.
## Risk to respect
`poller.mark_stale` runs only inside `poller.main()`, and `main()` returns early
on a `RequestException` — **before** `upsert` and **before** `mark_stale`.
That is existing single-provider behaviour and this plan does not change it.
But do not parameterise the fetch in a way that makes a future second provider
share one `main()` failure path: `docs/incidents.md` records the empty-`data`
array as the one route that can empty the candidate set, and a shared path
would let one provider's outage mark another's rows stale. **Isolating per
provider is the parked plan's job**; this plan's obligation is not to build a
structure that makes isolation harder later. Where a choice arises, prefer the
shape that keeps one provider's fetch, upsert, sanity floor and staleness
marking together.
## Non-goals
- Do not add a second provider, or provider config beyond what
`dispatch_providers` already holds.
- Do not build the `Provider` protocol, capability flags
(`has_energy_telemetry`), or a `type:` discriminator. Parked plan.
- Do not decide the proficiency-sharing question (see defect 3).
- Do not change the cost model, eco scoring, or any ranking behaviour. This
plan must be observably behaviour-neutral on a single-provider deployment.
- Do not rename the `neuralwatt` provider value itself. Existing rows,
`proficiency` keys and `energy_observations` reference it; a rename is a
migration and is not in scope.
## Success criteria
- `grep -rn neuralwatt src/*.py` returns only the places where the value is
*configured or named* — not places where behaviour is conditioned on it.
State the expected remaining count in the PR so it can be re-checked.
- `_model_exists` resolves across all configured providers, and
alias-stripping is decided by something other than "no NeuralWatt row has
this id". A test pins that a `vendor/model` id belonging to a non-NeuralWatt
row is **not** stripped.
- Catalog staleness is computed per provider; a test with two providers' rows
shows a stale second provider raising a warning.
- `set_leaderboard` receives its provider from the caller; no literal.
- The poller's sanity floor counts rows for the provider being polled, proven
by a test with rows from more than one provider present.
- **Behaviour-neutral on the live single-provider deployment**: routing
decisions, staleness warnings and poller output are unchanged. Pin this with
a before/after comparison on a real catalog, not by inspection.
- Full suite green with `local_energy.enabled` both true and false.
- The user's `config/config.local.yaml` values are untouched — tariff `0.159`
and `enabled: true` verbatim.

119
plans/tui-overhaul.md Normal file
View File

@@ -0,0 +1,119 @@
# TUI overhaul: catch the dashboard up to the schema
**Status: FINAL — decision-complete.** Written 2026-09-04 against `main` at
`a0c7e8e`.
## Why now
The TUI (`src/tui.py`, `tui_model.py`, `tui_screens.py`, `tui_sse.py` — ~1,000
lines total) has drifted behind the data it displays. Five columns were added
to `route_decisions` across this session's merges and none reached the
dashboard, and the decision table has never had a timestamp.
The admin portal got a Profile column in PR #25. The TUI did not. It is the
same data.
## 1. The decision table is missing a timestamp
Declared columns (`src/tui.py:314-316`):
```
"id", "kind", "category", "tier", "ctx", "selected", "est $", "flex"
```
`observed_at` **is already carried** by `decision_row` in `tui_model.py` — it
is fetched and then never rendered. So this is a display change, not a data
change.
Add a `time` column. Render it **short** (`HH:MM:SS`), not the full ISO
timestamp: the rows are dense, the date is almost always today, and the full
form would crowd out `selected`, which is the column people actually read.
Put it first — it is the natural scan axis for a live feed.
## 2. Five columns exist in the schema and are invisible
Measured by diffing `PRAGMA table_info(route_decisions)` against what
`decision_row` exposes:
| column | shipped in | why it matters |
|---|---|---|
| `profile` | PR #25 | which named profile served the request |
| `exploration` | proficiency branch | whether epsilon-greedy picked this, not the ranking |
| `pinch_original_tokens` | PR #18 | context pruning input |
| `pinch_final_tokens` | PR #18 | context pruning output |
| `request_id` | proficiency branch | the join key to `/outcome` reports |
| `session_key` | earlier | hashed session fingerprint |
**Do not add six more columns to the table.** It already has eight and the
terminal is not wide. Instead:
- Add **`profile`** to the table proper. It changes which models were even
considered, so a decision cannot be read without it — the same argument that
earned it a column in the admin portal.
- Add an **`E` flag** in the existing flags idiom for `exploration`, alongside
how `flex` is already rendered. An exploratory pick is not a ranking result
and must be visually distinguishable, or the operator reads a deliberate
random sample as the router's judgement.
- Put **`pinch_*`, `request_id`, `session_key`** in the **detail popup**
(`tui_screens.py`, opened with Enter or `e`), which already shows the full
decision JSON. They are per-decision forensics, not scan-axis data.
## 3. The quota panel shows the wrong thing
**This section depends on `plans/quota-balance-and-burn-rate.md` and must not
land before it.** That plan replaces percentage-of-plan with balance and burn
rate, because the current framing is measurably wrong: the warning says "a
quota is a wall, not a bill — requests fail rather than costing more" while
usage sat at 146% of plan and nothing failed, since the provider bills overage
against a credit balance.
Once that lands, the TUI panel (`#quota-panel`, `#quota-progress`,
`#quota-legend`) should lead with **balance and projected runway**
("$12.19 left, ~23h at current burn") and demote the percentage bar to
secondary. A progress bar against a plan figure that is routinely exceeded is
actively misleading — it implies a ceiling that does not exist.
**Sequencing:** if the quota plan has not landed when this one runs, do items
1, 2 and 4 and leave the panel alone. Do NOT reimplement balance/burn
independently in the TUI — `metrics.quota_burn` is the single source and the
TUI reads `/metrics`.
## 4. Surface the warnings that already exist
`coverage.warnings` from `/metrics` already carries catalog staleness, quota
burn, scoring coverage gaps and ceiling warnings. `#warnings-panel` exists.
Confirm every warning class actually reaches it — the vision-ceiling incident
on 2026-09-04 showed a whole warning family that was computed and never
displayed, and the fix there was surfacing, not computing.
This is a verification task as much as a feature: for each warning the
`/metrics` `coverage.warnings` list can emit, assert it renders.
## Non-goals
- No new data. Everything here is already in `route_decisions` or `/metrics`.
- Do not widen the decision table beyond one added column plus one flag.
- Do not reimplement any metric in the TUI. `tui_model.py` is the pure data
layer over `/metrics` and `/events/decisions`; keep the computation in
`metrics.py`.
- Do not import `textual` outside the TUI modules. The dispatch path must stay
free of the UI dependency — that separation is deliberate and tested.
- No colour/theme rework. This is about information, not appearance.
## Success criteria
- The decision table shows a short `HH:MM:SS` time column, first.
- `profile` is a column; `exploration` renders as a flag beside `flex`.
- `pinch_original_tokens`, `pinch_final_tokens`, `request_id` and
`session_key` appear in the detail popup.
- A test diffs `PRAGMA table_info(route_decisions)` against what the TUI
model exposes and fails if a column is added to the schema without a
decision about surfacing it. **This is the test that stops the drift
recurring** — the rest of this plan is a one-time catch-up, this is the part
that keeps it caught up.
- Every warning class `/metrics` can emit renders in `#warnings-panel`.
- Quota panel leads with balance and runway **if** the quota plan has landed;
otherwise untouched and noted.
- `textual` still imported only by TUI modules (existing test stays green).
- Full suite green with `local_energy.enabled` both true and false.
- The user's `config/config.yaml` values remain uncommitted and verbatim.

View File

@@ -302,9 +302,11 @@ _FLEX_VALUES = frozenset(v.value for v in FlexPreference)
# Dotted config.yaml paths an operator is allowed to edit. Everything else —
# classifier/verification/local_vision URLs and model names, api_key_env,
# provider blocks, dispatch_providers, and secrets — is deliberately OFF this
# list. Editing works only on the named scalars. Writes go through
# comment-preserving ruamel.yaml round-trip and are validated via ``RouterConfig``
# before any byte reaches disk; a backup is made first.
# list. Editing works only on the named scalars. Writes are persisted to
# ``config.local.yaml`` (the machine-local overlay) via a comment-preserving
# ruamel.yaml round-trip; whole-config validation via ``RouterConfig`` happens
# on the merged (base + overlay) result before any byte reaches disk; a backup
# of the overlay file precedes the write.
_CONFIG_ALLOWLIST: dict[str, tuple[str, ...]] = {
"logging.level": ("logging", "level"),
"objective.quality_tolerance": ("objective", "quality_tolerance"),
@@ -430,11 +432,169 @@ def _persist_config_value(
"""Atomically persist *value* at dotted *path* in *config_path*.
Thin wrapper around ``_persist_config_block`` preserving the historic
scalar-path signature used by tests.
scalar-path signature used by tests. On the overlay path this writes
to ``config.local.yaml`` (the machine-local gitignored file).
"""
_persist_config_block(config_path, path, value)
# --- overlay helpers ----------------------------------------------------------
_YAML_HEADER = (
"# Machine-local overlay — gitignored, never committed.\n"
"# Written by the admin portal (POST /admin/api/config/*).\n"
)
def _persist_to(
base_path: Path,
overlay_path: Path,
store_path: tuple[str, ...],
value: Any,
) -> None:
"""Persist *value* at *store_path* inside *overlay_path*, creating the file on first write.
The overlay file is created with a header comment if it does not yet exist.
Unlike ``_persist_config_block`` (which preserves the base file's comments),
the overlay starts fresh — no pre-existing comments to keep.
Whole-config validation via ``RouterConfig`` happens on the **merged**
(base + in-memory overlay with the requested mutation) before any byte
reaches disk; a backup of the overlay file precedes the write; and
``.tmp`` + ``os.replace`` provides an atomic swap.
"""
with _config_write_lock:
existed = overlay_path.exists()
if existed:
store = load_config_store(overlay_path) # CommentedMap | None
else:
store = CommentedMap()
if store is None:
store = CommentedMap()
cur = store
for part in store_path[:-1]:
nxt = cur.get(part)
if nxt is None:
nxt = CommentedMap()
cur[part] = nxt
cur = nxt
cur[store_path[-1]] = value
# Validate the MERGED config (base + intended overlay) before writing.
merged = _load_merged_config_store(
base_path, overlay_path, overlay_override=store
)
RouterConfig(**merged)
# Validation passed: create the overlay file now if this is the first
# write, so there is a pre-modification state to back up.
if not existed:
overlay_path.write_text(_YAML_HEADER)
# Backup the overlay (pre-modification), then atomically swap.
backup = overlay_path.with_name(
f"{overlay_path.name}.bak.{int(time.time())}"
)
if overlay_path.exists():
shutil.copyfile(overlay_path, backup)
# Atomic write. A fresh overlay gets the machine-local header comment;
# on later writes the header is already part of the CommentedMap and
# survives the ruamel round-trip, so we only prepend it explicitly when
# we are creating the file from scratch.
tmp = overlay_path.with_suffix(overlay_path.suffix + ".tmp")
with tmp.open("w") as fh:
if not existed:
fh.write(_YAML_HEADER)
YAML().dump(store, fh)
os.replace(tmp, overlay_path)
def _load_merged_config_store(
base_path: Path,
overlay_path: Path,
*,
overlay_override: Optional[dict[str, Any]] = None,
) -> dict[str, Any]:
"""Load and merge *base_path* with *overlay_path*, returning a plain dict.
The overlay values take precedence; any key in the overlay overrides the
corresponding base key. Both files are loaded via ruamel.CommentedMap
so existing comments survive the GET display.
When *overlay_override* is supplied, it is used instead of loading
*overlay_path*; this lets a write validate against the in-memory mutated
overlay before any byte reaches disk.
If the overlay does not exist and no override is provided, returns a
plain-copy of the base store that can be iterated by ``_CONFIG_ALLOWLIST``
paths.
"""
try:
base = load_config_store_safe(base_path) or {}
except HTTPException:
# Corrupt base — surface a clean error.
raise
if overlay_override is not None:
overlay = overlay_override
else:
try:
overlay = load_config_store(overlay_path) or {}
except YAMLError:
raise HTTPException(
status_code=503,
detail=(
"config.local.yaml could not be parsed — "
"fix the overlay file or delete it to recover"
),
)
except FileNotFoundError:
# Overlay does not exist yet — treat as empty.
overlay = {}
# Deep-merge overlay into base (same semantics as config._merge_overlay).
merged: dict[str, Any] = {}
for key, base_val in base.items():
if (
key in overlay
and isinstance(base_val, dict)
and isinstance(overlay[key], dict)
):
# Recursively merge nested dicts.
rec: dict[str, Any] = {}
ov = overlay[key]
for sk, sv in base_val.items():
if sk in ov and isinstance(sv, dict) and isinstance(ov[sk], dict):
# Flatten: reuse merged parents from outer level.
rec[sk] = _deep_merge_dicts(sv, ov[sk])
else:
rec[sk] = ov.get(sk, sv)
merged[key] = rec
else:
# Scalars, lists, or overlay-only keys: overlay wins.
merged[key] = overlay.get(key, base_val)
# Insert overlay-only keys not in base.
for key, ov_val in overlay.items():
if key not in merged:
merged[key] = ov_val
return merged
def _deep_merge_dicts(base: dict, overlay: dict) -> dict:
"""Recursively merge *overlay* into *base*, overlay winning on conflicts."""
result = {}
for key, bv in base.items():
if key in overlay and isinstance(bv, dict) and isinstance(overlay[key], dict):
result[key] = _deep_merge_dicts(bv, overlay[key])
else:
result[key] = overlay.get(key, bv)
for key, ov in overlay.items():
if key not in result:
result[key] = ov
return result
class _AvailabilityBody(BaseModel):
availability: str
reason: Optional[str] = None
@@ -506,8 +666,9 @@ def build_router(
"""Build the admin APIRouter bound to the caller's config and DB factory.
``base_dir`` is the repo root (the directory holding config.yaml and the
maintenance scripts). The persisted-config endpoints use it to locate
``config.yaml`` for comment-preserving writes; the maintenance triggers use
maintenance scripts). The persisted-config endpoints (allowlisted scalars)
read from ``config.yaml`` but write to ``config.local.yaml``; profile CRUD
endpoints write to ``config.yaml`` directly. The maintenance triggers use
it as their spawn CWD. Defaults to the repo root (parent of ``src/admin.py``),
so the router is portable and testable without an explicit base_dir.
"""
@@ -517,6 +678,7 @@ def build_router(
if base_dir is not None
else _REPO_ROOT / "config" / "config.yaml"
)
config_local_path = config_path.with_name("config.local.yaml")
router = APIRouter()
_admin_frontend = _REPO_ROOT / "admin" / "frontend" / "index.html"
@@ -800,13 +962,15 @@ def build_router(
status_code=403,
detail="built-in profiles are read-only",
)
# Read the CURRENT persisted default (the running cfg may be stale
# relative to the file) so the refusal names the setting explicitly.
store = load_config_store_safe(config_path) or {}
if name not in (store.get("profiles") or {}):
# Read the EFFECTIVE default from merged base + overlay so an admin
# cannot delete a profile that is currently the default, regardless of
# whether the default was set in config.yaml or config.local.yaml.
merged = _load_merged_config_store(config_path, config_local_path)
persisted_profiles = merged.get("profiles") or {}
if name not in persisted_profiles:
raise HTTPException(status_code=404, detail="profile not found")
raw_default = _dict_get_at(
store, _CONFIG_ALLOWLIST["routing.default_profile"]
merged, _CONFIG_ALLOWLIST["routing.default_profile"]
)
if raw_default == name:
raise HTTPException(
@@ -1219,21 +1383,46 @@ def build_router(
@router.get("/api/config")
def admin_config_get() -> dict:
"""The allowlisted config.yaml values, as {dotted_key: value}."""
return {key: _dict_get_at(load_config_store_safe(config_path), path)
for key, path in _CONFIG_ALLOWLIST.items()}
"""The allowlisted config values with provenance.
Each key maps to ``{"value": …, "source": "base" | "overlay"}``
where *overlay* means the key exists in ``config.local.yaml`` and
*base* means it comes only from ``config.yaml``.
"""
overlay_exists = config_local_path.exists()
overlay_raw = None
if overlay_exists:
try:
overlay_raw = load_config_store(config_local_path)
except YAMLError:
pass # Corrupt overlay — source all from base.
source_map: dict[str, str] = {}
for key, path in _CONFIG_ALLOWLIST.items():
try:
_dict_get_at(overlay_raw or {}, path)
source_map[key] = "overlay"
except (KeyError, TypeError):
source_map[key] = "base"
merged = _load_merged_config_store(config_path, config_local_path)
return {
key: {"value": _dict_get_at(merged, path), "source": source_map[key]}
for key, path in _CONFIG_ALLOWLIST.items()
}
@router.post("/api/config/{key}")
def admin_config_write(key: str, body: _ValueBody) -> dict:
"""Persist one allowlisted value to config.yaml (comment-preserving)."""
"""Persist one allowlisted value to the local overlay file."""
if key not in _CONFIG_ALLOWLIST:
raise HTTPException(
status_code=403,
detail=f"config key is not editable: {key}",
)
try:
_persist_config_value(
config_path, _CONFIG_ALLOWLIST[key], body.value
_persist_to(
config_path,
config_local_path,
_CONFIG_ALLOWLIST[key],
body.value,
)
except ValidationError as exc:
raise HTTPException(

View File

@@ -997,11 +997,55 @@ class RouterConfig(StrictModel):
return self
def _merge_overlay(base: dict, overlay: dict) -> dict:
"""Deep-merge *overlay* on top of *base* and return the merged dict.
Merge semantics:
* **Mappings** (``dict``): merged recursively, key by key. Keys present
only in *overlay* are inserted; keys only in *base* are preserved.
* **Lists** (``list``): replaced wholesale — the overlay list entirely
overwrites the base list at that key. Lists are never concatenated.
* **Scalars**: overlay value wins.
An absent overlay file leaves base unchanged, so this helper is
strictly additive. The merged dict is validated once by the caller
so that unknown keys in the overlay surface as Pydantic errors.
Precedence chain (highest → lowest): environment variables >
overlay > base (the base file).
"""
result = {}
# Start with all base keys
for key, base_val in base.items():
if (
key in overlay
and isinstance(base_val, dict)
and isinstance(overlay[key], dict)
):
# Both sides are dicts → recurse
result[key] = _merge_overlay(base_val, overlay[key])
else:
# Scalars, lists, or overlay-only keys: overlay wins (or base if absent)
result[key] = overlay.get(key, base_val)
# Insert overlay-only keys that had no counterpart in base
for key, overlay_val in overlay.items():
if key not in result:
result[key] = overlay_val
return result
def load_config(path: str | Path = "config/config.yaml") -> RouterConfig:
path = Path(path)
if not path.exists():
raise FileNotFoundError(f"Config file not found: {path}")
raw = yaml.safe_load(path.read_text())
local_path = path.with_name("config.local.yaml")
if local_path.exists():
local_raw = yaml.safe_load(local_path.read_text())
raw = _merge_overlay(raw, local_raw)
return RouterConfig(**raw)

View File

@@ -1,13 +1,15 @@
"""Tests for the /admin/api persisted-config endpoints.
``admin.py`` exposes ``GET /admin/api/config`` (the allowlisted config.yaml
values) and ``POST /admin/api/config/{key}`` (persist one allowlisted value to
config.yaml with comment-preserving ruamel.yaml round-trip, a backup copy, and
whole-config validation via ``RouterConfig`` BEFORE anything touches disk).
``admin.py`` exposes ``GET /admin/api/config`` (the allowlisted values with
provenance, ``{value, source}``) and ``POST /admin/api/config/{key}`` (persist
one allowlisted value to ``config.local.yaml`` with comment-preserving
ruamel.yaml round-trip, a backup copy, and whole-config validation of the
merged base + overlay via ``RouterConfig`` BEFORE anything touches disk).
Each test builds its own isolated router against a temp copy of config.yaml by
passing ``base_dir`` (a temp dir) to ``build_router`` — the real repo
``config.yaml`` is never written. It mounts the router on a fresh FastAPI
``config.yaml`` is never written. The overlay endpoints create and mutate
``<tmp_dir>/config/config.local.yaml``. It mounts the router on a fresh FastAPI
TestClient at ``prefix="/admin"``.
"""
@@ -16,6 +18,7 @@ from __future__ import annotations
import re
import shutil
import sqlite3
import subprocess
import threading
from pathlib import Path
@@ -91,14 +94,18 @@ def test_config_GET_returns_allowlisted_values(client):
"routing.default_flex_preference",
):
assert key in body
assert body["logging.level"] == "info"
assert body["objective.quality_tolerance"] == 0.10
assert body["routing.default_flex_preference"] == "auto"
assert body["logging.level"] == {"value": "info", "source": "base"}
assert body["objective.quality_tolerance"] == {"value": 0.10, "source": "base"}
assert body["routing.default_flex_preference"] == {
"value": "auto",
"source": "base",
}
def test_config_POST_preserves_comments_and_changes_value(client):
"""A valid write keeps the file's comments AND updates the value."""
tc, config_yaml = client
local_yaml = config_yaml.with_name("config.local.yaml")
_insert_sentinel(config_yaml)
resp = tc.post("/admin/api/config/logging.level", json={"value": "warning"})
@@ -108,9 +115,14 @@ def test_config_POST_preserves_comments_and_changes_value(client):
assert body["value"] == "warning"
assert "restart is required" in body["message"]
text = config_yaml.read_text()
assert _SENTINEL in text
assert re.search(r"^\s*level:\s*warning\s*$", text, re.MULTILINE) is not None
# The base file is untouched; the overlay now carries the changed value.
base_text = config_yaml.read_text()
assert _SENTINEL in base_text
assert re.search(r"^\s*level:\s*info\s*$", base_text, re.MULTILINE) is not None
local_text = local_yaml.read_text()
assert re.search(
r"^\s*level:\s*warning\s*$", local_text, re.MULTILINE
) is not None
def test_config_POST_rejects_non_allowlisted_key(client):
@@ -123,15 +135,19 @@ def test_config_POST_rejects_non_allowlisted_key(client):
def test_config_POST_invalid_value_leaves_file_unchanged(client):
"""quality_tolerance=1.5 (>1) fails RouterConfig validation -> 422, no write."""
tc, config_yaml = client
before = config_yaml.read_text()
before_base = config_yaml.read_text()
local_yaml = config_yaml.with_name("config.local.yaml")
before_local = local_yaml.read_text() if local_yaml.exists() else ""
resp = tc.post(
"/admin/api/config/objective.quality_tolerance", json={"value": 1.5}
)
assert resp.status_code == 422
after = config_yaml.read_text()
assert after == before
assert config_yaml.read_text() == before_base
assert (
local_yaml.read_text() if local_yaml.exists() else ""
) == before_local
def test_config_GET_includes_routing_default_profile(client):
@@ -141,12 +157,16 @@ def test_config_GET_includes_routing_default_profile(client):
assert resp.status_code == 200
body = resp.json()
assert "routing.default_profile" in body
assert body["routing.default_profile"] == "default"
assert body["routing.default_profile"] == {
"value": "default",
"source": "base",
}
def test_config_POST_default_profile_persists_valid_builtin(client):
"""POST routing.default_profile=batch persists and keeps comments."""
tc, config_yaml = client
local_yaml = config_yaml.with_name("config.local.yaml")
_insert_sentinel(config_yaml)
resp = tc.post(
@@ -158,17 +178,23 @@ def test_config_POST_default_profile_persists_valid_builtin(client):
assert body["value"] == "batch"
assert "restart is required" in body["message"]
text = config_yaml.read_text()
assert _SENTINEL in text
# Base comments preserved; value lives in overlay.
base_text = config_yaml.read_text()
assert _SENTINEL in base_text
local_text = local_yaml.read_text()
assert re.search(
r'^\s*default_profile:\s*["\']?batch["\']?\s*$', text, re.MULTILINE
r'^\s*default_profile:\s*["\']?batch["\']?\s*$', local_text, re.MULTILINE
) is not None
def test_config_POST_default_profile_rejects_unknown_and_leaves_file_unchanged(client):
"""Unknown default_profile returns 422 and does not touch config.yaml."""
def test_config_POST_default_profile_rejects_unknown_and_leaves_overlay_unchanged(
client,
):
"""Unknown default_profile returns 422 and does not touch config.local.yaml."""
tc, config_yaml = client
before = config_yaml.read_text()
local_yaml = config_yaml.with_name("config.local.yaml")
before_base = config_yaml.read_text()
before_local = local_yaml.read_text() if local_yaml.exists() else ""
resp = tc.post(
"/admin/api/config/routing.default_profile",
@@ -179,25 +205,29 @@ def test_config_POST_default_profile_rejects_unknown_and_leaves_file_unchanged(c
assert "nosuchprofile" in detail
assert "default" in detail or "batch" in detail
after = config_yaml.read_text()
assert after == before
assert config_yaml.read_text() == before_base
assert (
local_yaml.read_text() if local_yaml.exists() else ""
) == before_local
def test_config_POST_creates_backup_before_write(client, tmp_path):
"""A successful write produces a ``config.yaml.bak.<ts>`` backup copy."""
def test_config_POST_creates_overlay_backup_before_write(client, tmp_path):
"""A successful write produces a ``config.local.yaml.bak.<ts>`` backup copy."""
tc, config_yaml = client
local_yaml = config_yaml.with_name("config.local.yaml")
_insert_sentinel(config_yaml)
resp = tc.post("/admin/api/config/circuit_breaker.enabled", json={"value": True})
assert resp.status_code == 200
backups = sorted(tmp_path.glob("config/config.yaml.bak.*"))
backups = sorted(tmp_path.glob("config/config.local.yaml.bak.*"))
assert len(backups) == 1
# The backup captured the sentinel comment and the PRE-write value.
# The backup captured the pre-write empty overlay (just the header).
backup_text = backups[0].read_text()
assert _SENTINEL in backup_text
backup_yaml = yaml.safe_load(backup_text)
assert backup_yaml["circuit_breaker"]["enabled"] is True
assert backup_text.strip() != ""
# The current overlay carries the new value.
local_text = local_yaml.read_text()
assert yaml.safe_load(local_text)["circuit_breaker"]["enabled"] is True
def test_config_concurrent_writes_are_atomic_no_zero_byte_backups(tmp_path):
@@ -246,8 +276,10 @@ def test_config_concurrent_writes_are_atomic_no_zero_byte_backups(tmp_path):
assert b.read_text().strip() != ""
def test_profile_create_preserves_comments_and_user_local_energy(client):
"""A profile CRUD write keeps sentinel comments and the user's local_energy block."""
def test_profile_create_preserves_comments_and_does_not_create_overlay(
client, tmp_path
):
"""A profile CRUD write keeps sentinel comments and never touches config.local.yaml."""
tc, config_yaml = client
_insert_sentinel(config_yaml)
@@ -262,12 +294,8 @@ def test_profile_create_preserves_comments_and_user_local_energy(client):
text = config_yaml.read_text()
assert _SENTINEL in text
assert re.search(
r"^\s*tariff_usd_per_kwh:\s*0\.159\s*$", text, re.MULTILINE
) is not None
assert re.search(
r"^\s*enabled:\s*true\s*$", text, re.MULTILINE
) is not None
assert "local_energy:" in text
assert not (tmp_path / "config" / "config.local.yaml").exists()
def test_profile_create_creates_backup_before_write(client, tmp_path):
@@ -370,3 +398,130 @@ def test_config_concurrent_block_writes_are_atomic_no_zero_byte_backups(tmp_path
assert b.stat().st_size > 0, f"zero-byte backup: {b}"
assert b.read_text().strip() != ""
# ---------------------------------------------------------------------------
# Overlay survivability / provenance edge-case tests
# ---------------------------------------------------------------------------
_OVERLAY_HEADER_PREFIX = "# Machine-local overlay"
def test_overlay_survives_git_restore_of_base(tmp_path):
"""A machine-local overlay (config.local.yaml) survives a git checkout of
the base file — it is gitignored and therefore unaffected by ``git
checkout -- config/config.yaml``."""
repo = tmp_path / "repo"
repo.mkdir()
(repo / "config").mkdir()
base_yaml = repo / "config" / "config.yaml"
shutil.copyfile(ROOT / "config" / "config.yaml", base_yaml)
local_yaml = repo / "config" / "config.local.yaml"
local_yaml.write_text(
f"{_OVERLAY_HEADER_PREFIX} — gitignored, never committed.\n"
"logging:\n level: warning\n"
)
subprocess.run(
["git", "init", "-q", str(repo)],
check=True,
capture_output=True,
)
subprocess.run(
["git", "-C", str(repo), "config", "user.email", "test@test.com"],
check=True,
capture_output=True,
)
subprocess.run(
["git", "-C", str(repo), "config", "user.name", "test"],
check=True,
capture_output=True,
)
(repo / ".gitignore").write_text("config.local.yaml\n")
subprocess.run(
["git", "-C", str(repo), "add", "config/config.yaml", ".gitignore"],
check=True,
capture_output=True,
)
subprocess.run(
["git", "-C", str(repo), "commit", "-m", "init", "-q"],
check=True,
capture_output=True,
)
text = base_yaml.read_text()
base_yaml.write_text(text.replace('level: "info"', 'level: "debug"'))
subprocess.run(
["git", "-C", str(repo), "checkout", "--", "config/config.yaml"],
check=True,
capture_output=True,
)
assert local_yaml.exists()
overlay_text = local_yaml.read_text()
assert "level: warning" in overlay_text
restored = base_yaml.read_text()
assert 'level: "info"' in restored or "level: info" in restored
def test_config_GET_shows_overlay_source_when_overlay_exists(client):
"""When ``config.local.yaml`` overrides one allowlisted key, GET
/admin/api/config reports ``source: "overlay"`` for that key and
``source: "base"`` for the rest."""
tc, config_yaml = client
local_yaml = config_yaml.with_name("config.local.yaml")
local_yaml.write_text(
f"{_OVERLAY_HEADER_PREFIX} — gitignored, never committed.\n"
"logging:\n level: warning\n"
)
resp = tc.get("/admin/api/config")
assert resp.status_code == 200
body = resp.json()
assert body["logging.level"]["source"] == "overlay"
assert body["logging.level"]["value"] == "warning"
assert body["objective.quality_tolerance"]["source"] == "base"
def test_first_write_creates_overlay_with_header(client):
"""POST to an allowlisted key when no overlay exists creates
``config.local.yaml`` starting with the expected comment header."""
tc, config_yaml = client
local_yaml = config_yaml.with_name("config.local.yaml")
assert not local_yaml.exists(), "overlay must not exist before the write"
resp = tc.post(
"/admin/api/config/circuit_breaker.enabled", json={"value": True}
)
assert resp.status_code == 200
assert local_yaml.exists()
text = local_yaml.read_text()
assert text.startswith("# Machine-local overlay")
assert "# Written by the admin portal" in text
local = yaml.safe_load(text)
assert local["circuit_breaker"]["enabled"] is True
def test_second_write_keeps_header(client):
"""A second admin write preserves the machine-local header comment."""
tc, config_yaml = client
local_yaml = config_yaml.with_name("config.local.yaml")
resp = tc.post("/admin/api/config/logging.level", json={"value": "warning"})
assert resp.status_code == 200
resp = tc.post("/admin/api/config/circuit_breaker.enabled", json={"value": True})
assert resp.status_code == 200
text = local_yaml.read_text()
assert text.startswith("# Machine-local overlay")
# The header must appear exactly once.
assert text.count("# Machine-local overlay") == 1
local = yaml.safe_load(text)
assert local["logging"]["level"] == "warning"
assert local["circuit_breaker"]["enabled"] is True

View File

@@ -152,3 +152,16 @@ def test_quota_modal_has_not_configured_fallback():
index_path = ROOT / "admin" / "frontend" / "index.html"
html = index_path.read_text()
assert "not configured" in html
def test_admin_controls_persisted_config_handles_wrapped_value_shape(admin_client):
"""GET /admin/controls contains the updated Persisted Config hint and the
frontend unwrapping logic for the {value, source} response shape."""
resp = admin_client.get("/admin/controls")
assert resp.status_code == 200
assert resp.headers["content-type"].startswith("text/html")
text = resp.text
assert "config/config.local.yaml" in text
assert "entry.value" in text
assert "source === 'overlay'" in text
assert "data-orig" in text

View File

@@ -11,8 +11,9 @@ from pathlib import Path
import pytest
import yaml
from pydantic import ValidationError
from config import RouterConfig
from config import RouterConfig, _merge_overlay, load_config
ROOT = Path(__file__).resolve().parent.parent
@@ -179,3 +180,93 @@ def test_default_profile_accepts_builtin_name(raw, builtin_name):
cfg.setdefault("routing", {})["default_profile"] = builtin_name
loaded = RouterConfig(**cfg)
assert loaded.routing.default_profile == builtin_name
def _write_configs(tmp_path: Path, base: dict | None, overlay: dict | None):
"""Write *base* and *overlay* YAML files into *tmp_path* and return the base path."""
cfg_dir = tmp_path / "config"
cfg_dir.mkdir(exist_ok=True)
base_path = cfg_dir / "config.yaml"
if base is not None:
base_path.write_text(yaml.safe_dump(base))
if overlay is not None:
(cfg_dir / "config.local.yaml").write_text(yaml.safe_dump(overlay))
return base_path
def test_load_config_without_overlay_returns_base_only(tmp_path: Path, raw):
"""No overlay file ⇒ load_config returns the same as RouterConfig(**base_raw)."""
base_path = _write_configs(tmp_path, raw, overlay=None)
loaded = load_config(str(base_path))
assert loaded.objective.quality_tolerance == 0.1
assert loaded.routing.default_profile == "default"
assert loaded.routing.allowed_access_levels == ["public"]
assert loaded.local_energy.enabled is False
def test_load_config_overlay_deep_merge_preserves_sibling_keys(tmp_path: Path, raw):
"""Overlay sets local_energy.tariff_usd_per_kwh only; base keys remain intact."""
overlay = {"local_energy": {"tariff_usd_per_kwh": 0.999}}
base_path = _write_configs(tmp_path, copy.deepcopy(raw), overlay)
loaded = load_config(str(base_path))
assert loaded.local_energy.tariff_usd_per_kwh == 0.999
assert loaded.local_energy.enabled is False
assert loaded.local_energy.meter == "nvidia_smi"
assert loaded.objective.quality_tolerance == 0.1
def test_load_config_overlay_list_replaces_wholesale(tmp_path: Path, raw):
"""An overlay list replaces the base list — it does NOT append."""
overlay = {"routing": {"allowed_access_levels": ["public", "canary"]}}
base_path = _write_configs(tmp_path, copy.deepcopy(raw), overlay)
loaded = load_config(str(base_path))
assert loaded.routing.allowed_access_levels == ["public", "canary"]
assert "private" not in loaded.routing.allowed_access_levels
assert loaded.routing.default_profile == "default"
def test_load_config_overlay_unknown_key_raises(tmp_path: Path):
"""Overlay containing a top-level unknown key triggers Pydantic validation."""
base = {"objective": {"quality_tolerance": 0.1}}
overlay = {"objective": {"quality_tolerance": 0.1, "total_bullshit_key": 42}}
base_path = _write_configs(tmp_path, base, overlay)
with pytest.raises((ValueError, ValidationError)) as exc_info:
load_config(str(base_path))
assert "total_bullshit_key" in str(exc_info.value)
def test_merge_overlay_helper_list_replacement():
"""_merge_overlay replaces lists rather than appending."""
base = {"routing": {"allowed_access_levels": ["public"]}}
overlay = {"routing": {"allowed_access_levels": ["private"]}}
merged = _merge_overlay(base, overlay)
assert merged["routing"]["allowed_access_levels"] == ["private"]
assert len(merged["routing"]["allowed_access_levels"]) == 1
def test_overlay_enables_local_energy(tmp_path: Path, raw):
"""The overlay can flip local_energy.enabled without touching the tracked base file."""
overlay = {"local_energy": {"enabled": True, "tariff_usd_per_kwh": 0.12}}
base_path = _write_configs(tmp_path, copy.deepcopy(raw), overlay)
loaded = load_config(str(base_path))
assert loaded.local_energy.enabled is True
assert loaded.local_energy.tariff_usd_per_kwh == 0.12
def test_merge_overlay_helper_recursive_mapping_merge():
"""_merge_overlay recursively merges nested mappings, preserving sibling keys."""
base = {
"a": {"x": 1, "y": 2},
"b": "base",
}
overlay = {
"a": {"y": 99, "z": 3},
"c": "new",
}
merged = _merge_overlay(base, overlay)
assert merged["a"]["x"] == 1
assert merged["a"]["y"] == 99
assert merged["a"]["z"] == 3
assert merged["b"] == "base"
assert merged["c"] == "new"