feat: report quota balance, burn rate, and runway #30

Merged
alee merged 6 commits from feat/quota-balance-and-burn-rate into main 2026-09-05 04:47:16 +00:00
12 changed files with 881 additions and 40 deletions

View File

@@ -113,6 +113,14 @@ the one case that was checked live it agrees with the measurement in direction
and magnitude (7.8x predicted vs 5.0x measured). It is also free, needs no
sweep, and refreshes whenever the poller runs.
`objective.plan_kwh_per_period` is a planning figure only: per-request traffic
is **never** refused for exceeding it — it **gates nothing**. Overage is billed
against the account's credit balance (`allowance_remaining_usd` from the
provider). `/metrics` now reports balance, estimated burn rate, and projected
runway (hours remaining) derived from the provider-reported
`allowance_remaining_usd` over a configurable window, using a configurable
minimum segment length and sample count to avoid wild extrapolations.
Three signals said `deepseek-v4-flash` — catalog token price (7.8x cheaper),
NeuralWatt's own published per-request energy (~10x lower), and a live 70k
measurement (5.0x cheaper). Only the 400-token benchmark disagreed. Trust the

View File

@@ -460,7 +460,7 @@ header.navbar{padding-top:2px!important;padding-bottom:2px!important}
<div class="modal-dialog modal-dialog-centered" role="document">
<div class="modal-content">
<div class="modal-header">
<h5 class="modal-title">Quota &mdash; 30 day window</h5>
<h5 class="modal-title">Quota &mdash; Balance &amp; Runway</h5>
<button type="button" class="btn-close" data-bs-dismiss="modal" aria-label="Close"></button>
</div>
<div class="modal-body" id="quota-modal-body">
@@ -657,17 +657,57 @@ function renderQuotaModal(quota) {
el.innerHTML = `<div class="empty">No quota plan configured</div>`;
return;
}
// ── balance-led rendering ───────────────────────────────────────────
if (quota.balance_usd != null && typeof quota.balance_usd === 'number') {
const balanceStr = '$' + quota.balance_usd.toFixed(2);
const burn = quota.burn_rate_usd_per_hour ?? null;
const proj = quota.projected_hours_remaining ?? null;
const low = !!quota.runway_low_warning;
const hours = proj != null ? '~' + Math.round(proj) + 'h runway' : 'runway unknown';
const hourColor = low ? 'var(--tblr-danger)' : !burn ? 'var(--tblr-warning)' : 'var(--tblr-success)';
const burnLabel = burn != null ? '$' + (Math.round(burn * 100) / 100).toFixed(2) + '/hr' : (quota.runway_note ? '' : 'n/a');
// Build detail rows
const detailRows = [];
if (proj != null) detailRows.push('Projected');
if (burn != null && !quota.runway_note) detailRows.push('Burn rate');
detailRows.push('Plan');
detailRows.push('Metered');
detailRows.push('Reset');
let rowIdx = -1;
const runways = ['Low Runway', 'Burn N/A'];
const meteredLabel = quota.reset_date
? `Metered (period since ${escapeHtml(quota.reset_date)})`
: 'Metered (30d window)';
el.innerHTML = `
<div class="mb-3">
<div style="font-size:1.8rem;font-weight:700;color:#4ade80">${escapeHtml(balanceStr)}</div>
<div style="font-size:1rem;color:${hourColor};margin-top:2px">${hours}</div>
${low ? '<div class="badge bg-warning text-warning mt-1">Low Runway</div>' : ''}
</div>
<div class="quota-details">
${burn != null && !quota.runway_note ? '<span class="quota-label">Burn rate</span><span class="quota-value">~' + escapeHtml(burnLabel) + '/hr</span>' : ''}
${quota.runway_note ? '<span class="quota-label">Burn</span><span class="quota-value">' + escapeHtml(quota.runway_note) + '</span>' : ''}
${proj != null ? '<span class="quota-label">Projected</span><span class="quota-value">~' + Math.round(proj) + 'h</span>' : ''}
<span class="quota-label">Plan</span><span class="quota-value">${quota.plan_kwh} kWh</span>
<span class="quota-label">${escapeHtml(meteredLabel)}</span><span class="quota-value">${quota.metered_kwh_period != null ? quota.metered_kwh_period + ' kWh' : quota.metered_kwh_30d + ' kWh'}</span>
<span class="quota-label">Fraction</span><span class="quota-value">${(quota.metered_fraction_of_plan * 100).toFixed(1)}%</span>
<span class="quota-label">Reset</span><span class="quota-value">${quota.next_reset_date || 'not configured'}</span>
</div>`;
return;
}
// ── fallback: percentage-only rendering ─────────────────────────────
const pct = Math.min(quota.metered_fraction_of_plan * 100, 100);
const color = pct > 90 ? 'var(--tblr-danger)' : pct > 75 ? 'var(--tblr-warning)' : 'var(--tblr-success)';
const displayPct = (quota.metered_fraction_of_plan * 100).toFixed(1);
const calls = Number(quota.metered_calls_30d || 0).toLocaleString();
// One % label. When the fill is wide enough (>22%) the % sits at the fill's
// right end, INSIDE the capsule, white on the colored fill. Otherwise it sits
// just right of the fill in the same row, in the threshold color.
const pctInside = pct > 22;
const pctStyle = pctInside
? `left:calc(${pct}% - 30px);color:#fff`
: `left:calc(${pct}% + 8px);color:${color}`;
el.innerHTML = `
<div>
<div class="d-flex justify-content-between align-items-center mb-2">
@@ -697,6 +737,64 @@ function renderQuotaChip(quota) {
return;
}
chip.hidden = false;
// ── balance-led display ------------------------------------------------
if (quota.balance_usd != null) {
const burn = quota.burn_rate_usd_per_hour ?? null;
const proj = quota.projected_hours_remaining ?? null;
const low = !!quota.runway_low_warning;
const hours = proj != null ? Math.round(proj) : null;
let chipStyle = {};
if (low) {
// Red: runway critically low
chipStyle = {
'--chip-accent': 'var(--tblr-danger)',
background: 'linear-gradient(135deg, rgba(239,68,68,.12), rgba(59,130,246,.08))',
borderColor: 'rgba(239,68,68,.35)',
};
} else if (!burn) {
// Amber: burn unknown, not alarming
chipStyle = {
'--chip-accent': 'var(--tblr-warning)',
background: 'linear-gradient(135deg, rgba(245,158,11,.12), rgba(59,130,246,.08))',
borderColor: 'rgba(245,158,11,.38)',
};
} else if (chipStyle._unset !== true) {
// Green: healthy runway
chipStyle = {
'--chip-accent': 'var(--tblr-success)',
};
}
const balanceStr = typeof quota.balance_usd === 'number'
? '$' + quota.balance_usd.toFixed(2)
: '$' + String(quota.balance_usd);
if (hours != null) {
text.textContent = balanceStr + ' · ~' + hours + 'h';
} else if (burn != null) {
text.textContent = balanceStr;
} else {
text.textContent = balanceStr;
}
// Apply chip accent colours (border + text tint)
for (const [k, v] of Object.entries(chipStyle)) {
if (k !== '_unset') chip.style.setProperty(k, v);
}
if (chipStyle['--chip-accent']) {
text.style.color = chipStyle['--chip-accent'];
}
// Build operator-readable tooltip
let title = 'Account balance';
if (burn != null) title = '$' + quota.balance_usd.toFixed(2) + ' · ~' + (Math.round(burn * 100) / 100) + '/hr';
if (hours != null) title += ' · ~' + hours + 'h runway';
text.setAttribute('title', title);
fill.style.width = `${Math.min(quota.metered_fraction_of_plan * 100, 100)}%`;
return;
}
// ── fallback: percentage only ------------------------------------------------
const pct = Math.min(quota.metered_fraction_of_plan * 100, 100);
fill.style.width = `${pct}%`;
text.textContent = `${(quota.metered_fraction_of_plan * 100).toFixed(1)}%`;

View File

@@ -47,19 +47,36 @@ objective:
# Per-request ceiling on measured ENERGY, in kWh. null disables it.
#
# Denominated in kWh rather than dollars because the plan is a subscription
# with a 6.25 kWh quota, not pay-per-request. Dollars accrue; a quota is a
# wall you hit mid-task. So this is the cost mandate stated as a guarantee.
# For scale: the reference task runs ~5e-06 kWh on the cheapest model and
# ~2.2e-04 on the most expensive.
# Denominated in kWh rather than dollars. For scale: the reference task runs
# ~5e-06 kWh on the cheapest model and ~2.2e-04 on the most expensive.
# Overage is billed against the account's credit balance;
# plan_kwh_per_period gates nothing.
max_energy_per_request:
# The subscription's kWh allowance per billing period, for reporting burn in
# /health. Set to match your plan; null disables the report. NeuralWatt also
# returns allowance_remaining_usd per request, which is logged, but that is a
# dollar figure while the plan is denominated in energy.
# /health. This is a planning figure only: per-request traffic is never
# refused for exceeding plan_kwh_per_period — it gates nothing. Set to match
# your plan; null disables the report. NeuralWatt also returns
# allowance_remaining_usd per request, which is logged for /metrics, but that
# is a dollar figure while the plan is denominated in energy.
plan_kwh_per_period: 6.25
# Hours of recent balance history used to estimate the burn rate. The most
# recent monotonically-decreasing segment of allowance_remaining_usd values
# (segments split at each balance increase) is examined over this window.
quota_burn_window_hours: 24
# Projected hours of runway below which /metrics and dashboards emit a
# low-warning boolean; only fires when a burn rate estimate exists and the
# projected remainder is positive but short.
quota_runway_warning_hours: 6
# A burn estimate needs at least this many balance samples in the most recent
# monotonically-decreasing segment (post-top-up resets the segment); below
# which the estimate is None with an explanatory runway note.
quota_burn_min_segment_samples: 3
# ...and the segment must span at least this many hours, otherwise the
# estimate is None with an explanatory note (never a wild extrapolation).
quota_burn_min_segment_hours: 0.5
# The day-of-month your NeuralWatt subscription billing cycle resets. Set
# this to YOUR real billing day so the admin quota modal shows a genuine
# next-reset date instead of a misleading rolling-window start. null disables

View File

@@ -128,7 +128,15 @@ it appears only in `metrics.py` reporting and the admin allowlist.
- The admin quota chip leads with balance and runway.
- Existing `/metrics` `quota` keys still present.
- Full suite green with `local_energy.enabled` both true and false.
- The user's `config/config.yaml` tariff lines remain uncommitted and verbatim.
- The user's `config/config.local.yaml` is byte-identical after the run
(`local_energy.enabled: true`, `tariff_usd_per_kwh: 0.159`). **Corrected
2026-09-04:** this criterion previously said the tariff lines in
`config/config.yaml` must stay uncommitted. That is stale — PR #26 moved
deployment values into the gitignored overlay, so `config/config.yaml` is
now clean and tracked. Note the consequence for this plan specifically:
§3 edits comments *in* `config/config.yaml`, which is now an ordinary
tracked edit to commit deliberately, not a file to tiptoe around. The file
to protect is the overlay, and it is not in git history at all.
## The pattern worth naming

View File

@@ -50,6 +50,10 @@ class Objective(StrictModel):
max_energy_per_request: Optional[float] = None
plan_kwh_per_period: Optional[float] = None
billing_reset_day: Optional[int] = None
quota_burn_window_hours: Optional[int] = None
quota_runway_warning_hours: Optional[int] = None
quota_burn_min_segment_samples: Optional[int] = None
quota_burn_min_segment_hours: Optional[float] = None
@field_validator("quality_tolerance")
@classmethod
@@ -75,6 +79,42 @@ class Objective(StrictModel):
)
return v
@field_validator("quota_burn_window_hours")
@classmethod
def burn_window_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.quota_burn_window_hours must be > 0"
)
return v
@field_validator("quota_runway_warning_hours")
@classmethod
def runway_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.quota_runway_warning_hours must be > 0"
)
return v
@field_validator("quota_burn_min_segment_samples")
@classmethod
def segment_samples_positive(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v <= 0:
raise ValueError(
"objective.quota_burn_min_segment_samples must be > 0"
)
return v
@field_validator("quota_burn_min_segment_hours")
@classmethod
def segment_hours_positive(cls, v: Optional[float]) -> Optional[float]:
if v is not None and v <= 0:
raise ValueError(
"objective.quota_burn_min_segment_hours must be > 0"
)
return v
class ContextOverride(StrictModel):
"""Per-model context handling, for a row whose real limits are known.

View File

@@ -9,8 +9,10 @@ optionally a ``RouterConfig`` instance; none rely on module-level globals.
Functions
---------
quota_burn — kWh metered in the last 30 d, against the plan allowance;
also reports the 30-day reset date
quota_burn — kWh metered in the last 30 d and in the current billing
period, against the plan allowance; also reports the account
credit balance, burn rate and runway from
allowance_remaining_usd
scoring_coverage — which scoring axes actually have data
recent_decisions — last N rows from the route_decisions observability table
per_model — per-model aggregates over energy_observations (last 30 d)
@@ -68,6 +70,165 @@ def _next_reset_date(billing_reset_day: int, today: Optional[date] = None) -> st
return date(next_year, next_month, billing_reset_day).isoformat()
def _billing_period_start(billing_reset_day: int, today: Optional[date] = None) -> str:
"""Return the most recent billing-period start as an ISO date string.
When *today*'s day-of-month is on or after *billing_reset_day*, returns
this month's reset day; otherwise returns the previous month's reset day
(handling January → December rollover).
"""
if today is None:
today = datetime.now(timezone.utc).date()
if today.day >= billing_reset_day:
return date(today.year, today.month, billing_reset_day).isoformat()
if today.month == 1:
prev_year = today.year - 1
prev_month = 12
else:
prev_year = today.year
prev_month = today.month - 1
return date(prev_year, prev_month, billing_reset_day).isoformat()
def quota_balance_and_burn(conn: sqlite3.Connection, cfg: Any) -> dict:
"""Latest account balance and its burn/runway from ``allowance_remaining_usd``.
Pure read of ``energy_observations`` — no dispatcher import. The burn
rate is intentionally conservative: only the most recent
monotonically-decreasing balance segment is used, and two guards stop a
fresh credit top-up from producing a wild extrapolation.
"""
# Read optional knobs from config; treat None as unset and use code defaults.
# Test configs use SimpleNamespace without these attributes, so getattr
# must have a fallback and then a second default when the attr is None.
burn_window_hours = getattr(cfg.objective, "quota_burn_window_hours", None)
if burn_window_hours is None:
burn_window_hours = 24
warning_hours = getattr(cfg.objective, "quota_runway_warning_hours", None)
if warning_hours is None:
warning_hours = 6
min_samples = getattr(cfg.objective, "quota_burn_min_segment_samples", None)
if min_samples is None:
min_samples = 3
min_hours = getattr(cfg.objective, "quota_burn_min_segment_hours", None)
if min_hours is None:
min_hours = 0.5
result: dict[str, Any] = {
"balance_usd": None,
"balance_at": None,
"burn_window_hours": burn_window_hours,
"burn_rate_usd_per_hour": None,
"projected_hours_remaining": None,
"runway_low_warning": False,
"runway_note": None,
}
# 1. Latest non-NULL allowance across ALL rows (no time-window filter).
balance_row = conn.execute(
"""
SELECT observed_at, allowance_remaining_usd
FROM energy_observations
WHERE allowance_remaining_usd IS NOT NULL
ORDER BY observed_at DESC, id DESC
LIMIT 1
"""
).fetchone()
if balance_row is not None:
result["balance_usd"] = float(balance_row["allowance_remaining_usd"])
result["balance_at"] = balance_row["observed_at"]
# 2. In-window rows for burn estimation.
rows = conn.execute(
"""
SELECT observed_at, allowance_remaining_usd
FROM energy_observations
WHERE allowance_remaining_usd IS NOT NULL
AND julianday(observed_at) > julianday('now', '-' || ? || ' hours')
ORDER BY observed_at ASC, id ASC
""",
(str(burn_window_hours),),
).fetchall()
if not rows:
result["runway_note"] = (
"burn estimate unavailable: no decreasing balance samples in the current window"
)
return result
# 3. Split into monotonically-decreasing segments at every balance INCREASE.
# A top-up (credit jump) starts a new segment; only the latest survives.
segments: list[list[tuple[datetime, float]]] = [[]]
for row in rows:
raw_ts = row["observed_at"]
try:
ts = datetime.fromisoformat(raw_ts)
except ValueError:
# Defensive: malformed timestamp would otherwise break /metrics.
continue
value = float(row["allowance_remaining_usd"])
current = segments[-1]
if not current:
current.append((ts, value))
elif value > current[-1][1]:
segments.append([(ts, value)])
else:
current.append((ts, value))
latest_segment = segments[-1]
if not latest_segment:
result["runway_note"] = (
"burn estimate unavailable: no decreasing balance samples in the current window"
)
return result
# 4. Guarded burn-rate computation.
if len(latest_segment) < min_samples:
result["runway_note"] = (
f"burn estimate unavailable: segment after last balance increase has only "
f"{len(latest_segment)} sample(s), need {min_samples}"
)
return result
first_ts, first_balance = latest_segment[0]
last_ts, last_balance = latest_segment[-1]
elapsed_hours = (last_ts - first_ts).total_seconds() / 3600.0
if elapsed_hours < min_hours:
minutes = elapsed_hours * 60
if minutes < 60:
duration_str = f"{minutes:.0f} minutes"
else:
duration_str = f"{elapsed_hours:.1f} hours"
result["runway_note"] = (
f"burn estimate unavailable: segment after last balance increase spans only "
f"{duration_str}, need at least {min_hours} h"
)
return result
total_decrease = first_balance - last_balance
if total_decrease <= 0.0:
# Treat a flat or increasing-only segment like the no-burn case.
result["runway_note"] = (
"burn estimate unavailable: no decreasing balance samples in the current window"
)
return result
burn_rate = round(total_decrease / elapsed_hours, 6)
result["burn_rate_usd_per_hour"] = burn_rate
# 5. Projected runway and warning.
balance = result["balance_usd"]
if balance is not None and burn_rate > 0.0:
projected = balance / burn_rate
result["projected_hours_remaining"] = projected
result["runway_low_warning"] = projected < warning_hours
return result
def quota_burn(
conn: sqlite3.Connection,
cfg: Any,
@@ -80,13 +241,21 @@ def quota_burn(
dispatcher.
Returns a dict with ``plan_kwh``, ``metered_kwh_30d``,
``metered_fraction_of_plan``, ``metered_calls_30d``, ``reset_date`` (the
ISO date of today minus 30 days, the rolling-window start) and ``note``.
``metered_kwh_period``, ``metered_fraction_of_plan``,
``metered_calls_30d``, ``reset_date`` (the billing-period start),
``window_start_30d`` (the rolling 30-day window start), ``note`` and the
balance/burn/runway fields from ``quota_balance_and_burn``.
When ``cfg.objective.billing_reset_day`` is set, also returns
``next_reset_date`` — the upcoming billing-period reset day.
"""
# This gate removes the report only; it is intentionally not used to refuse
# or alter request dispatch — routing decisions remain independent of quota.
if not cfg.objective.plan_kwh_per_period:
return None
plan = cfg.objective.plan_kwh_per_period
window_start_30d = (datetime.now(timezone.utc).date() - timedelta(days=30)).isoformat()
row = conn.execute(
"""
SELECT COALESCE(SUM(energy_kwh), 0) kwh, COUNT(*) n
@@ -94,17 +263,43 @@ def quota_burn(
WHERE julianday(observed_at) > julianday('now', '-30 days')
"""
).fetchone()
metered_kwh_30d = round(float(row["kwh"]), 5)
metered_calls_30d = row["n"]
reset_day = getattr(cfg.objective, "billing_reset_day", None)
if reset_day is not None:
period_start = _billing_period_start(reset_day)
period_row = conn.execute(
"""
SELECT COALESCE(SUM(energy_kwh), 0) kwh
FROM energy_observations
WHERE julianday(observed_at) >= julianday(?)
""",
(period_start,),
).fetchone()
metered_kwh_period = round(float(period_row["kwh"]), 5)
metered_fraction_of_plan = round(metered_kwh_period / plan, 4)
reset_date = period_start
else:
metered_kwh_period = None
metered_fraction_of_plan = round(metered_kwh_30d / plan, 4)
reset_date = None
plan = cfg.objective.plan_kwh_per_period
result = {
"plan_kwh": plan,
"metered_kwh_30d": round(float(row["kwh"]), 5),
"metered_fraction_of_plan": round(float(row["kwh"]) / plan, 4),
"metered_calls_30d": row["n"],
"reset_date": (datetime.now(timezone.utc).date() - timedelta(days=30)).isoformat(),
"metered_kwh_30d": metered_kwh_30d,
"metered_kwh_period": metered_kwh_period,
"metered_fraction_of_plan": metered_fraction_of_plan,
"metered_calls_30d": metered_calls_30d,
"reset_date": reset_date,
"window_start_30d": window_start_30d,
"note": "router-metered only; traffic bypassing the router is not counted",
}
reset_day = getattr(cfg.objective, "billing_reset_day", None)
# Merge the balance/burn/runway block; quota_balance_and_burn also supplies
# the burn_window_hours default.
result.update(quota_balance_and_burn(conn, cfg))
if reset_day is not None:
result["next_reset_date"] = _next_reset_date(reset_day)
return result
@@ -152,8 +347,8 @@ def scoring_coverage(
if quota and quota["metered_fraction_of_plan"] > 0.8:
warnings.append(
f"metered usage is {quota['metered_fraction_of_plan']*100:.0f}% of the "
f"{quota['plan_kwh']} kWh plan allowance. A quota is a wall, not a bill — "
"requests fail rather than costing more."
f"{quota['plan_kwh']} kWh plan allowance. Overage is billed against the "
"account's credit balance; plan_kwh_per_period gates nothing."
)
if missing_energy:
warnings.append(
@@ -544,10 +739,12 @@ def local_energy_summary(conn: sqlite3.Connection, cfg) -> Optional[dict]:
"metered_cost_usd_30d": round(float(total["cost"]), 5),
"calls_30d": total["n"],
"by_type": by_type,
"reset_date": (datetime.now(timezone.utc).date() - timedelta(days=30)).isoformat(),
"reset_date": None,
"window_start_30d": (datetime.now(timezone.utc).date() - timedelta(days=30)).isoformat(),
}
reset_day = getattr(cfg.objective, "billing_reset_day", None)
if reset_day is not None:
result["reset_date"] = _billing_period_start(reset_day)
result["next_reset_date"] = _next_reset_date(reset_day)
return result
@@ -639,10 +836,12 @@ def pinch_summary(conn: sqlite3.Connection, cfg: Any) -> Optional[dict]:
"total_tokens_saved": total_tokens_saved,
"median_tokens_saved": median_tokens_saved,
"dollars_saved_usd_30d": round(dollars_saved_usd_30d, 6),
"reset_date": None,
"window_start_30d": (datetime.now(timezone.utc).date() - timedelta(days=30)).isoformat(),
}
reset_day = getattr(cfg.objective, "billing_reset_day", None)
if reset_day is not None:
result["reset_date"] = (datetime.now(timezone.utc).date() - timedelta(days=30)).isoformat()
result["reset_date"] = _billing_period_start(reset_day)
result["next_reset_date"] = _next_reset_date(reset_day)
return result

View File

@@ -447,12 +447,26 @@ class DashboardApp(App):
except Exception: # noqa: BLE001 — a Static still takes focus harmlessly
pass
def _format_quota_legend(self, plan, metered, frac, calls, reset_date, next_reset_date=None, flex_default=None) -> str:
def _format_quota_legend(
self,
plan,
metered,
frac,
calls,
reset_date,
next_reset_date=None,
flex_default=None,
window_start_30d=None,
) -> str:
"""Build the quota legend string, suppressing ``None`` values.
``metered``/``frac``/``calls`` come from ``rows.get(...)`` and may be
``None`` (unmetered plan); render those as ``n/a`` rather than the
Python literal so the legend never contains the substring ``"None"``.
``reset_date`` is the billing-period start (shown as "period since …").
``window_start_30d`` is the rolling 30-day window start (shown as
"window start …").
"""
metered_fmt = "n/a" if metered is None else metered
frac_fmt = "n/a" if frac is None else frac
@@ -466,7 +480,12 @@ class DashboardApp(App):
legend += f" · [rgb(128,128,128)]flex default {flex_default}[/rgb(128,128,128)]"
if reset_date:
legend += (
f" · [rgb(128,128,128)]window start {reset_date}"
f" · [rgb(128,128,128)]period since {reset_date}"
f"[/rgb(128,128,128)]"
)
if window_start_30d:
legend += (
f" · [rgb(128,128,128)]window start {window_start_30d}"
f"[/rgb(128,128,128)]"
)
if next_reset_date:
@@ -495,10 +514,14 @@ class DashboardApp(App):
bar.progress = float(metered or 0)
bar.display = True
reset_date = rows.get("reset_date")
window_start = rows.get("window_start_30d")
flex_default = model.get("flex_default")
next_reset_date = rows.get("next_reset_date")
legend.update(
self._format_quota_legend(plan, metered, frac, calls, reset_date, next_reset_date, flex_default)
self._format_quota_legend(
plan, metered, frac, calls, reset_date,
next_reset_date, flex_default, window_start,
)
)
else:
bar.display = False

View File

@@ -55,6 +55,12 @@ def build_model(data: dict) -> dict:
{"label": "calls", "value": quota.get("metered_calls_30d")},
{"label": "reset_date", "value": quota.get("reset_date")},
{"label": "next_reset_date", "value": quota.get("next_reset_date")},
{"label": "balance_usd", "value": quota.get("balance_usd")},
{"label": "burn_rate_usd_per_hour", "value": quota.get("burn_rate_usd_per_hour")},
{"label": "projected_hours_remaining", "value": quota.get("projected_hours_remaining")},
{"label": "runway_low_warning", "value": quota.get("runway_low_warning")},
{"label": "runway_note", "value": quota.get("runway_note")},
{"label": "window_start_30d", "value": quota.get("window_start_30d")},
]
else:
quota_rows = []

View File

@@ -154,6 +154,13 @@ def test_quota_modal_has_not_configured_fallback():
assert "not configured" in html
def test_admin_quota_chip_references_balance_and_runway():
"""The index page references the new balance / runway quota fields."""
index_path = ROOT / "admin" / "frontend" / "index.html"
html = index_path.read_text()
assert "balance_usd" in html and "runway_low_warning" in html
def test_admin_controls_persisted_config_handles_wrapped_value_shape(admin_client):
"""GET /admin/controls contains the updated Persisted Config hint and the
frontend unwrapping logic for the {value, source} response shape."""

View File

@@ -19,7 +19,7 @@ move, and that ``import metrics`` alone succeeds (no circular import).
from __future__ import annotations
import sqlite3
from datetime import datetime, timedelta, timezone
from datetime import date, datetime, timedelta, timezone
from pathlib import Path
from types import SimpleNamespace
@@ -31,6 +31,7 @@ from config import load_config
from metrics import (
_next_reset_date,
context_ceilings,
local_energy_summary,
per_model,
pinch_summary,
quota_burn,
@@ -198,10 +199,13 @@ def test_quota_burn_aggregates_last_30_days(tmp_path):
assert result["metered_calls_30d"] == 2
assert result["plan_kwh"] == 6.25
assert result["metered_fraction_of_plan"] == pytest.approx(0.25 / 6.25)
expected_reset = (datetime.now(timezone.utc).date() - timedelta(days=30)).isoformat()
assert result["reset_date"] == expected_reset
assert "T" not in result["reset_date"]
assert not result["reset_date"].endswith(("Z", "+00:00"))
# With no billing_reset_day configured, reset_date is None (billing period
# not defined) and the rolling 30-day window start lives in window_start_30d.
assert result["reset_date"] is None
expected_window = (datetime.now(timezone.utc).date() - timedelta(days=30)).isoformat()
assert result["window_start_30d"] == expected_window
assert "T" not in result["window_start_30d"]
assert not result["window_start_30d"].endswith(("Z", "+00:00"))
def test_quota_burn_empty_db(tmp_path):
@@ -215,6 +219,352 @@ def test_quota_burn_empty_db(tmp_path):
assert result["metered_calls_30d"] == 0
# --- quota_burn balance / burn / runway tests --------------------------------
def test_quota_burn_with_top_up_ignores_credit_jump(tmp_path):
"""A credit top-up splits the window; burn uses only the latest segment.
Series before top-up: 3.00 → 2.00 → 1.00. Then balance jumps to 10.00 and
resumes decreasing 10.00 → 9.25 → 8.50. Burn must come from the post-top-up
segment (1.50 USD / 1.5 h = 1.0 USD/h), not from the overall MAX-MIN which
would invent a phantom burn of ~7.00 USD.
"""
cfg = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6)
)
conn = _make_db(tmp_path)
now = _now()
rows = [
("m", 3.00, now - timedelta(hours=5)),
("m", 2.00, now - timedelta(hours=4)),
("m", 1.00, now - timedelta(hours=3)),
("m", 10.00, now - timedelta(hours=2)),
("m", 9.25, now - timedelta(minutes=90)),
("m", 8.50, now - timedelta(minutes=30)),
]
for model_id, balance, ts in rows:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
(model_id, balance, ts.isoformat()),
)
conn.commit()
result = quota_burn(conn, cfg)
assert result["balance_usd"] == pytest.approx(8.50)
assert result["balance_at"] == rows[-1][2].isoformat()
assert result["burn_window_hours"] == 24
assert result["burn_rate_usd_per_hour"] == pytest.approx(1.0)
assert result["projected_hours_remaining"] == pytest.approx(8.5)
assert result["runway_low_warning"] is False
def test_quota_burn_with_top_up_immediately_before_window_end_returns_none(tmp_path):
"""Top-up with only a short final tail → burn is unavailable, never wild."""
cfg = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6)
)
conn = _make_db(tmp_path)
now = _now()
rows = [
("m", 5.00, now - timedelta(hours=4)),
("m", 4.00, now - timedelta(hours=3)),
("m", 3.00, now - timedelta(hours=2)),
("m", 20.00, now - timedelta(minutes=10)),
("m", 19.90, now - timedelta(minutes=6)),
("m", 19.80, now - timedelta(minutes=2)),
]
for model_id, balance, ts in rows:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
(model_id, balance, ts.isoformat()),
)
conn.commit()
result = quota_burn(conn, cfg)
assert result["burn_rate_usd_per_hour"] is None
assert result["runway_note"] is not None
assert "burn estimate unavailable" in result["runway_note"]
assert result["projected_hours_remaining"] is None
assert result["runway_low_warning"] is False
def test_quota_burn_all_null_allowance_degrades_gracefully(tmp_path):
"""All allowance_remaining_usd NULL → no balance, no burn, no false alarm."""
cfg = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6)
)
conn = _make_db(tmp_path)
now = _now()
rows = [
("m", 2.0, now - timedelta(hours=2)),
("m", 3.0, now - timedelta(hours=1)),
]
for model_id, kwh, ts in rows:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', ?, NULL, ?)",
(model_id, kwh, ts.isoformat()),
)
conn.commit()
result = quota_burn(conn, cfg)
assert result is not None
assert result["balance_usd"] is None
assert result["balance_at"] is None
assert result["burn_rate_usd_per_hour"] is None
assert result["projected_hours_remaining"] is None
assert result["runway_low_warning"] is False
assert result["runway_note"] is not None
assert result["metered_kwh_30d"] == pytest.approx(5.0)
def test_quota_burn_runway_warning_when_below_threshold(tmp_path):
"""Low balance and positive burn below warning threshold triggers warning."""
cfg = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6)
)
conn = _make_db(tmp_path)
now = _now()
rows = [
("m", 1.60, now - timedelta(hours=3)),
("m", 1.20, now - timedelta(hours=2)),
("m", 0.80, now - timedelta(hours=1)),
("m", 0.60, now - timedelta(minutes=30)),
]
for model_id, balance, ts in rows:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
(model_id, balance, ts.isoformat()),
)
conn.commit()
result = quota_burn(conn, cfg)
assert result["burn_rate_usd_per_hour"] == pytest.approx(0.4)
assert result["projected_hours_remaining"] == pytest.approx(1.5)
assert result["runway_low_warning"] is True
assert result["burn_window_hours"] == 24
def test_quota_burn_billing_period_kwh_excludes_rolling_window(tmp_path):
"""metered_kwh_period uses billing-period window; metered_kwh_30d uses 30 d.
Rows are pinned on either side of each boundary so the expected values can
be computed from first principles without relying on a helper.
"""
cfg = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6)
)
conn = _make_db(tmp_path)
now = _now()
today = now.date()
reset_day = 6
if today.day >= reset_day:
period_start = datetime(today.year, today.month, reset_day, tzinfo=timezone.utc)
else:
if today.month == 1:
period_start = datetime(today.year - 1, 12, reset_day, tzinfo=timezone.utc)
else:
period_start = datetime(today.year, today.month - 1, reset_day, tzinfo=timezone.utc)
recent = now - timedelta(days=2)
rolling_edge = now - timedelta(days=30) + timedelta(days=1)
period_edge = period_start.replace(hour=1)
seeded = [
("recent", recent, 2.0),
("rolling_edge", rolling_edge, 3.0),
("period_edge", period_edge, 5.0),
]
for model_id, ts, kwh in seeded:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', ?, 10.0, ?)",
(model_id, kwh, ts.isoformat()),
)
conn.commit()
expected_rolling = sum(
kwh for _, ts, kwh in seeded
if ts > now - timedelta(days=30)
)
expected_period = sum(
kwh for _, ts, kwh in seeded
if ts >= period_start
)
result = quota_burn(conn, cfg)
assert "metered_kwh_period" in result
assert "window_start_30d" in result
assert result["metered_kwh_period"] == pytest.approx(expected_period)
assert result["metered_kwh_30d"] == pytest.approx(expected_rolling)
assert result["metered_fraction_of_plan"] == pytest.approx(expected_period / 6.25)
def test_quota_burn_includes_balance_and_runway_keys(tmp_path):
"""quota_burn returns the full balance/burn/runway key set, and the
standalone quota_balance_and_burn helper matches it.
"""
cfg = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6)
)
conn = _make_db(tmp_path)
now = _now()
rows = [
("m", 1.60, now - timedelta(hours=3)),
("m", 1.20, now - timedelta(hours=2)),
("m", 0.80, now - timedelta(hours=1)),
("m", 0.60, now - timedelta(minutes=30)),
]
for model_id, balance, ts in rows:
conn.execute(
"INSERT INTO energy_observations "
"(model_id, provider, energy_kwh, allowance_remaining_usd, observed_at) "
"VALUES (?, 'neuralwatt', 0.001, ?, ?)",
(model_id, balance, ts.isoformat()),
)
conn.commit()
result = quota_burn(conn, cfg)
for key in (
"plan_kwh",
"metered_kwh_30d",
"metered_kwh_period",
"metered_fraction_of_plan",
"metered_calls_30d",
"note",
"reset_date",
"window_start_30d",
"balance_usd",
"balance_at",
"burn_window_hours",
"burn_rate_usd_per_hour",
"projected_hours_remaining",
"runway_low_warning",
"runway_note",
"next_reset_date",
):
assert key in result, f"missing key {key!r}"
from metrics import quota_balance_and_burn
balance_only = quota_balance_and_burn(conn, cfg)
for key in (
"balance_usd",
"balance_at",
"burn_window_hours",
"burn_rate_usd_per_hour",
"projected_hours_remaining",
"runway_low_warning",
"runway_note",
):
assert key in balance_only, f"missing key {key!r}"
assert balance_only[key] == result[key]
def test_quota_burn_reset_date_is_billing_period_start(tmp_path):
"""reset_date is the billing-period start; window_start_30d is rolling 30 d."""
cfg = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25, billing_reset_day=6)
)
conn = _make_db(tmp_path)
now = datetime.now(timezone.utc)
today = now.date()
reset_day = 6
if today.day >= reset_day:
expected_period_start = date(today.year, today.month, reset_day).isoformat()
else:
if today.month == 1:
expected_period_start = date(today.year - 1, 12, reset_day).isoformat()
else:
expected_period_start = date(today.year, today.month - 1, reset_day).isoformat()
expected_rolling_start = (today - timedelta(days=30)).isoformat()
result = quota_burn(conn, cfg)
assert result["reset_date"] == expected_period_start
assert result["window_start_30d"] == expected_rolling_start
cfg_unconfigured = SimpleNamespace(
objective=SimpleNamespace(plan_kwh_per_period=6.25)
)
result_unconfigured = quota_burn(conn, cfg_unconfigured)
assert "reset_date" in result_unconfigured
assert result_unconfigured["reset_date"] is None
assert result_unconfigured["window_start_30d"] == expected_rolling_start
def test_local_energy_summary_reset_date_is_billing_period_start(tmp_path):
"""local_energy_summary mirrors the same reset_date/window_start split."""
cfg = SimpleNamespace(
local_energy=SimpleNamespace(enabled=True),
objective=SimpleNamespace(billing_reset_day=6),
)
conn = _make_db(tmp_path)
now = datetime.now(timezone.utc)
today = now.date()
reset_day = 6
if today.day >= reset_day:
expected_period_start = date(today.year, today.month, reset_day).isoformat()
else:
if today.month == 1:
expected_period_start = date(today.year - 1, 12, reset_day).isoformat()
else:
expected_period_start = date(today.year, today.month - 1, reset_day).isoformat()
expected_rolling_start = (today - timedelta(days=30)).isoformat()
result = local_energy_summary(conn, cfg)
assert result["reset_date"] == expected_period_start
assert result["window_start_30d"] == expected_rolling_start
cfg_unconfigured = SimpleNamespace(
local_energy=SimpleNamespace(enabled=True),
objective=SimpleNamespace(),
)
result_unconfigured = local_energy_summary(conn, cfg_unconfigured)
assert "reset_date" in result_unconfigured
assert result_unconfigured["reset_date"] is None
assert result_unconfigured["window_start_30d"] == expected_rolling_start
def test_pinch_summary_reset_date_is_billing_period_start(tmp_path):
"""pinch_summary mirrors the same reset_date/window_start split."""
cfg = SimpleNamespace(
pinch=SimpleNamespace(enabled=True),
objective=SimpleNamespace(assumed_cache_rate=0.917, billing_reset_day=6),
)
conn = _make_db(tmp_path)
now = datetime.now(timezone.utc)
today = now.date()
reset_day = 6
if today.day >= reset_day:
expected_period_start = date(today.year, today.month, reset_day).isoformat()
else:
if today.month == 1:
expected_period_start = date(today.year - 1, 12, reset_day).isoformat()
else:
expected_period_start = date(today.year, today.month - 1, reset_day).isoformat()
expected_rolling_start = (today - timedelta(days=30)).isoformat()
result = pinch_summary(conn, cfg)
assert result["reset_date"] == expected_period_start
assert result["window_start_30d"] == expected_rolling_start
cfg_unconfigured = SimpleNamespace(
pinch=SimpleNamespace(enabled=True),
objective=SimpleNamespace(assumed_cache_rate=0.917),
)
result_unconfigured = pinch_summary(conn, cfg_unconfigured)
assert "reset_date" in result_unconfigured
assert result_unconfigured["reset_date"] is None
assert result_unconfigured["window_start_30d"] == expected_rolling_start
# --- _next_reset_date helper --------------------------------------------------

View File

@@ -226,6 +226,38 @@ def test_metrics_contains_no_session_dir(seeded_client):
assert "session_dir" not in body
def test_metrics_quota_carries_balance_and_runway_keys(seeded_client):
"""GET /metrics quota section includes the new balance/burn/runway keys
alongside the existing kWh/keys. With no allowance seeded, kWh fields are
present and balance/burn values are None or empty, but keys must exist.
"""
resp = seeded_client.get("/metrics")
assert resp.status_code == 200
data = resp.json()
assert data["quota"] is not None
quota = data["quota"]
legacy_keys = (
"plan_kwh",
"metered_kwh_30d",
"metered_fraction_of_plan",
"metered_calls_30d",
"note",
)
new_keys = (
"balance_usd",
"balance_at",
"burn_window_hours",
"burn_rate_usd_per_hour",
"projected_hours_remaining",
"runway_low_warning",
"runway_note",
"window_start_30d",
"metered_kwh_period",
)
for key in legacy_keys + new_keys:
assert key in quota, f"missing quota key {key!r}"
def test_metrics_local_energy_omitted_when_disabled(seeded_client):
"""With metering off the section is omitted — independent of deployment config."""
resp = seeded_client.get("/metrics")

View File

@@ -33,6 +33,15 @@ def _fixture() -> dict:
"metered_calls_30d": 18,
"reset_date": "2026-07-26",
"note": "router-metered only",
"balance_usd": 8.50,
"balance_at": "2026-08-23T09:58:00+00:00",
"burn_window_hours": 24,
"burn_rate_usd_per_hour": 1.0,
"projected_hours_remaining": 8.5,
"runway_low_warning": False,
"runway_note": None,
"window_start_30d": "2026-07-24",
"metered_kwh_period": 0.5,
},
"coverage": {
"routable_models": 13,
@@ -45,6 +54,15 @@ def _fixture() -> dict:
"metered_calls_30d": 18,
"reset_date": "2026-07-26",
"note": "router-metered only",
"balance_usd": 8.50,
"balance_at": "2026-08-23T09:58:00+00:00",
"burn_window_hours": 24,
"burn_rate_usd_per_hour": 1.0,
"projected_hours_remaining": 8.5,
"runway_low_warning": False,
"runway_note": None,
"window_start_30d": "2026-07-24",
"metered_kwh_period": 0.5,
},
"warnings": [
"3/13 routable models have no reference-workload observations",
@@ -164,6 +182,22 @@ def test_build_model_quota_panel_includes_reset_date():
rows = m["quota"]
by_label = {r["label"]: r["value"] for r in rows}
assert by_label["reset_date"] == "2026-07-26"
assert by_label["window_start_30d"] == "2026-07-24"
def test_build_model_quota_panel_includes_balance_and_runway_keys():
"""The new quota payload keys must reach the TUI data model so the
quota panel can surface balance, burn rate, and runway alongside the
existing plan/metered/reset rows."""
m = build_model(_fixture())
rows = m["quota"]
by_label = {r["label"]: r["value"] for r in rows}
assert by_label["balance_usd"] == 8.50
assert by_label["burn_rate_usd_per_hour"] == 1.0
assert by_label["projected_hours_remaining"] == 8.5
assert by_label["runway_low_warning"] is False
assert by_label["window_start_30d"] == "2026-07-24"
assert by_label["reset_date"] == "2026-07-26"
def test_build_model_per_model_lists_seeded_models():
@@ -361,7 +395,8 @@ def test_app_run_test_quota_panel_progress_bar_and_legend():
assert "6.25" in text
assert "1.25" in text
assert "calls 18" in text
assert "window start 2026-07-26" in text
assert "period since 2026-07-26" in text
assert "window start 2026-07-24" in text
_run_app(app, _assert)
@@ -375,25 +410,43 @@ def test_format_quota_legend_omits_none_when_unmetered():
"""
app = tui.DashboardApp(fetcher=_StubFetcher())
legend = app._format_quota_legend(
plan=6.25, metered=None, frac=None, calls=None, reset_date="2026-07-26"
plan=6.25, metered=None, frac=None, calls=None,
reset_date="2026-07-26", window_start_30d="2026-07-24",
)
assert isinstance(legend, str)
assert "None" not in legend
assert "n/a" in legend
assert "window start 2026-07-26" in legend
assert "period since 2026-07-26" in legend
assert "window start 2026-07-24" in legend
def test_format_quota_legend_happy_path_contains_numbers():
"""A fully-populated legend carries the metered / plan / calls values."""
app = tui.DashboardApp(fetcher=_StubFetcher())
legend = app._format_quota_legend(
plan=6.25, metered=1.25, frac=0.2, calls=18, reset_date="2026-07-26"
plan=6.25, metered=1.25, frac=0.2, calls=18,
reset_date="2026-07-26", window_start_30d="2026-07-24",
)
assert "6.25" in legend
assert "1.25" in legend
assert "calls 18" in legend
assert "20%" in legend
assert "window start 2026-07-26" in legend
assert "period since 2026-07-26" in legend
assert "window start 2026-07-24" in legend
assert "None" not in legend
def test_format_quota_legend_distinguishes_period_from_window():
"""The legend labels the two distinct dates: reset_date (billing period
start) and window_start_30d (rolling 30-day window start). They are
different concepts, so the labels must be visible side by side."""
app = tui.DashboardApp(fetcher=_StubFetcher())
legend = app._format_quota_legend(
plan=6.25, metered=1.25, frac=0.2, calls=18,
reset_date="2026-07-26", window_start_30d="2026-07-24",
)
assert "period since 2026-07-26" in legend
assert "window start 2026-07-24" in legend
assert "None" not in legend