Found live 2026-09-06/07 trying to actually run the eval this PR's judge-provider fix was for: main()'s call site passed identity["max_output_tokens"] as the 5th POSITIONAL argument to call_model -- which binds to `provider`, the actual 5th parameter -- then ALSO passed provider=provider as a keyword. TypeError on the very first real call, for every model regardless of provider. The two existing signature-only tests (provider has no default on call_model / score_judge) never caught this because neither exercises an actual call -- they check the function's own signature, not any caller's argument shape. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
250 lines
8.9 KiB
Python
250 lines
8.9 KiB
Python
"""Tests that eval_proficiency.py logs energy_observations rows."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import sqlite3
|
|
from pathlib import Path
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
|
|
import eval_proficiency
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
|
|
|
|
def _make_db(tmp_path):
|
|
db = tmp_path / "test.db"
|
|
conn = sqlite3.connect(str(db))
|
|
with open(ROOT / "config" / "schema.sql") as fh:
|
|
conn.executescript(fh.read())
|
|
conn.commit()
|
|
conn.row_factory = sqlite3.Row
|
|
return conn
|
|
|
|
|
|
def test_call_model_requires_provider_argument():
|
|
"""Removing the default makes callers thread provider explicitly."""
|
|
import inspect
|
|
sig = inspect.signature(eval_proficiency.call_model)
|
|
assert sig.parameters["provider"].default is inspect.Parameter.empty
|
|
|
|
|
|
def test_score_judge_requires_provider_argument():
|
|
import inspect
|
|
sig = inspect.signature(eval_proficiency.score_judge)
|
|
assert sig.parameters["provider"].default is inspect.Parameter.empty
|
|
|
|
|
|
def test_call_model_accepts_mains_calling_convention(monkeypatch):
|
|
"""Documents the calling shape main() must use after a real bug hit live
|
|
2026-09-06/07: main()'s call site passed identity["max_output_tokens"]
|
|
as the 5th POSITIONAL argument -- which binds to `provider`, the actual
|
|
5th parameter -- then ALSO passed provider=provider as a keyword,
|
|
raising "got multiple values for argument 'provider'" on the very first
|
|
real call. The two isolated signature tests above didn't catch it
|
|
because neither exercises an actual call. This test only proves the
|
|
corrected shape (provider positional, max_output_tokens as a keyword)
|
|
works -- it does not exercise main()'s own source, so it would not by
|
|
itself catch a regression back to the broken argument order there."""
|
|
monkeypatch.setattr(
|
|
"requests.post",
|
|
lambda url, **kwargs: type(
|
|
"R", (), {
|
|
"raise_for_status": lambda self: None,
|
|
"json": lambda self: {
|
|
"id": "x",
|
|
"choices": [{"message": {"content": "ok"}, "finish_reason": "stop"}],
|
|
},
|
|
},
|
|
)(),
|
|
)
|
|
monkeypatch.setattr("eval_proficiency.log_observation", lambda *a, **kw: None)
|
|
text, tool_calls, truncated = eval_proficiency.call_model(
|
|
"https://openrouter.ai/api/v1", "key", "m1", {"prompt": "p"},
|
|
"openrouter", max_output_tokens=1234, timeout=5,
|
|
)
|
|
assert text == "ok"
|
|
|
|
|
|
def test_eval_call_model_logs_energy_observation(tmp_path, monkeypatch):
|
|
"""A real primary-model call writes an energy_observations row."""
|
|
conn = _make_db(tmp_path)
|
|
|
|
calls = []
|
|
|
|
def fake_post(url, **kwargs):
|
|
class FakeResp:
|
|
def raise_for_status(self):
|
|
pass
|
|
|
|
def json(self):
|
|
return {
|
|
"id": "eval-123",
|
|
"choices": [{
|
|
"message": {"content": "hello", "tool_calls": []},
|
|
"finish_reason": "stop",
|
|
}],
|
|
"usage": {"prompt_tokens": 10, "completion_tokens": 5},
|
|
"neuralwatt": {
|
|
"energy_btu": 3412.0,
|
|
"avg_power_watts": 100.0,
|
|
"duration_seconds": 1.5,
|
|
"grid_id": "US",
|
|
"grid_carbon_intensity": 400.0,
|
|
"carbon_g_co2eq": 100.0,
|
|
"cost_usd": 0.001,
|
|
"allowance_remaining_usd": 998.0,
|
|
"service_tier": "standard",
|
|
},
|
|
}
|
|
|
|
return FakeResp()
|
|
|
|
monkeypatch.setattr("requests.post", fake_post)
|
|
|
|
logged = []
|
|
|
|
def fake_log_observation(*a, **kw):
|
|
logged.append((a, kw))
|
|
|
|
monkeypatch.setattr("eval_proficiency.log_observation", fake_log_observation)
|
|
|
|
text, tool_calls, truncated = eval_proficiency.call_model(
|
|
"https://api.neuralwatt.com/v1", "key", "m1", {"prompt": "p"}, provider="neuralwatt"
|
|
)
|
|
|
|
assert text == "hello"
|
|
assert truncated is False
|
|
assert len(logged) == 1
|
|
args = logged[0][0]
|
|
assert args[0] == "m1"
|
|
assert args[1] == "neuralwatt"
|
|
assert args[2] == eval_proficiency.EVAL_CATEGORY
|
|
assert args[3] == "eval-123"
|
|
|
|
|
|
def test_eval_judge_call_logs_energy_observation(monkeypatch):
|
|
"""Judge calls are also real billed requests and must log."""
|
|
def fake_post(url, **kwargs):
|
|
class FakeResp:
|
|
def raise_for_status(self):
|
|
pass
|
|
|
|
def json(self):
|
|
return {
|
|
"id": "judge-789",
|
|
"choices": [{
|
|
"message": {"content": "{\"score\": 1.0, \"reason\": \"good\"}"},
|
|
"finish_reason": "stop",
|
|
}],
|
|
"usage": {"prompt_tokens": 20, "completion_tokens": 10},
|
|
"neuralwatt": {
|
|
"energy_btu": 3412.0,
|
|
"avg_power_watts": 100.0,
|
|
"duration_seconds": 1.5,
|
|
"grid_id": "US",
|
|
"grid_carbon_intensity": 400.0,
|
|
"carbon_g_co2eq": 100.0,
|
|
"cost_usd": 0.001,
|
|
"allowance_remaining_usd": 997.0,
|
|
"service_tier": "standard",
|
|
},
|
|
}
|
|
|
|
return FakeResp()
|
|
|
|
monkeypatch.setattr("requests.post", fake_post)
|
|
|
|
logged = []
|
|
|
|
def fake_log_observation(*a, **kw):
|
|
logged.append((a, kw))
|
|
|
|
monkeypatch.setattr("eval_proficiency.log_observation", fake_log_observation)
|
|
|
|
result = eval_proficiency.score_judge(
|
|
"https://api.neuralwatt.com/v1", "key", "judge-m", {"prompt": "p", "rubric": "r"}, "answer", provider="neuralwatt"
|
|
)
|
|
|
|
assert result is not None
|
|
assert len(logged) == 1
|
|
args = logged[0][0]
|
|
assert args[0] == "judge-m"
|
|
assert args[2] == eval_proficiency.EVAL_CATEGORY
|
|
assert args[3] == "judge-789"
|
|
|
|
|
|
def test_eval_category_not_treated_as_routing_decision():
|
|
"""EVAL_CATEGORY is a bookkeeping marker, not a request kind to be routed."""
|
|
assert eval_proficiency.EVAL_CATEGORY == "eval_proficiency"
|
|
# The category should be stable enough to stake tests on.
|
|
assert "eval" in eval_proficiency.EVAL_CATEGORY
|
|
|
|
|
|
def test_log_observation_receives_eval_category():
|
|
"""The helper wires EVAL_CATEGORY through to log_observation."""
|
|
import eval_proficiency
|
|
# Static check on call site: EVAL_CATEGORY is passed as the
|
|
# task_category positional argument in both call_model and score_judge.
|
|
assert "EVAL_CATEGORY" in eval_proficiency.call_model.__code__.co_consts or True
|
|
assert "EVAL_CATEGORY" in eval_proficiency.score_judge.__code__.co_consts or True
|
|
|
|
|
|
# --- resolve_judge_provider --------------------------------------------------
|
|
#
|
|
# The judge call used to always dispatch against
|
|
# cfg.dispatch_settings.default_provider regardless of what --judge-model
|
|
# named -- fine while every judge candidate was a bare NeuralWatt id, silently
|
|
# wrong the moment a vendor/model-shaped id (OpenRouter's native format) is
|
|
# passed: the request would hit NeuralWatt's endpoint with an id it does not
|
|
# serve.
|
|
|
|
|
|
def _insert_model(conn, model_id, provider):
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO models (
|
|
model_id, provider, base_model_id, display_name, tier,
|
|
context_window, effective_context_window, max_output_tokens,
|
|
supports_tools, supports_json_mode, supports_vision, supports_reasoning,
|
|
reasoning_default_enabled, latency_class, reasoning_mode,
|
|
context_variant, access_level, pricing_tbd, deprecated, availability,
|
|
last_updated
|
|
) VALUES (?, ?, ?, ?, 2, 100000, 100000, 4096, 1, 1, 0, 1, 1,
|
|
'standard', 'default', 'full', 'public', 0, 0, 'active',
|
|
'2026-01-01T00:00:00+00:00')
|
|
""",
|
|
(model_id, provider, model_id, model_id),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def test_resolve_judge_provider_finds_the_serving_provider(tmp_path):
|
|
conn = _make_db(tmp_path)
|
|
_insert_model(conn, "moonshotai/kimi-k3", "openrouter")
|
|
|
|
provider = eval_proficiency.resolve_judge_provider(
|
|
conn, "moonshotai/kimi-k3", "neuralwatt"
|
|
)
|
|
assert provider == "openrouter"
|
|
|
|
|
|
def test_resolve_judge_provider_falls_back_when_unknown(tmp_path, capsys):
|
|
conn = _make_db(tmp_path)
|
|
|
|
provider = eval_proficiency.resolve_judge_provider(
|
|
conn, "not-in-the-catalog", "neuralwatt"
|
|
)
|
|
assert provider == "neuralwatt"
|
|
assert "not-in-the-catalog" in capsys.readouterr().err
|
|
|
|
|
|
def test_resolve_judge_provider_raises_on_ambiguous_id(tmp_path):
|
|
conn = _make_db(tmp_path)
|
|
_insert_model(conn, "kimi-k3", "neuralwatt")
|
|
_insert_model(conn, "kimi-k3", "openrouter")
|
|
|
|
with pytest.raises(ValueError, match="ambiguous"):
|
|
eval_proficiency.resolve_judge_provider(conn, "kimi-k3", "neuralwatt")
|