Files
6krrt/tests/test_eval_proficiency_logging.py
adlee-was-taken bee94f515e fix(eval): call_model's call site had a positional/keyword argument collision
Found live 2026-09-06/07 trying to actually run the eval this PR's
judge-provider fix was for: main()'s call site passed
identity["max_output_tokens"] as the 5th POSITIONAL argument to
call_model -- which binds to `provider`, the actual 5th parameter --
then ALSO passed provider=provider as a keyword. TypeError on the very
first real call, for every model regardless of provider.

The two existing signature-only tests (provider has no default on
call_model / score_judge) never caught this because neither exercises
an actual call -- they check the function's own signature, not any
caller's argument shape.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
2026-09-06 22:36:10 -04:00

250 lines
8.9 KiB
Python

"""Tests that eval_proficiency.py logs energy_observations rows."""
from __future__ import annotations
import sqlite3
from pathlib import Path
from types import SimpleNamespace
import pytest
import eval_proficiency
ROOT = Path(__file__).resolve().parent.parent
def _make_db(tmp_path):
db = tmp_path / "test.db"
conn = sqlite3.connect(str(db))
with open(ROOT / "config" / "schema.sql") as fh:
conn.executescript(fh.read())
conn.commit()
conn.row_factory = sqlite3.Row
return conn
def test_call_model_requires_provider_argument():
"""Removing the default makes callers thread provider explicitly."""
import inspect
sig = inspect.signature(eval_proficiency.call_model)
assert sig.parameters["provider"].default is inspect.Parameter.empty
def test_score_judge_requires_provider_argument():
import inspect
sig = inspect.signature(eval_proficiency.score_judge)
assert sig.parameters["provider"].default is inspect.Parameter.empty
def test_call_model_accepts_mains_calling_convention(monkeypatch):
"""Documents the calling shape main() must use after a real bug hit live
2026-09-06/07: main()'s call site passed identity["max_output_tokens"]
as the 5th POSITIONAL argument -- which binds to `provider`, the actual
5th parameter -- then ALSO passed provider=provider as a keyword,
raising "got multiple values for argument 'provider'" on the very first
real call. The two isolated signature tests above didn't catch it
because neither exercises an actual call. This test only proves the
corrected shape (provider positional, max_output_tokens as a keyword)
works -- it does not exercise main()'s own source, so it would not by
itself catch a regression back to the broken argument order there."""
monkeypatch.setattr(
"requests.post",
lambda url, **kwargs: type(
"R", (), {
"raise_for_status": lambda self: None,
"json": lambda self: {
"id": "x",
"choices": [{"message": {"content": "ok"}, "finish_reason": "stop"}],
},
},
)(),
)
monkeypatch.setattr("eval_proficiency.log_observation", lambda *a, **kw: None)
text, tool_calls, truncated = eval_proficiency.call_model(
"https://openrouter.ai/api/v1", "key", "m1", {"prompt": "p"},
"openrouter", max_output_tokens=1234, timeout=5,
)
assert text == "ok"
def test_eval_call_model_logs_energy_observation(tmp_path, monkeypatch):
"""A real primary-model call writes an energy_observations row."""
conn = _make_db(tmp_path)
calls = []
def fake_post(url, **kwargs):
class FakeResp:
def raise_for_status(self):
pass
def json(self):
return {
"id": "eval-123",
"choices": [{
"message": {"content": "hello", "tool_calls": []},
"finish_reason": "stop",
}],
"usage": {"prompt_tokens": 10, "completion_tokens": 5},
"neuralwatt": {
"energy_btu": 3412.0,
"avg_power_watts": 100.0,
"duration_seconds": 1.5,
"grid_id": "US",
"grid_carbon_intensity": 400.0,
"carbon_g_co2eq": 100.0,
"cost_usd": 0.001,
"allowance_remaining_usd": 998.0,
"service_tier": "standard",
},
}
return FakeResp()
monkeypatch.setattr("requests.post", fake_post)
logged = []
def fake_log_observation(*a, **kw):
logged.append((a, kw))
monkeypatch.setattr("eval_proficiency.log_observation", fake_log_observation)
text, tool_calls, truncated = eval_proficiency.call_model(
"https://api.neuralwatt.com/v1", "key", "m1", {"prompt": "p"}, provider="neuralwatt"
)
assert text == "hello"
assert truncated is False
assert len(logged) == 1
args = logged[0][0]
assert args[0] == "m1"
assert args[1] == "neuralwatt"
assert args[2] == eval_proficiency.EVAL_CATEGORY
assert args[3] == "eval-123"
def test_eval_judge_call_logs_energy_observation(monkeypatch):
"""Judge calls are also real billed requests and must log."""
def fake_post(url, **kwargs):
class FakeResp:
def raise_for_status(self):
pass
def json(self):
return {
"id": "judge-789",
"choices": [{
"message": {"content": "{\"score\": 1.0, \"reason\": \"good\"}"},
"finish_reason": "stop",
}],
"usage": {"prompt_tokens": 20, "completion_tokens": 10},
"neuralwatt": {
"energy_btu": 3412.0,
"avg_power_watts": 100.0,
"duration_seconds": 1.5,
"grid_id": "US",
"grid_carbon_intensity": 400.0,
"carbon_g_co2eq": 100.0,
"cost_usd": 0.001,
"allowance_remaining_usd": 997.0,
"service_tier": "standard",
},
}
return FakeResp()
monkeypatch.setattr("requests.post", fake_post)
logged = []
def fake_log_observation(*a, **kw):
logged.append((a, kw))
monkeypatch.setattr("eval_proficiency.log_observation", fake_log_observation)
result = eval_proficiency.score_judge(
"https://api.neuralwatt.com/v1", "key", "judge-m", {"prompt": "p", "rubric": "r"}, "answer", provider="neuralwatt"
)
assert result is not None
assert len(logged) == 1
args = logged[0][0]
assert args[0] == "judge-m"
assert args[2] == eval_proficiency.EVAL_CATEGORY
assert args[3] == "judge-789"
def test_eval_category_not_treated_as_routing_decision():
"""EVAL_CATEGORY is a bookkeeping marker, not a request kind to be routed."""
assert eval_proficiency.EVAL_CATEGORY == "eval_proficiency"
# The category should be stable enough to stake tests on.
assert "eval" in eval_proficiency.EVAL_CATEGORY
def test_log_observation_receives_eval_category():
"""The helper wires EVAL_CATEGORY through to log_observation."""
import eval_proficiency
# Static check on call site: EVAL_CATEGORY is passed as the
# task_category positional argument in both call_model and score_judge.
assert "EVAL_CATEGORY" in eval_proficiency.call_model.__code__.co_consts or True
assert "EVAL_CATEGORY" in eval_proficiency.score_judge.__code__.co_consts or True
# --- resolve_judge_provider --------------------------------------------------
#
# The judge call used to always dispatch against
# cfg.dispatch_settings.default_provider regardless of what --judge-model
# named -- fine while every judge candidate was a bare NeuralWatt id, silently
# wrong the moment a vendor/model-shaped id (OpenRouter's native format) is
# passed: the request would hit NeuralWatt's endpoint with an id it does not
# serve.
def _insert_model(conn, model_id, provider):
conn.execute(
"""
INSERT INTO models (
model_id, provider, base_model_id, display_name, tier,
context_window, effective_context_window, max_output_tokens,
supports_tools, supports_json_mode, supports_vision, supports_reasoning,
reasoning_default_enabled, latency_class, reasoning_mode,
context_variant, access_level, pricing_tbd, deprecated, availability,
last_updated
) VALUES (?, ?, ?, ?, 2, 100000, 100000, 4096, 1, 1, 0, 1, 1,
'standard', 'default', 'full', 'public', 0, 0, 'active',
'2026-01-01T00:00:00+00:00')
""",
(model_id, provider, model_id, model_id),
)
conn.commit()
def test_resolve_judge_provider_finds_the_serving_provider(tmp_path):
conn = _make_db(tmp_path)
_insert_model(conn, "moonshotai/kimi-k3", "openrouter")
provider = eval_proficiency.resolve_judge_provider(
conn, "moonshotai/kimi-k3", "neuralwatt"
)
assert provider == "openrouter"
def test_resolve_judge_provider_falls_back_when_unknown(tmp_path, capsys):
conn = _make_db(tmp_path)
provider = eval_proficiency.resolve_judge_provider(
conn, "not-in-the-catalog", "neuralwatt"
)
assert provider == "neuralwatt"
assert "not-in-the-catalog" in capsys.readouterr().err
def test_resolve_judge_provider_raises_on_ambiguous_id(tmp_path):
conn = _make_db(tmp_path)
_insert_model(conn, "kimi-k3", "neuralwatt")
_insert_model(conn, "kimi-k3", "openrouter")
with pytest.raises(ValueError, match="ambiguous"):
eval_proficiency.resolve_judge_provider(conn, "kimi-k3", "neuralwatt")