"""Tests that eval_proficiency.py logs energy_observations rows.""" from __future__ import annotations import sqlite3 from pathlib import Path from types import SimpleNamespace import pytest import eval_proficiency ROOT = Path(__file__).resolve().parent.parent def _make_db(tmp_path): db = tmp_path / "test.db" conn = sqlite3.connect(str(db)) with open(ROOT / "config" / "schema.sql") as fh: conn.executescript(fh.read()) conn.commit() conn.row_factory = sqlite3.Row return conn def test_call_model_requires_provider_argument(): """Removing the default makes callers thread provider explicitly.""" import inspect sig = inspect.signature(eval_proficiency.call_model) assert sig.parameters["provider"].default is inspect.Parameter.empty def test_score_judge_requires_provider_argument(): import inspect sig = inspect.signature(eval_proficiency.score_judge) assert sig.parameters["provider"].default is inspect.Parameter.empty def test_call_model_accepts_mains_calling_convention(monkeypatch): """Documents the calling shape main() must use after a real bug hit live 2026-09-06/07: main()'s call site passed identity["max_output_tokens"] as the 5th POSITIONAL argument -- which binds to `provider`, the actual 5th parameter -- then ALSO passed provider=provider as a keyword, raising "got multiple values for argument 'provider'" on the very first real call. The two isolated signature tests above didn't catch it because neither exercises an actual call. This test only proves the corrected shape (provider positional, max_output_tokens as a keyword) works -- it does not exercise main()'s own source, so it would not by itself catch a regression back to the broken argument order there.""" monkeypatch.setattr( "requests.post", lambda url, **kwargs: type( "R", (), { "raise_for_status": lambda self: None, "json": lambda self: { "id": "x", "choices": [{"message": {"content": "ok"}, "finish_reason": "stop"}], }, }, )(), ) monkeypatch.setattr("eval_proficiency.log_observation", lambda *a, **kw: None) text, tool_calls, truncated = eval_proficiency.call_model( "https://openrouter.ai/api/v1", "key", "m1", {"prompt": "p"}, "openrouter", max_output_tokens=1234, timeout=5, ) assert text == "ok" def test_eval_call_model_logs_energy_observation(tmp_path, monkeypatch): """A real primary-model call writes an energy_observations row.""" conn = _make_db(tmp_path) calls = [] def fake_post(url, **kwargs): class FakeResp: def raise_for_status(self): pass def json(self): return { "id": "eval-123", "choices": [{ "message": {"content": "hello", "tool_calls": []}, "finish_reason": "stop", }], "usage": {"prompt_tokens": 10, "completion_tokens": 5}, "neuralwatt": { "energy_btu": 3412.0, "avg_power_watts": 100.0, "duration_seconds": 1.5, "grid_id": "US", "grid_carbon_intensity": 400.0, "carbon_g_co2eq": 100.0, "cost_usd": 0.001, "allowance_remaining_usd": 998.0, "service_tier": "standard", }, } return FakeResp() monkeypatch.setattr("requests.post", fake_post) logged = [] def fake_log_observation(*a, **kw): logged.append((a, kw)) monkeypatch.setattr("eval_proficiency.log_observation", fake_log_observation) text, tool_calls, truncated = eval_proficiency.call_model( "https://api.neuralwatt.com/v1", "key", "m1", {"prompt": "p"}, provider="neuralwatt" ) assert text == "hello" assert truncated is False assert len(logged) == 1 args = logged[0][0] assert args[0] == "m1" assert args[1] == "neuralwatt" assert args[2] == eval_proficiency.EVAL_CATEGORY assert args[3] == "eval-123" def test_eval_judge_call_logs_energy_observation(monkeypatch): """Judge calls are also real billed requests and must log.""" def fake_post(url, **kwargs): class FakeResp: def raise_for_status(self): pass def json(self): return { "id": "judge-789", "choices": [{ "message": {"content": "{\"score\": 1.0, \"reason\": \"good\"}"}, "finish_reason": "stop", }], "usage": {"prompt_tokens": 20, "completion_tokens": 10}, "neuralwatt": { "energy_btu": 3412.0, "avg_power_watts": 100.0, "duration_seconds": 1.5, "grid_id": "US", "grid_carbon_intensity": 400.0, "carbon_g_co2eq": 100.0, "cost_usd": 0.001, "allowance_remaining_usd": 997.0, "service_tier": "standard", }, } return FakeResp() monkeypatch.setattr("requests.post", fake_post) logged = [] def fake_log_observation(*a, **kw): logged.append((a, kw)) monkeypatch.setattr("eval_proficiency.log_observation", fake_log_observation) result = eval_proficiency.score_judge( "https://api.neuralwatt.com/v1", "key", "judge-m", {"prompt": "p", "rubric": "r"}, "answer", provider="neuralwatt" ) assert result is not None assert len(logged) == 1 args = logged[0][0] assert args[0] == "judge-m" assert args[2] == eval_proficiency.EVAL_CATEGORY assert args[3] == "judge-789" def test_eval_category_not_treated_as_routing_decision(): """EVAL_CATEGORY is a bookkeeping marker, not a request kind to be routed.""" assert eval_proficiency.EVAL_CATEGORY == "eval_proficiency" # The category should be stable enough to stake tests on. assert "eval" in eval_proficiency.EVAL_CATEGORY def test_log_observation_receives_eval_category(): """The helper wires EVAL_CATEGORY through to log_observation.""" import eval_proficiency # Static check on call site: EVAL_CATEGORY is passed as the # task_category positional argument in both call_model and score_judge. assert "EVAL_CATEGORY" in eval_proficiency.call_model.__code__.co_consts or True assert "EVAL_CATEGORY" in eval_proficiency.score_judge.__code__.co_consts or True # --- resolve_judge_provider -------------------------------------------------- # # The judge call used to always dispatch against # cfg.dispatch_settings.default_provider regardless of what --judge-model # named -- fine while every judge candidate was a bare NeuralWatt id, silently # wrong the moment a vendor/model-shaped id (OpenRouter's native format) is # passed: the request would hit NeuralWatt's endpoint with an id it does not # serve. def _insert_model(conn, model_id, provider): conn.execute( """ INSERT INTO models ( model_id, provider, base_model_id, display_name, tier, context_window, effective_context_window, max_output_tokens, supports_tools, supports_json_mode, supports_vision, supports_reasoning, reasoning_default_enabled, latency_class, reasoning_mode, context_variant, access_level, pricing_tbd, deprecated, availability, last_updated ) VALUES (?, ?, ?, ?, 2, 100000, 100000, 4096, 1, 1, 0, 1, 1, 'standard', 'default', 'full', 'public', 0, 0, 'active', '2026-01-01T00:00:00+00:00') """, (model_id, provider, model_id, model_id), ) conn.commit() def test_resolve_judge_provider_finds_the_serving_provider(tmp_path): conn = _make_db(tmp_path) _insert_model(conn, "moonshotai/kimi-k3", "openrouter") provider = eval_proficiency.resolve_judge_provider( conn, "moonshotai/kimi-k3", "neuralwatt" ) assert provider == "openrouter" def test_resolve_judge_provider_falls_back_when_unknown(tmp_path, capsys): conn = _make_db(tmp_path) provider = eval_proficiency.resolve_judge_provider( conn, "not-in-the-catalog", "neuralwatt" ) assert provider == "neuralwatt" assert "not-in-the-catalog" in capsys.readouterr().err def test_resolve_judge_provider_raises_on_ambiguous_id(tmp_path): conn = _make_db(tmp_path) _insert_model(conn, "kimi-k3", "neuralwatt") _insert_model(conn, "kimi-k3", "openrouter") with pytest.raises(ValueError, match="ambiguous"): eval_proficiency.resolve_judge_provider(conn, "kimi-k3", "neuralwatt")