"""Tests for the baseline_report.py read-only routing comparator. Seeds a throwaway temp SQLite DB with a small catalog, some proficiency rows, and a handful of route_decisions, then asserts on the aggregate analysis and the reconstructed baselines. No network, no provider calls, no dispatcher import — the report is pure read-over-seeded-tables. """ from __future__ import annotations import csv import io import sqlite3 from pathlib import Path import pytest import baseline_report from config import load_config ROOT = Path(__file__).resolve().parent.parent SCHEMA_SQL = (ROOT / "config" / "schema.sql").read_text() CFG = load_config(str(ROOT / "config" / "config.yaml")) @pytest.fixture() def conn(tmp_path: Path) -> sqlite3.Connection: c = sqlite3.connect(str(tmp_path / "test.db")) c.row_factory = sqlite3.Row c.executescript(SCHEMA_SQL) return c def _seed_model( conn: sqlite3.Connection, model_id: str, *, cost: float = 1.0, tier: int = 1, context: int = 262128, vision: int = 1, json_mode: int = 1, latency: str = "standard", access: str = "public", ) -> None: conn.execute( """ INSERT INTO models ( model_id, provider, base_model_id, tier, context_window, effective_context_window, max_output_tokens, cost_per_1m_prompt, cost_per_1m_completion, cost_per_1m_prompt_cached, supports_vision, supports_json_mode, latency_class, reasoning_mode, context_variant, access_level, availability, last_updated ) VALUES (?, 'neuralwatt', ?, ?, ?, ?, 16384, ?, ?, ?, ?, ?, ?, 'default', 'full', ?, 'active', '2026-08-22T00:00:00+00:00') """, ( model_id, model_id, tier, context, context, cost, cost / 3, cost / 2, vision, json_mode, latency, access, ), ) def _seed_proficiency( conn: sqlite3.Connection, model_id: str, category: str, score: float, ) -> None: conn.execute( """ INSERT INTO proficiency ( model_id, provider, category, blended_score, source, last_updated ) VALUES (?, 'neuralwatt', ?, ?, 'blended', '2026-08-22T00:00:00+00:00') """, (model_id, category, score), ) def _seed_decision( conn: sqlite3.Connection, *, observed_at: str, category: str, tier: int = 1, context: int = 1000, latency: str = "interactive", selected: str, est_cost: float, est_prof: float, tools: int = 0, images: int = 0, json_mode: int = 0, ) -> None: conn.execute( """ INSERT INTO route_decisions ( observed_at, kind, task_category, task_tier, required_context_tokens, confidence, classifier_ms, classification_source, latency_tolerance, candidates_considered, selected_model, selected_provider, est_cost_usd, est_proficiency, tools, images, json_mode, streamed, flex_preference, flex_swapped, flex_forced ) VALUES (?, 'route', ?, ?, ?, 0.95, 200, 'classifier', ?, 3, ?, 'neuralwatt', ?, ?, ?, ?, ?, 0, 'auto', 0, 0) """, ( observed_at, category, tier, context, latency, selected, est_cost, est_prof, tools, images, json_mode, ), ) def _seed_basic_catalog(conn: sqlite3.Connection) -> None: # cheap: low proficiency, tier 1, large context _seed_model(conn, "cheap", cost=0.10, tier=1, context=131072) # dear: high proficiency, tier 1, large context _seed_model(conn, "dear", cost=9.00, tier=1, context=131072) # mid-small: cheap but small context (fails when the request needs more) _seed_model(conn, "small", cost=0.05, tier=1, context=4096) # low-tier: cheap and large but tier 3 (fails tier-1 requests) _seed_model(conn, "frontier", cost=7.00, tier=3, context=262128) _seed_proficiency(conn, "cheap", "coding_general", 0.40) _seed_proficiency(conn, "dear", "coding_general", 0.90) _seed_proficiency(conn, "small", "coding_general", 0.35) _seed_proficiency(conn, "frontier", "coding_general", 0.95) conn.commit() class TestBaselineSelection: def test_cheapest_picks_lowest_cost_eligible(self, conn) -> None: _seed_basic_catalog(conn) candidates = baseline_report.load_candidates(conn, "coding_general", CFG) eligible = baseline_report.select_candidates( candidates, required_context_tokens=1000, required_tier=1, latency_tolerance="interactive", allowed_access_levels=CFG.routing.allowed_access_levels, exclude_stale=CFG.freshness.exclude_stale, exclude_deprecated=CFG.freshness.exclude_deprecated, ) # cheap, small, dear, frontier are all eligible at tier 1 / 1000 tokens. cheapest, best = baseline_report.baseline_selection( eligible, decision_fake(context=1000), CFG ) assert cheapest["model_id"] == "small" # lowest list price assert best["model_id"] == "frontier" # highest proficiency, ties by cost def test_empty_eligible_set_yields_none(self, conn) -> None: _seed_basic_catalog(conn) cheapest, best = baseline_report.baseline_selection([], decision_fake(1000), CFG) assert cheapest is None assert best is None def test_missing_proficiency_defaults_to_half(self, conn) -> None: """A model with no proficiency row is treated as neutral (0.5).""" _seed_model(conn, "scored", cost=1.0, tier=1, context=131072) _seed_model(conn, "unscored", cost=1.0, tier=1, context=131072) _seed_proficiency(conn, "scored", "coding_general", 0.40) candidates = baseline_report.load_candidates(conn, "coding_general", CFG) eligible = baseline_report.select_candidates( candidates, required_context_tokens=1000, required_tier=1, latency_tolerance="interactive", allowed_access_levels=CFG.routing.allowed_access_levels, exclude_stale=CFG.freshness.exclude_stale, exclude_deprecated=CFG.freshness.exclude_deprecated, ) cheapest, best = baseline_report.baseline_selection( eligible, decision_fake(context=1000), CFG ) assert cheapest["model_id"] == "scored" assert best["model_id"] == "unscored" def decision_fake(context: int) -> dict: """A minimal decision-shaped mapping for baseline/reconstruct helpers.""" return { "required_context_tokens": context, "task_tier": 1, "latency_tolerance": "interactive", "tools": 0, "images": 0, "json_mode": 0, "task_category": "coding_general", "selected_model": None, "est_cost_usd": 0.0, "est_proficiency": 0.0, } class TestLoadDecisions: def test_chat_rows_are_loaded(self, conn) -> None: """The report must load scored routing decisions, not just 'route'. Real dogfooding traffic is recorded with kinds ``route``, ``chat`` and ``dispatch``; only rows where a model was actually selected carry a ``selected_model``. ``load_decisions`` should include those rows. """ _seed_basic_catalog(conn) conn.execute( """ INSERT INTO route_decisions ( observed_at, kind, task_category, task_tier, required_context_tokens, confidence, classifier_ms, classification_source, latency_tolerance, candidates_considered, selected_model, selected_provider, est_cost_usd, est_proficiency, tools, images, json_mode, streamed, flex_preference, flex_swapped, flex_forced ) VALUES (?, 'chat', ?, 1, 1000, 0.95, 200, 'classifier', 'interactive', 3, 'cheap', 'neuralwatt', 0.001, 0.40, 0, 0, 0, 0, 'auto', 0, 0) """, ("2026-08-20T00:00:00+00:00", "coding_general"), ) conn.commit() decisions = baseline_report.load_decisions(conn, since=None, category=None) assert len(decisions) == 1 assert decisions[0]["kind"] == "chat" class TestReconstruct: def test_constraints_respected(self, conn) -> None: # A decision needing a huge context must not offer "small". _seed_basic_catalog(conn) decision = decision_fake(context=10_000) cheapest, best = baseline_report.reconstruct_decision(conn, decision, CFG) # small dropped by context; frontier outranks dear on proficiency. assert cheapest["model_id"] == "cheap" assert best["model_id"] == "frontier" def test_vision_flag_filters_eligible_set(self, conn) -> None: _seed_basic_catalog(conn) conn.execute("UPDATE models SET supports_vision = 0 WHERE model_id != 'cheap'") conn.commit() decision = decision_fake(context=1000) decision["images"] = 1 cheapest, best = baseline_report.reconstruct_decision(conn, decision, CFG) assert cheapest["model_id"] == "cheap" assert best["model_id"] == "cheap" def test_json_mode_flag_filters_eligible_set(self, conn) -> None: _seed_basic_catalog(conn) conn.execute("UPDATE models SET supports_json_mode = 0 WHERE model_id != 'dear'") conn.commit() decision = decision_fake(context=1000) decision["json_mode"] = 1 cheapest, best = baseline_report.reconstruct_decision(conn, decision, CFG) assert cheapest["model_id"] == "dear" assert best["model_id"] == "dear" class TestAnalyze: def test_aggregate_and_dominance(self, conn) -> None: _seed_basic_catalog(conn) # Two decisions: one selects the cheapest eligible (small -> dominant), # one selects dear (not dominant). _seed_decision( conn, observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="small", est_cost=0.001, est_prof=0.35, ) _seed_decision( conn, observed_at="2026-08-21T00:00:00+00:00", category="coding_general", selected="dear", est_cost=0.09, est_prof=0.90, ) total, cat_rows = baseline_report.analyze(conn, CFG) assert total["count"] == 2 assert total["dominance_count"] == 1 assert total["dominance_pct"] == 50.0 assert len(cat_rows) == 1 assert cat_rows[0]["category"] == "coding_general" def test_category_filter(self, conn) -> None: _seed_basic_catalog(conn) _seed_decision( conn, observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="cheap", est_cost=0.001, est_prof=0.40, ) _seed_decision( conn, observed_at="2026-08-20T00:00:00+00:00", category="translation", selected="dear", est_cost=0.09, est_prof=0.70, ) total, _ = baseline_report.analyze(conn, CFG, category="translation") assert total["count"] == 1 assert total["actual_cost"] == pytest.approx(0.09) def test_since_filter(self, conn) -> None: _seed_basic_catalog(conn) _seed_decision( conn, observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="cheap", est_cost=0.001, est_prof=0.40, ) _seed_decision( conn, observed_at="2026-08-25T00:00:00+00:00", category="coding_general", selected="dear", est_cost=0.09, est_prof=0.90, ) total, _ = baseline_report.analyze(conn, CFG, since="2026-08-21") assert total["count"] == 1 assert total["actual_cost"] == pytest.approx(0.09) def test_empty_window_returns_zero_counts(self, conn) -> None: _seed_basic_catalog(conn) total, cat_rows = baseline_report.analyze(conn, CFG, since="2099-01-01") assert total["count"] == 0 assert total["dominance_pct"] is None assert cat_rows == [] class TestOutput: def test_format_summary_has_dominance(self, conn) -> None: _seed_basic_catalog(conn) _seed_decision( conn, observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="small", est_cost=0.001, est_prof=0.35, ) total, cat_rows = baseline_report.analyze(conn, CFG) text = baseline_report.format_summary(total, cat_rows) assert "dominance" in text assert "100.0%" in text def test_csv_parses(self, conn) -> None: _seed_basic_catalog(conn) _seed_decision( conn, observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="small", est_cost=0.001, est_prof=0.35, ) total, cat_rows = baseline_report.analyze(conn, CFG) buf = io.StringIO() baseline_report.write_csv(total, cat_rows, buf) buf.seek(0) rows = list(csv.DictReader(buf)) assert rows[0]["category"] == "total" assert rows[1]["category"] == "coding_general" assert rows[0]["count"] == "1" assert rows[0]["dominance_pct"] == "100.0" def _seed_energy_observation( conn: sqlite3.Connection, provider: str, model_id: str, prompt_tokens: int, cached_prompt_tokens: int, observed_at: str = "2026-08-20T00:00:00+00:00", ) -> None: conn.execute( """ INSERT INTO energy_observations ( request_id, provider, model_id, task_category, prompt_tokens, cached_prompt_tokens, completion_tokens, observed_at ) VALUES (?, ?, ?, NULL, ?, ?, ?, ?) """, (f"req-{model_id}", provider, model_id, prompt_tokens, cached_prompt_tokens, 500, observed_at), ) def _seed_session_decision( conn: sqlite3.Connection, *, session_key: str, observed_at: str, category: str, selected: str, context: int = 1000, ) -> None: conn.execute( """ INSERT INTO route_decisions ( observed_at, kind, task_category, task_tier, required_context_tokens, confidence, classifier_ms, classification_source, latency_tolerance, candidates_considered, selected_model, selected_provider, est_cost_usd, est_proficiency, tools, images, json_mode, streamed, flex_preference, flex_swapped, flex_forced, session_key ) VALUES (?, 'route', ?, 1, ?, 0.95, 200, 'classifier', 'interactive', 3, ?, 'neuralwatt', 0.01, 0.50, 0, 0, 0, 0, 'auto', 0, 0, ?) """, (observed_at, category, context, selected, session_key), ) class TestIncumbentFree: """Tests for the incumbent-free counterfactual in baseline_report.""" def _hot_cold_data(self, conn: sqlite3.Connection, hot_selected: bool) -> None: """Seed hot and cold models with same proficiency but different costs. hot: cost_per_1m_prompt = 2.00 — base price, higher cold: cost_per_1m_prompt = 1.80 — base price, lower completion: 0.01/1m for both — small, won't matter. With 1000 prompt tokens, 500 completion: - at base: hot = 1.977, cold = 1.868 → cold cheaper - at cache=80%: cold even cheaper We need to seed energy_observations so hot has measured cache rate. The neutral counterfactual produces cold winner (same as real at neutral). At challenger=0.0, cold is still cheaper → same winner. For a more dramatic test, see test_challenger_switches below. """ _seed_model(conn, "hot", cost=2.00, tier=1, context=131072) _seed_model(conn, "cold", cost=1.80, tier=1, context=131072) _seed_proficiency(conn, "hot", "coding_general", 0.50) _seed_proficiency(conn, "cold", "coding_general", 0.50) # Seed cache data for hot: 950 cached out of 1000 prompt tokens _seed_energy_observation(conn, "neuralwatt", "hot", 1000, 950, "2026-08-20T00:00:00+00:00") if hot_selected: _seed_session_decision(conn, session_key="s1", observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="hot", context=1000) else: _seed_session_decision(conn, session_key="s1", observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="cold", context=1000) conn.commit() def test_neutral_reproduces_real_choice(self, conn) -> None: """The neutral dial reproduces the real routing choice. At neutral dial the incumbent-free ranking is identical to the pre-feature ranking (no incumbency pricing). Both paths produce the same winner. """ self._hot_cold_data(conn, hot_selected=False) total, cat_rows = baseline_report.analyze(conn, CFG, since=None, category=None) assert total["count"] == 1 # Neutral matched: the neutral counterfactual's choice # matches the real decision's selected_model. assert total["incumbent_neutral_match"] == 1 def test_challenger_switches_when_strictly_cheaper(self, conn) -> None: """Challenger dial 0.0 changes the chosen model when the cold challenger is strictly cheaper in the same quality band. hot (base 100.00, measured cache rate 0.90) vs cold (base 5.00). Both proficiency 0.50 — same band. At challenger=0.0 the challenger prices at full (5.00) and beats the incumbent's discounted price (10.00), so the counterfactual's choice differs from the real row's hot selection. """ _seed_model(conn, "hot", cost=100.00, tier=1, context=131072) _seed_model(conn, "cold", cost=5.00, tier=1, context=131072) _seed_proficiency(conn, "hot", "coding_general", 0.50) _seed_proficiency(conn, "cold", "coding_general", 0.50) _seed_energy_observation(conn, "neuralwatt", "hot", 1000, 900, "2026-08-20T00:00:00+00:00") _seed_session_decision(conn, session_key="s1", observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="hot", context=1000) conn.commit() total, cat_rows = baseline_report.analyze(conn, CFG, since=None, category=None) assert total["count"] == 1 # Challenger=0.0: cold's full price (5.00) beats hot's discounted # price (10.00) → the counterfactual picks cold, differing from # the real row's hot. assert total["incumbent_challenger_match"] == 0 assert total["incumbent_challenger_winner"] == "cold" def test_incumbent_retained_when_not_strictly_cheaper(self, conn) -> None: """The other arm: at challenger=0.0 the incumbent is retained when the cold challenger is NOT strictly cheaper. hot (base 1.00, measured cache rate 0.90) vs cold (base 5.00). hot's discounted price (0.10) beats cold's full price (5.00). """ _seed_model(conn, "hot", cost=1.00, tier=1, context=131072) _seed_model(conn, "cold", cost=5.00, tier=1, context=131072) _seed_proficiency(conn, "hot", "coding_general", 0.50) _seed_proficiency(conn, "cold", "coding_general", 0.50) _seed_energy_observation(conn, "neuralwatt", "hot", 1000, 900, "2026-08-20T00:00:00+00:00") _seed_session_decision(conn, session_key="s1", observed_at="2026-08-20T00:00:00+00:00", category="coding_general", selected="hot", context=1000) conn.commit() total, cat_rows = baseline_report.analyze(conn, CFG, since=None, category=None) assert total["count"] == 1 # hot's discounted price (0.10) beats cold's full price (5.00) assert total["incumbent_challenger_match"] == 1 assert total["incumbent_challenger_winner"] == "hot" def test_eviction_path_no_crash(self, conn) -> None: """A session whose incumbent is evicted by a hard filter must not crash the incumbent-free replay. A prior chat turn selects cold-small (its incumbent), then the replayed decision needs 10000 tokens — cold-small (4096 context) is evicted by the hard filter, so the incumbent is absent from the eligible set. The counterfactual must not crash and must still produce a choice from the surviving candidates. """ _seed_model(conn, "hot-big", cost=1.00, tier=1, context=262128) _seed_model(conn, "cold-small", cost=0.50, tier=1, context=4096) _seed_proficiency(conn, "hot-big", "coding_general", 0.50) _seed_proficiency(conn, "cold-small", "coding_general", 0.50) # Prior chat turn: session s2's incumbent is cold-small conn.execute( """ INSERT INTO route_decisions ( observed_at, kind, task_category, task_tier, required_context_tokens, confidence, classifier_ms, classification_source, latency_tolerance, candidates_considered, selected_model, selected_provider, est_cost_usd, est_proficiency, tools, images, json_mode, streamed, flex_preference, flex_swapped, flex_forced, session_key ) VALUES (?, 'chat', 'coding_general', 1, 1000, 0.95, 200, 'classifier', 'interactive', 3, 'cold-small', 'neuralwatt', 0.01, 0.50, 0, 0, 0, 0, 'auto', 0, 0, 's2') """, ("2026-08-20T00:00:00+00:00",), ) # The replayed decision needs 10000 tokens: cold-small evicted _seed_session_decision(conn, session_key="s2", observed_at="2026-08-21T00:00:00+00:00", category="coding_general", selected="hot-big", context=10000) conn.commit() total, cat_rows = baseline_report.analyze(conn, CFG, since=None, category=None) # load_decisions only returns rows with selected_model, and both # rows have one — so count is 2 (chat + route). assert total["count"] == 2 # The route row (hot-big, only surviving candidate) matches on # both dials. The chat row's incumbent lookup sees itself as the # incumbent (kind='chat' matches the allowlist), and cold-small # is eligible at 1000 tokens — so the challenger dial may pick # hot-big there; either way no crash and the aggregate is sane. assert total["incumbent_neutral_match"] >= 1 assert total["incumbent_challenger_match"] >= 1