Files
6krrt/tests/test_eval_heldout.py
adlee-was-taken 0b85b6366c feat(eval): promote held-out set to evals/heldout.yaml
Copy plans/local-decision-classifier-heldout.yaml (30 tasks) to
evals/heldout.yaml unchanged, and add a loader test asserting the
file loads 30 scoreable tasks with the expected categories.

The held-out set is weak: single author, short prompts, no true
distributional shift from the training set. Treat 100% as a ceiling,
not a forecast -- it cannot measure generalization.
2026-09-28 19:56:55 -04:00

63 lines
1.8 KiB
Python

"""Tests that evals/heldout.yaml is a loadable, scoreable task set.
The held-out set is a fixed benchmark: 30 tasks across 10 categories, with
tool-use tasks excluded from the scoreable set (as in the plan). These tests
guard the loader contract against a malformed or drifted file.
"""
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))
from eval_classifier import EXCLUDED_CATEGORIES, load_scoreable_tasks
HELDOUT_PATH = Path(__file__).resolve().parent.parent / "evals" / "heldout.yaml"
# The 10 categories in the held-out set, one per 3-task group.
EXPECTED_CATEGORIES = [
"coding_general",
"coding_refactor",
"debugging",
"docs_writing",
"summarization",
"file_summarization",
"diff_checking",
"translation",
"reasoning_math",
"general_chat",
]
def test_heldout_file_exists():
assert HELDOUT_PATH.exists(), "evals/heldout.yaml must be committed alongside the loader"
def test_heldout_has_30_scoreable_tasks():
tasks = load_scoreable_tasks(str(HELDOUT_PATH))
assert len(tasks) == 30
def test_heldout_categories_match_expected():
tasks = load_scoreable_tasks(str(HELDOUT_PATH))
categories = [t["category"] for t in tasks]
# The held-out set is grouped: three tasks per category, in order.
expected = [c for c in EXPECTED_CATEGORIES for _ in range(3)]
assert categories == expected
def test_heldout_tasks_have_id_and_prompt():
tasks = load_scoreable_tasks(str(HELDOUT_PATH))
for t in tasks:
assert t["id"]
assert t["prompt"]
def test_heldout_excludes_tool_use():
# The scoreable set must never admit tool-use tasks (per the plan).
tasks = load_scoreable_tasks(str(HELDOUT_PATH))
cats = {t["category"] for t in tasks}
assert not (cats & set(EXCLUDED_CATEGORIES))