Copy plans/local-decision-classifier-heldout.yaml (30 tasks) to evals/heldout.yaml unchanged, and add a loader test asserting the file loads 30 scoreable tasks with the expected categories. The held-out set is weak: single author, short prompts, no true distributional shift from the training set. Treat 100% as a ceiling, not a forecast -- it cannot measure generalization.
63 lines
1.8 KiB
Python
63 lines
1.8 KiB
Python
"""Tests that evals/heldout.yaml is a loadable, scoreable task set.
|
|
|
|
The held-out set is a fixed benchmark: 30 tasks across 10 categories, with
|
|
tool-use tasks excluded from the scoreable set (as in the plan). These tests
|
|
guard the loader contract against a malformed or drifted file.
|
|
"""
|
|
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))
|
|
|
|
from eval_classifier import EXCLUDED_CATEGORIES, load_scoreable_tasks
|
|
|
|
HELDOUT_PATH = Path(__file__).resolve().parent.parent / "evals" / "heldout.yaml"
|
|
|
|
# The 10 categories in the held-out set, one per 3-task group.
|
|
EXPECTED_CATEGORIES = [
|
|
"coding_general",
|
|
"coding_refactor",
|
|
"debugging",
|
|
"docs_writing",
|
|
"summarization",
|
|
"file_summarization",
|
|
"diff_checking",
|
|
"translation",
|
|
"reasoning_math",
|
|
"general_chat",
|
|
]
|
|
|
|
|
|
def test_heldout_file_exists():
|
|
assert HELDOUT_PATH.exists(), "evals/heldout.yaml must be committed alongside the loader"
|
|
|
|
|
|
def test_heldout_has_30_scoreable_tasks():
|
|
tasks = load_scoreable_tasks(str(HELDOUT_PATH))
|
|
assert len(tasks) == 30
|
|
|
|
|
|
def test_heldout_categories_match_expected():
|
|
tasks = load_scoreable_tasks(str(HELDOUT_PATH))
|
|
categories = [t["category"] for t in tasks]
|
|
# The held-out set is grouped: three tasks per category, in order.
|
|
expected = [c for c in EXPECTED_CATEGORIES for _ in range(3)]
|
|
assert categories == expected
|
|
|
|
|
|
def test_heldout_tasks_have_id_and_prompt():
|
|
tasks = load_scoreable_tasks(str(HELDOUT_PATH))
|
|
for t in tasks:
|
|
assert t["id"]
|
|
assert t["prompt"]
|
|
|
|
|
|
def test_heldout_excludes_tool_use():
|
|
# The scoreable set must never admit tool-use tasks (per the plan).
|
|
tasks = load_scoreable_tasks(str(HELDOUT_PATH))
|
|
cats = {t["category"] for t in tasks}
|
|
assert not (cats & set(EXCLUDED_CATEGORIES))
|