Gate G results: - Eval-set accuracy: 97.8% (PASS >= 85%) - Held-out accuracy: 90.0% (PASS >= 90%) - p95 latency: 502ms (PASS <= 800ms) - 14b router resident after 100 calls: EVICTED (FAIL) VRAM constraint: 14b router (17.76GB) + qwen3.5:4b (>=6.15GB) = 25.1GB > 24GB RTX 6000. No num_ctx allows coexistence. STOP per plan: todos 4-17 NOT started.
162 lines
5.0 KiB
Python
162 lines
5.0 KiB
Python
"""Structural tests for local_decision._DECISION_DESCRIPTIONS.
|
|
|
|
Verifies:
|
|
- All 10 expected category keys are present.
|
|
- `tool_use_agentic` is excluded.
|
|
- The dict is NOT a shallow copy of local_encoder._CATEGORY_DESCRIPTIONS.
|
|
- local_encoder.py is NOT modified by this change (git diff is empty).
|
|
- classify_choice would work with these descriptions (they're valid option values).
|
|
"""
|
|
|
|
import pathlib
|
|
import subprocess
|
|
|
|
|
|
def _encoder_descriptions():
|
|
"""Import and return local_encoder._CATEGORY_DESCRIPTIONS."""
|
|
import local_encoder
|
|
return local_encoder._CATEGORY_DESCRIPTIONS
|
|
|
|
|
|
def _decision_descriptions():
|
|
"""Import and return local_decision._DECISION_DESCRIPTIONS."""
|
|
import local_decision
|
|
return local_decision._DECISION_DESCRIPTIONS
|
|
|
|
|
|
# Expected keys: all 11 minus tool_use_agentic = 10
|
|
_EXPECTED_KEYS = sorted([
|
|
"coding_general",
|
|
"coding_refactor",
|
|
"debugging",
|
|
"docs_writing",
|
|
"summarization",
|
|
"file_summarization",
|
|
"diff_checking",
|
|
"translation",
|
|
"reasoning_math",
|
|
"general_chat",
|
|
])
|
|
|
|
|
|
# --- Key presence -----------------------------------------------------------
|
|
|
|
|
|
def test_decision_descriptions_has_exactly_10_keys():
|
|
"""_DECISION_DESCRIPTIONS has exactly 10 keys."""
|
|
d = _decision_descriptions()
|
|
assert len(d) == 10
|
|
|
|
|
|
def test_decision_descriptions_keys_match_expected():
|
|
"""All 10 expected category keys are present, no extras."""
|
|
d = _decision_descriptions()
|
|
assert sorted(d.keys()) == _EXPECTED_KEYS
|
|
|
|
|
|
def test_tool_use_agentic_excluded():
|
|
"""tool_use_agentic must NOT appear in _DECISION_DESCRIPTIONS."""
|
|
d = _decision_descriptions()
|
|
assert "tool_use_agentic" not in d
|
|
|
|
|
|
# --- Not a shallow copy of encoder descriptions ------------------------------
|
|
|
|
|
|
def test_decision_descriptions_not_same_object_as_encoder():
|
|
"""_DECISION_DESCRIPTIONS is a distinct dict, not local_encoder's."""
|
|
d = _decision_descriptions()
|
|
e = _encoder_descriptions()
|
|
assert d is not e
|
|
|
|
|
|
def test_decision_descriptions_differ_from_encoder():
|
|
"""At least one value differs — must not be a shallow copy."""
|
|
d = _decision_descriptions()
|
|
e = _encoder_descriptions()
|
|
# They must NOT be equal at the value level
|
|
assert d != e
|
|
|
|
|
|
def test_coding_general_differ():
|
|
"""coding_general was tuned — should include 'tracing' wording."""
|
|
d = _decision_descriptions()
|
|
e = _encoder_descriptions()
|
|
assert d["coding_general"] != e["coding_general"]
|
|
assert "tracing" in d["coding_general"]
|
|
|
|
|
|
def test_debugging_differ():
|
|
"""debugging was tuned — should include 'fixing a bug' wording."""
|
|
d = _decision_descriptions()
|
|
e = _encoder_descriptions()
|
|
assert d["debugging"] != e["debugging"]
|
|
assert "fixing a bug" in d["debugging"]
|
|
|
|
|
|
def test_reasoning_math_differ():
|
|
"""reasoning_math was tuned — should be a pure word problem, no code."""
|
|
d = _decision_descriptions()
|
|
e = _encoder_descriptions()
|
|
assert d["reasoning_math"] != e["reasoning_math"]
|
|
assert "no code" in d["reasoning_math"]
|
|
|
|
|
|
def test_file_summarization_differ():
|
|
"""file_summarization was tuned — should emphasize 'single source file'."""
|
|
d = _decision_descriptions()
|
|
e = _encoder_descriptions()
|
|
assert d["file_summarization"] != e["file_summarization"]
|
|
assert "single source file" in d["file_summarization"]
|
|
|
|
|
|
def test_docs_writing_differ():
|
|
"""docs_writing was tuned — should include '(not code)' qualifier."""
|
|
d = _decision_descriptions()
|
|
e = _encoder_descriptions()
|
|
assert d["docs_writing"] != e["docs_writing"]
|
|
assert "(not code)" in d["docs_writing"]
|
|
|
|
|
|
# --- Keep-same keys still differ from encoder --------------------------------
|
|
# Even categories marked "keep" differ because the encoder has 11 keys
|
|
# and decision has 10 — the decision dict is a new dict, so `is` must fail.
|
|
|
|
|
|
def test_unmodified_keys_still_distinct_object():
|
|
"""summarization key value is the same text but the dict itself differs."""
|
|
d = _decision_descriptions()
|
|
e = _encoder_descriptions()
|
|
assert d is not e
|
|
|
|
|
|
# --- local_encoder.py is NOT modified ----------------------------------------
|
|
|
|
|
|
def test_local_encoder_file_unmodified():
|
|
"""git diff src/local_encoder.py must be empty — no encoder edits."""
|
|
repo_root = pathlib.Path(__file__).parent.parent.parent
|
|
result = subprocess.run(
|
|
["git", "diff", "src/local_encoder.py"],
|
|
capture_output=True,
|
|
text=True,
|
|
check=False,
|
|
cwd=str(repo_root),
|
|
)
|
|
assert result.stdout == "", (
|
|
f"local_encoder.py was modified! Diff:\n{result.stdout}"
|
|
)
|
|
|
|
|
|
# --- classify_choice compatibility -------------------------------------------
|
|
|
|
|
|
def test_decision_descriptions_keys_can_be_used_as_options():
|
|
"""The 10 keys are valid option identifiers for classify_choice."""
|
|
d = _decision_descriptions()
|
|
# Keys should all be valid strings suitable for option labels
|
|
for key, value in d.items():
|
|
assert isinstance(key, str)
|
|
assert isinstance(value, str)
|
|
assert len(value) > 0, f"{key!r} has empty description"
|