Files
6krrt/tests/test_progress_backtest.py

317 lines
12 KiB
Python

"""Integration tests for scripts/progress_backtest.py.
Tests the fixture-based backtest output format, label correctness,
boundary conditions, and ensures no real opencode calls are made.
"""
import json
import os
import re
import sys
from progress_detect import DetectConfig, evaluate
_SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts")
_PROGRESS_BACKTEST = os.path.join(_SCRIPTS_DIR, "progress_backtest.py")
_FIXTURE_PATH = os.path.join(
os.path.dirname(__file__), "fixtures", "progress", "fixture.json"
)
_ATLAS_ID = "ses_f2506ac70ffebLUxBsq0aauOn9"
_FLAG_RE = re.compile(
r"^(FLAG\s+|ok\s+ )"
r"calls=(\s*\d+)"
r" first_flag_at=(\d+|N/A)"
r" dup=(\d+\.\d+)"
r" top=(\d+)"
r" landed=(\d+)"
r" ro=(\d+)"
r" \| (.+)$"
)
# ---------------------------------------------------------------------------
# helpers
# ---------------------------------------------------------------------------
def _calls_from_fixture(session):
"""Convert a fixture session dict to call tuples for evaluate()."""
return [
(c["t"], c["tool"], json.dumps(c["args"]), c["landed"])
for c in session["calls"]
]
def _file_lines_fn(file_lines_dict):
"""Return a callable for coverage() from a file_path→line_count dict."""
def fn(fp):
if fp is None:
return 0
return file_lines_dict.get(fp, 0) or 0
return fn
def _run_backtest(fixture_path):
"""Run the backtest script with --fixture and return parsed lines + stderr."""
import subprocess
worktree_root = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
result = subprocess.run(
[sys.executable, _PROGRESS_BACKTEST, "--fixture", fixture_path],
capture_output=True,
text=True,
check=False,
cwd=worktree_root,
env={**os.environ, "PYTHONPATH": os.path.join(worktree_root, "src")},
)
lines = [l for l in result.stdout.strip().split("\n") if l.strip() and not l.startswith("#")]
return result, lines
def _load_fixture():
"""Load fixture JSON."""
with open(_FIXTURE_PATH) as f:
return json.load(f)
def _parse_line(line):
"""Parse a backtest output line into a dict, or None if invalid."""
m = _FLAG_RE.match(line)
if not m:
return None
return {
"status": "flag" if m.group(1).startswith("FLAG") else "ok",
"calls": int(m.group(2)),
"first_flag_at": None if m.group(3) == "N/A" else int(m.group(3)),
"dup": float(m.group(4)),
"top": int(m.group(5)),
"landed": int(m.group(6)),
"ro": int(m.group(7)),
"title": m.group(8),
}
# ---------------------------------------------------------------------------
# fixture label correctness
# ---------------------------------------------------------------------------
class TestFixtureLabels:
"""Verify backtest results match fixture labels using the detector directly."""
@staticmethod
def _check_session(session):
"""Run sliding-window backtest returns (first_flag_at, reason)."""
cfg = DetectConfig()
calls = _calls_from_fixture(session)
fl = session.get("file_lines", {})
fl_fn = _file_lines_fn(fl)
title = session.get("title", "")
if len(calls) < cfg.min_calls:
return (False, None)
for end_idx in range(cfg.min_calls, len(calls) + 1, 5):
flag, reason = evaluate(calls, cfg, fl_fn, title, end_idx=end_idx)
if flag and reason:
return (True, reason)
return (False, None)
def test_must_flag_sessions_all_flag(self):
"""Every must_flag session has at least one window where flag=True."""
sessions = _load_fixture()
must_flag = [s for s in sessions if s["label"] == "must_flag"]
for s in must_flag:
flagged, _ = self._check_session(s)
assert flagged, (
f"Session {s['session_id']} (must_flag) should flag but did not"
)
def test_must_not_flag_sessions_never_flag(self):
"""Every must_not_flag session has NO window where flag=True."""
sessions = _load_fixture()
must_not = [s for s in sessions if s["label"] == "must_not_flag"]
for s in must_not:
sid = s["session_id"]
flagged, _ = self._check_session(s)
assert not flagged, (
f"Session {sid} (must_not_flag) should not flag but did"
)
# ---------------------------------------------------------------------------
# Atlas 02:07 boundary
# ---------------------------------------------------------------------------
class TestAtlasBoundary:
"""Atlas session: no flag before 02:07; flag exists post-02:07."""
def test_atlas_no_flag_before_cutoff(self):
"""No flagged window before 02:07 wall-clock time."""
sessions = _load_fixture()
atlas = next(s for s in sessions if s["session_id"] == _ATLAS_ID)
calls = _calls_from_fixture(atlas)
fl_fn = _file_lines_fn(atlas.get("file_lines", {}))
cfg = DetectConfig()
cutoff = 1790402856356 # 02:07:36 UTC wall-clock
pre_flagged = []
for i in range(cfg.min_calls, len(calls) + 1, 5):
subset = calls[:i]
if len(subset) < cfg.min_calls:
continue
win = subset[-cfg.window:]
if len(win) < cfg.min_calls:
continue
win_end = win[-1][0]
flag, _ = evaluate(calls, cfg, fl_fn, atlas["title"], end_idx=i)
if flag and win_end < cutoff:
pre_flagged.append(i)
assert len(pre_flagged) == 0, (
f"Expected no flags before 02:07, got {len(pre_flagged)} "
f"at indices {pre_flagged}"
)
def test_atlas_flag_after_cutoff(self):
"""At least one flag after 02:07 wall-clock time."""
sessions = _load_fixture()
atlas = next(s for s in sessions if s["session_id"] == _ATLAS_ID)
calls = _calls_from_fixture(atlas)
fl_fn = _file_lines_fn(atlas.get("file_lines", {}))
cfg = DetectConfig()
cutoff = 1790402856356
post_flagged = []
for i in range(cfg.min_calls, len(calls) + 1, 5):
subset = calls[:i]
if len(subset) < cfg.min_calls:
continue
win = subset[-cfg.window:]
if len(win) < cfg.min_calls:
continue
win_end = win[-1][0]
flag, _ = evaluate(calls, cfg, fl_fn, atlas["title"], end_idx=i)
if flag and win_end >= cutoff:
post_flagged.append(i)
assert len(post_flagged) >= 1, (
"Expected at least one flag after 02:07, got none"
)
# ---------------------------------------------------------------------------
# backtest output format
# ---------------------------------------------------------------------------
class TestOutputFormat:
"""Test that the backtest script produces correctly formatted output."""
def test_backtest_output_parseable(self):
"""All non-comment, non-empty output lines parse correctly."""
result, lines = _run_backtest(_FIXTURE_PATH)
assert result.returncode == 0, f"Backtest failed: {result.stderr}"
assert len(lines) == 15, f"Expected 15 session lines, got {len(lines)}"
parsed = [_parse_line(l) for l in lines]
assert all(p is not None for p in parsed), (
f"Some lines failed parsing. Sample: {lines[0] if lines else '(empty)'}"
)
def test_flag_count_matches_fixture(self):
"""FLAG count equals must_flag count."""
sessions = _load_fixture()
must_flag = len([s for s in sessions if s["label"] == "must_flag"])
expected_flags = must_flag
_, lines = _run_backtest(_FIXTURE_PATH)
parsed = [p for p in [_parse_line(l) for l in lines] if p]
actual_flags = sum(1 for p in parsed if p["status"] == "flag")
assert actual_flags == expected_flags, (
f"Expected {expected_flags} flagged, got {actual_flags}"
)
def test_flagged_session_has_first_flag_at(self):
"""FLAG lines have a numeric first_flag_at; ok lines have None."""
_, lines = _run_backtest(_FIXTURE_PATH)
parsed = [_parse_line(l) for l in lines if _parse_line(l)]
for p in parsed:
if p["status"] == "flag":
assert p["first_flag_at"] is not None, (
f"FLAG line should have numeric first_flag_at: {p}"
)
else:
assert p["first_flag_at"] is None, (
f"ok line should have None first_flag_at: {p}"
)
def test_call_counts_match_fixture(self):
"""Each line's calls= field matches the fixture call count."""
sessions = _load_fixture()
sid_to_calls = {s["session_id"]: len(s["calls"]) for s in sessions}
# We need to match lines to sessions — titles are unique
title_to_sid = {}
for s in sessions:
title_to_sid[s["title"]] = s["session_id"]
_, lines = _run_backtest(_FIXTURE_PATH)
parsed = [_parse_line(l) for l in lines if _parse_line(l)]
assert len(parsed) == len(sessions)
for p in parsed:
sid = title_to_sid.get(p["title"])
if sid:
assert p["calls"] == sid_to_calls[sid], (
f"Call count mismatch for {sid}: expected {sid_to_calls[sid]}, got {p['calls']}"
)
def test_output_includes_summary_in_stderr(self):
"""Stderr includes a summary line like '# backtest complete: N/M flagged'."""
result, _lines = _run_backtest(_FIXTURE_PATH)
summary_re = re.compile(r"# backtest complete: \d+/\d+ flagged")
assert summary_re.search(result.stderr), (
f"Expected summary in stderr, got: {result.stderr}"
)
# ---------------------------------------------------------------------------
# empty fixture edge case
# ---------------------------------------------------------------------------
class TestEmptyFixture:
"""Test edge cases with empty or minimal fixture data."""
def test_empty_fixture_file(self, tmp_path):
"""Empty fixture produces zero session lines."""
fixture_file = tmp_path / "empty_fixture.json"
fixture_file.write_text("[]")
result, lines = _run_backtest(str(fixture_file))
assert result.returncode == 0
# Only the summary line should remain after filtering
assert len(lines) == 0, f"Expected 0 lines for empty fixture, got: {lines}"
def test_fixture_with_too_few_calls(self, tmp_path):
"""Fixture with sessions under min_calls produces all ok."""
fixture_file = tmp_path / "few_calls.json"
# Build a minimal fixture with 20 calls (below min_calls=40)
calls = []
for i in range(20):
calls.append({
"t": 1790000000000 + i,
"tool": "read",
"args": {"filePath": f"/tmp/file_{i}.py"},
"landed": False,
})
fixture_data = [{
"session_id": "ses_empty_test",
"title": "Tiny session",
"agent": "test",
"calls": calls,
"file_lines": {},
"label": "must_not_flag",
}]
fixture_file.write_text(json.dumps(fixture_data))
result, lines = _run_backtest(str(fixture_file))
assert result.returncode == 0
assert len(lines) == 1
parsed = _parse_line(lines[0])
assert parsed is not None
assert parsed["status"] == "ok"
assert parsed["first_flag_at"] is None