"""Integration tests for scripts/progress_backtest.py. Tests the fixture-based backtest output format, label correctness, boundary conditions, and ensures no real opencode calls are made. """ import json import os import re import sys from progress_detect import DetectConfig, evaluate _SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") _PROGRESS_BACKTEST = os.path.join(_SCRIPTS_DIR, "progress_backtest.py") _FIXTURE_PATH = os.path.join( os.path.dirname(__file__), "fixtures", "progress", "fixture.json" ) _ATLAS_ID = "ses_f2506ac70ffebLUxBsq0aauOn9" _FLAG_RE = re.compile( r"^(FLAG\s+|ok\s+ )" r"calls=(\s*\d+)" r" first_flag_at=(\d+|N/A)" r" dup=(\d+\.\d+)" r" top=(\d+)" r" landed=(\d+)" r" ro=(\d+)" r" \| (.+)$" ) # --------------------------------------------------------------------------- # helpers # --------------------------------------------------------------------------- def _calls_from_fixture(session): """Convert a fixture session dict to call tuples for evaluate().""" return [ (c["t"], c["tool"], json.dumps(c["args"]), c["landed"]) for c in session["calls"] ] def _file_lines_fn(file_lines_dict): """Return a callable for coverage() from a file_path→line_count dict.""" def fn(fp): if fp is None: return 0 return file_lines_dict.get(fp, 0) or 0 return fn def _run_backtest(fixture_path): """Run the backtest script with --fixture and return parsed lines + stderr.""" import subprocess worktree_root = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) result = subprocess.run( [sys.executable, _PROGRESS_BACKTEST, "--fixture", fixture_path], capture_output=True, text=True, check=False, cwd=worktree_root, env={**os.environ, "PYTHONPATH": os.path.join(worktree_root, "src")}, ) lines = [l for l in result.stdout.strip().split("\n") if l.strip() and not l.startswith("#")] return result, lines def _load_fixture(): """Load fixture JSON.""" with open(_FIXTURE_PATH) as f: return json.load(f) def _parse_line(line): """Parse a backtest output line into a dict, or None if invalid.""" m = _FLAG_RE.match(line) if not m: return None return { "status": "flag" if m.group(1).startswith("FLAG") else "ok", "calls": int(m.group(2)), "first_flag_at": None if m.group(3) == "N/A" else int(m.group(3)), "dup": float(m.group(4)), "top": int(m.group(5)), "landed": int(m.group(6)), "ro": int(m.group(7)), "title": m.group(8), } # --------------------------------------------------------------------------- # fixture label correctness # --------------------------------------------------------------------------- class TestFixtureLabels: """Verify backtest results match fixture labels using the detector directly.""" @staticmethod def _check_session(session): """Run sliding-window backtest returns (first_flag_at, reason).""" cfg = DetectConfig() calls = _calls_from_fixture(session) fl = session.get("file_lines", {}) fl_fn = _file_lines_fn(fl) title = session.get("title", "") if len(calls) < cfg.min_calls: return (False, None) for end_idx in range(cfg.min_calls, len(calls) + 1, 5): flag, reason = evaluate(calls, cfg, fl_fn, title, end_idx=end_idx) if flag and reason: return (True, reason) return (False, None) def test_must_flag_sessions_all_flag(self): """Every must_flag session has at least one window where flag=True.""" sessions = _load_fixture() must_flag = [s for s in sessions if s["label"] == "must_flag"] for s in must_flag: flagged, _ = self._check_session(s) assert flagged, ( f"Session {s['session_id']} (must_flag) should flag but did not" ) def test_must_not_flag_sessions_never_flag(self): """Every must_not_flag session has NO window where flag=True.""" sessions = _load_fixture() must_not = [s for s in sessions if s["label"] == "must_not_flag"] for s in must_not: sid = s["session_id"] flagged, _ = self._check_session(s) assert not flagged, ( f"Session {sid} (must_not_flag) should not flag but did" ) # --------------------------------------------------------------------------- # Atlas 02:07 boundary # --------------------------------------------------------------------------- class TestAtlasBoundary: """Atlas session: no flag before 02:07; flag exists post-02:07.""" def test_atlas_no_flag_before_cutoff(self): """No flagged window before 02:07 wall-clock time.""" sessions = _load_fixture() atlas = next(s for s in sessions if s["session_id"] == _ATLAS_ID) calls = _calls_from_fixture(atlas) fl_fn = _file_lines_fn(atlas.get("file_lines", {})) cfg = DetectConfig() cutoff = 1790402856356 # 02:07:36 UTC wall-clock pre_flagged = [] for i in range(cfg.min_calls, len(calls) + 1, 5): subset = calls[:i] if len(subset) < cfg.min_calls: continue win = subset[-cfg.window:] if len(win) < cfg.min_calls: continue win_end = win[-1][0] flag, _ = evaluate(calls, cfg, fl_fn, atlas["title"], end_idx=i) if flag and win_end < cutoff: pre_flagged.append(i) assert len(pre_flagged) == 0, ( f"Expected no flags before 02:07, got {len(pre_flagged)} " f"at indices {pre_flagged}" ) def test_atlas_flag_after_cutoff(self): """At least one flag after 02:07 wall-clock time.""" sessions = _load_fixture() atlas = next(s for s in sessions if s["session_id"] == _ATLAS_ID) calls = _calls_from_fixture(atlas) fl_fn = _file_lines_fn(atlas.get("file_lines", {})) cfg = DetectConfig() cutoff = 1790402856356 post_flagged = [] for i in range(cfg.min_calls, len(calls) + 1, 5): subset = calls[:i] if len(subset) < cfg.min_calls: continue win = subset[-cfg.window:] if len(win) < cfg.min_calls: continue win_end = win[-1][0] flag, _ = evaluate(calls, cfg, fl_fn, atlas["title"], end_idx=i) if flag and win_end >= cutoff: post_flagged.append(i) assert len(post_flagged) >= 1, ( "Expected at least one flag after 02:07, got none" ) # --------------------------------------------------------------------------- # backtest output format # --------------------------------------------------------------------------- class TestOutputFormat: """Test that the backtest script produces correctly formatted output.""" def test_backtest_output_parseable(self): """All non-comment, non-empty output lines parse correctly.""" result, lines = _run_backtest(_FIXTURE_PATH) assert result.returncode == 0, f"Backtest failed: {result.stderr}" assert len(lines) == 15, f"Expected 15 session lines, got {len(lines)}" parsed = [_parse_line(l) for l in lines] assert all(p is not None for p in parsed), ( f"Some lines failed parsing. Sample: {lines[0] if lines else '(empty)'}" ) def test_flag_count_matches_fixture(self): """FLAG count equals must_flag count.""" sessions = _load_fixture() must_flag = len([s for s in sessions if s["label"] == "must_flag"]) expected_flags = must_flag _, lines = _run_backtest(_FIXTURE_PATH) parsed = [p for p in [_parse_line(l) for l in lines] if p] actual_flags = sum(1 for p in parsed if p["status"] == "flag") assert actual_flags == expected_flags, ( f"Expected {expected_flags} flagged, got {actual_flags}" ) def test_flagged_session_has_first_flag_at(self): """FLAG lines have a numeric first_flag_at; ok lines have None.""" _, lines = _run_backtest(_FIXTURE_PATH) parsed = [_parse_line(l) for l in lines if _parse_line(l)] for p in parsed: if p["status"] == "flag": assert p["first_flag_at"] is not None, ( f"FLAG line should have numeric first_flag_at: {p}" ) else: assert p["first_flag_at"] is None, ( f"ok line should have None first_flag_at: {p}" ) def test_call_counts_match_fixture(self): """Each line's calls= field matches the fixture call count.""" sessions = _load_fixture() sid_to_calls = {s["session_id"]: len(s["calls"]) for s in sessions} # We need to match lines to sessions — titles are unique title_to_sid = {} for s in sessions: title_to_sid[s["title"]] = s["session_id"] _, lines = _run_backtest(_FIXTURE_PATH) parsed = [_parse_line(l) for l in lines if _parse_line(l)] assert len(parsed) == len(sessions) for p in parsed: sid = title_to_sid.get(p["title"]) if sid: assert p["calls"] == sid_to_calls[sid], ( f"Call count mismatch for {sid}: expected {sid_to_calls[sid]}, got {p['calls']}" ) def test_output_includes_summary_in_stderr(self): """Stderr includes a summary line like '# backtest complete: N/M flagged'.""" result, _lines = _run_backtest(_FIXTURE_PATH) summary_re = re.compile(r"# backtest complete: \d+/\d+ flagged") assert summary_re.search(result.stderr), ( f"Expected summary in stderr, got: {result.stderr}" ) # --------------------------------------------------------------------------- # empty fixture edge case # --------------------------------------------------------------------------- class TestEmptyFixture: """Test edge cases with empty or minimal fixture data.""" def test_empty_fixture_file(self, tmp_path): """Empty fixture produces zero session lines.""" fixture_file = tmp_path / "empty_fixture.json" fixture_file.write_text("[]") result, lines = _run_backtest(str(fixture_file)) assert result.returncode == 0 # Only the summary line should remain after filtering assert len(lines) == 0, f"Expected 0 lines for empty fixture, got: {lines}" def test_fixture_with_too_few_calls(self, tmp_path): """Fixture with sessions under min_calls produces all ok.""" fixture_file = tmp_path / "few_calls.json" # Build a minimal fixture with 20 calls (below min_calls=40) calls = [] for i in range(20): calls.append({ "t": 1790000000000 + i, "tool": "read", "args": {"filePath": f"/tmp/file_{i}.py"}, "landed": False, }) fixture_data = [{ "session_id": "ses_empty_test", "title": "Tiny session", "agent": "test", "calls": calls, "file_lines": {}, "label": "must_not_flag", }] fixture_file.write_text(json.dumps(fixture_data)) result, lines = _run_backtest(str(fixture_file)) assert result.returncode == 0 assert len(lines) == 1 parsed = _parse_line(lines[0]) assert parsed is not None assert parsed["status"] == "ok" assert parsed["first_flag_at"] is None