feat: agent guardrails (verify_commit, dispatch audit, opencode guardrails plugin) #110

Merged
alee merged 45 commits from feat/agent-guardrails into main 2026-10-05 01:31:12 +00:00
25 changed files with 8511 additions and 0 deletions

View File

@@ -51,6 +51,22 @@ happened on 2026-09-28: a worker rewrote the main checkout's live
- Before ticking a todo, confirm its commit exists: - Before ticking a todo, confirm its commit exists:
`git -C <worktree> log --oneline`. `git -C <worktree> log --oneline`.
Additionally:
- A worker, after committing, runs
`python3 scripts/verify_commit.py --repo <worktree> HEAD` and pastes its
output before reporting done.
- An orchestrator, before reporting a wave done, runs
`scripts/oc_dispatch_audit.py <session> --worktree <path>` and pastes its
output.
- A guardrail block is a rule, not an error to route around: fix the call,
do not work around the plugin.
- A test of code that calls an external service (Ollama, a provider, the
opencode API) fakes it at the HTTP boundary (`requests.post`, `fetch`),
never by replacing the project's own wrapper. On 2026-09-29 the
local_decision dispatcher bug (category names passed where option letters
were expected) passed tests that faked `classify_choice` and failed every
live call.
The Git hygiene section below says why the main checkout is shared. The Git hygiene section below says why the main checkout is shared.
## Stack snapshot ## Stack snapshot

View File

@@ -355,6 +355,11 @@ rather than from months of history.
- `watchdog_store.py` — `watchdog_ticks`, `watchdog_verdicts`, `watchdog_alerts`, `watchdog_channel_settings` (four tables + indexes). See [watchdog](docs/watchdog.md). - `watchdog_store.py` — `watchdog_ticks`, `watchdog_verdicts`, `watchdog_alerts`, `watchdog_channel_settings` (four tables + indexes). See [watchdog](docs/watchdog.md).
- `router-link.js` — the opencode plugin; exported as a factory with `parentCache` as a property, because opencode 1.18.x rejects the whole plugin when any export is not a function. See [watchdog](docs/watchdog.md). - `router-link.js` — the opencode plugin; exported as a factory with `parentCache` as a property, because opencode 1.18.x rejects the whole plugin when any export is not a function. See [watchdog](docs/watchdog.md).
- `tests/` — 2414 tests across 108 files, offline, verified on Python 3.10 and 3.14. [README](README.md). - `tests/` — 2414 tests across 108 files, offline, verified on Python 3.10 and 3.14. [README](README.md).
- Agent guardrails — `scripts/verify_commit.py` (is this commit good?),
`scripts/oc_dispatch_audit.py` (audit an orchestrator session's dispatches),
and the opencode plugin `deploy/opencode-plugin/guardrails.js` whose
`tool.execute.before` hook blocks rule-breaking tool calls.
[agent-guardrails](docs/agent-guardrails.md).
## #45 — OpenRouter is an opt-in allowlist provider ## #45 — OpenRouter is an opt-in allowlist provider

View File

@@ -0,0 +1,282 @@
import { readFileSync, existsSync, mkdirSync, writeFileSync, rmSync, openSync, fsyncSync, closeSync } from "node:fs";
import { join } from "node:path";
import { mkdtempSync } from "node:fs";
import os from "node:os";
import { Guardrails } from "./guardrails.js";
/**
* Guardrails Replay CLI
*
* Replays a set of tool calls through the guardrails engine to calibrate rules.
*
* Usage:
* node guardrails-replay.mjs [--fixture FILE] [--url URL] [--directory DIR] [--limit N] [--assume-in-scope WT]
*/
/** Rule IDs that go in the "Always-Scope" table (always fire regardless of boulder). */
const ALWAYS_RULES = new Set(["bash_banned", "bash_protected_port"]);
/** Flush a file path to ensure disk sync. */
function _fsync(path) {
try {
const fd = openSync(path, "r");
fsyncSync(fd);
closeSync(fd);
} catch { /* non-fatal */ }
}
/**
* Extract worktree path from a prompt string.
*
* Parses the first `WORKTREE: <path>` line. Path is the first whitespace-delimited
* token after `WORKTREE: `, with one trailing `.` stripped.
* Returns null if no WORKTREE line is found.
*/
function parseWorktreePath(prompt) {
if (typeof prompt !== "string") return null;
const m = prompt.match(/WORKTREE:\s*(\S+)/m);
if (!m) return null;
let path = m[1];
if (path.endsWith(".")) path = path.slice(0, -1);
return path;
}
async function main() {
const args = process.argv.slice(2);
const options = {
fixture: null,
url: null,
directory: process.cwd(),
limit: null,
assumeInScope: null,
listAll: false,
};
for (let i = 0; i < args.length; i++) {
if (args[i] === "--fixture") options.fixture = args[++i];
else if (args[i] === "--url") options.url = args[++i];
else if (args[i] === "--directory") options.directory = args[++i];
else if (args[i] === "--limit") options.limit = parseInt(args[++i], 10);
else if (args[i] === "--assume-in-scope") options.assumeInScope = args[++i];
else if (args[i] === "--list-all") options.listAll = true;
}
const directory = options.directory;
// Create a temp dir for synthetic config/boulder — never under --directory.
const omoDir = mkdtempSync(join(os.tmpdir(), "guardrails-replay-"));
// 1. Setup a "report" config (all rules in block mode, no logging)
const reportConfig = {
rules: {
task_needs_agent: "block",
task_banned_agent: "block",
task_worktree_line: "block",
write_outside_worktree: "block",
bash_main_checkout: "block",
bash_banned: "block",
bash_protected_port: "block",
plan_tick_gate: "block",
},
protected_ports: [8080],
};
writeFileSync(join(omoDir, "guardrails.json"), JSON.stringify(reportConfig));
// 2. Load tool calls
let calls = [];
if (options.fixture) {
const raw = readFileSync(options.fixture, "utf-8");
calls = JSON.parse(raw);
} else if (options.url) {
const sessions = await fetchSessions(options.url, directory);
for (const s of sessions) {
const messages = await fetchMessages(options.url, s.id, directory);
for (const m of messages) {
if (m.parts) {
for (const part of m.parts) {
if (part.type === "tool" && part.state?.status === "completed") {
calls.push({
sessionID: s.id,
callID: part.callID || part.partID || part.id || `part_${s.id}_${part.index}`,
tool: part.tool,
args: part.state?.input || {},
prompt: part.state?.input?.prompt,
});
}
}
}
}
}
}
if (options.limit) calls = calls.slice(0, options.limit);
// 3. Build per-session boulders.
// Group calls by sessionID. Each session gets its own boulder with
// worktree_path derived from its first WORKTREE: line (or --assume-in-scope).
const sessionCalls = {};
for (const call of calls) {
const sid = call.sessionID;
if (!sessionCalls[sid]) sessionCalls[sid] = [];
sessionCalls[sid].push(call);
}
const sessionBoulders = {};
for (const [sid, sessionCallsList] of Object.entries(sessionCalls)) {
// Find first WORKTREE: line from any call in this session.
let wtPath = null;
for (const c of sessionCallsList) {
wtPath = parseWorktreePath(c.prompt ?? "");
if (wtPath) break;
}
// Fallback: --assume-in-scope.
if (!wtPath) {
wtPath = options.assumeInScope;
}
sessionBoulders[sid] = {
status: "active",
session_ids: [sid],
worktree_path: wtPath || null,
};
}
// Write all session boulders to omoDir *before* Guardrails reads them.
for (const [sid, boulder] of Object.entries(sessionBoulders)) {
const boulderPath = join(omoDir, `boulder-${sid}.json`);
writeFileSync(boulderPath, JSON.stringify(boulder));
}
try {
// 4. Replay — create a fresh Guardrails factory per session so readBoulder
// picks up that session's boulder from disk. Group by err.ruleId:
// Always table = bash_banned + bash_protected_port
// (B) table = the other six
const stats = {
always: {},
scoped: {},
};
const blockedCalls = [];
const client = {
session: {
get: async ({ path }) => ({ data: { id: path.id, parentID: null } }),
},
};
for (const [sid, sessionCallList] of Object.entries(sessionCalls)) {
const boulder = sessionBoulders[sid];
// Write this session's boulder to omoDir *before* creating the factory.
const boulderPath = join(omoDir, "boulder.json");
writeFileSync(boulderPath, JSON.stringify(boulder));
_fsync(boulderPath);
// Create a fresh Guardrails factory so readBoulder picks up this boulder.
const { "tool.execute.before": hook } = await Guardrails({
client,
directory,
omoDir: omoDir,
});
for (const call of sessionCallList) {
const input = { sessionID: sid, tool: call.tool, callID: call.callID };
const output = { args: { ...call.args } };
try {
await hook(input, output);
// Check if prompt was rewritten by task_worktree_line
if (call.tool === "task" && output.args?.prompt !== call.args?.prompt) {
updateStat("task_worktree_line", "rewrite", stats);
}
} catch (err) {
const ruleId = err.ruleId;
if (!ruleId) {
console.error(`Unexpected error during replay of ${call.callID}:`, err);
continue;
}
const type = "block";
updateStat(ruleId, type, stats);
if (options.listAll) {
const content = call.args?.command || call.args?.description || "";
const truncated = content.length > 160 ? content.slice(0, 160) : content;
blockedCalls.push(`${ruleId} | ${call.tool} | ${truncated}`);
}
// Collect examples (max 5 per rule)
const bucket = getBucket(ruleId, stats);
if (!bucket.examples) bucket.examples = [];
if (bucket.examples.length < 5) {
bucket.examples.push(JSON.stringify(call.args));
}
}
}
}
// 6. Print Results
const printTable = (title, data) => {
console.log(`\n${title}`);
console.log("Rule | Blocked | Rewritten | Examples");
console.log("-----|----------|-----------|----------");
for (const [id, s] of Object.entries(data)) {
console.log(`${id} | ${s.block} | ${s.rewrite || 0} | ${s.examples ? s.examples.join("; ") : ""}`);
}
};
printTable("Always-Scope Rules", stats.always);
printTable("(B) Scoped Rules", stats.scoped);
if (options.listAll && blockedCalls.length > 0) {
console.log("\nBlocked Calls Detail");
console.log("Rule | Tool | Command/Description");
console.log("----|------|---------------------");
for (const line of blockedCalls) {
console.log(line);
}
}
} finally {
rmSync(omoDir, { recursive: true, force: true });
}
}
/** Get the stat bucket for a ruleId based on grouping. */
function getBucket(ruleId, stats) {
const bucket = ALWAYS_RULES.has(ruleId)
? stats.always
: stats.scoped;
if (!bucket[ruleId]) {
bucket[ruleId] = { block: 0, rewrite: 0, examples: [] };
}
return bucket[ruleId];
}
/** Increment a stat counter. */
function updateStat(ruleId, type, stats) {
const bucket = getBucket(ruleId, stats);
if (type === "block") bucket.block++;
if (type === "rewrite") bucket.rewrite++;
}
/** Fetch sessions from the opencode API. */
async function fetchSessions(url, directory) {
const params = new URLSearchParams({ directory });
const resp = await fetch(`${url}/session?${params}`);
if (!resp.ok) throw new Error(`GET ${url}/session failed: ${resp.status} ${resp.statusText}`);
return resp.json();
}
/** Fetch messages for a session from the opencode API. */
async function fetchMessages(url, sessionId, directory) {
const params = new URLSearchParams({ directory });
const resp = await fetch(`${url}/session/${sessionId}/message?${params}`);
if (!resp.ok) throw new Error(`GET ${url}/session/${sessionId}/message failed: ${resp.status} ${resp.statusText}`);
return resp.json();
}
main().catch(console.error);

View File

@@ -0,0 +1,468 @@
import { describe, it } from "node:test";
import assert from "node:assert/strict";
import { writeFileSync, mkdirSync, rmSync, existsSync, readFileSync, readdirSync } from "node:fs";
import { join } from "node:path";
import { execFileSync, execFile } from "node:child_process";
import { mkdtempSync } from "node:fs";
import os from "node:os";
import { fileURLToPath } from "node:url";
import { dirname } from "node:path";
import http from "node:http";
import { promisify } from "node:util";
const __filename = fileURLToPath(import.meta.url);
const __dirname = dirname(__filename);
const execFileAsync = promisify(execFile);
const PYTHON3 = execFileSync("which", ["python3"], { encoding: "utf-8" }).trim();
// ---------------------------------------------------------------------------
// Fixture mode tests (unchanged)
// ---------------------------------------------------------------------------
describe("guardrails-replay CLI", () => {
const FIXTURE_PATH = join(__dirname, "replay-fixture.json");
it("should summarize violations from fixture", async () => {
const dir = mkdtempSync(join(os.tmpdir(), "replay-test-"));
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
const output = execFileSync(
"node",
[join(__dirname, "guardrails-replay.mjs"), "--fixture", FIXTURE_PATH, "--assume-in-scope", wt],
{
cwd: dir,
encoding: "utf-8",
env: { ...process.env, NODE_PATH: __dirname },
},
);
assert.ok(output.includes("bash_banned"), "Should report bash_banned");
assert.ok(output.includes("bash_protected_port"), "Should report bash_protected_port");
assert.ok(output.includes("task_banned_agent"), "Should report task_banned_agent");
});
it("should correctly identify rewrites for task_worktree_line", async () => {
const dir = mkdtempSync(join(os.tmpdir(), "replay-test-"));
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
const output = execFileSync(
"node",
[join(__dirname, "guardrails-replay.mjs"), "--fixture", FIXTURE_PATH, "--assume-in-scope", wt],
{
cwd: dir,
encoding: "utf-8",
env: { ...process.env, NODE_PATH: __dirname },
},
);
assert.ok(output.includes("task_worktree_line"), "Should report task_worktree_line");
});
});
// ---------------------------------------------------------------------------
// URL mode tests (local node:http server)
// ---------------------------------------------------------------------------
describe("guardrails-replay CLI — URL mode", () => {
/**
* Build a local HTTP server that responds with the opencode API endpoint shapes.
*/
function createServer(fixture) {
return new Promise((resolve, reject) => {
const server = http.createServer((req, res) => {
const url = new URL(req.url, `http://${req.headers.host}`);
if (url.pathname === "/session" && req.method === "GET") {
res.writeHead(200, { "Content-Type": "application/json" });
res.end(JSON.stringify(fixture.sessions));
return;
}
const messageMatch = url.pathname.match(/^\/session\/([^/]+)\/message$/);
if (messageMatch && req.method === "GET") {
const sessionId = decodeURIComponent(messageMatch[1]);
const msgs = fixture.messagesBySession[sessionId] || [];
res.writeHead(200, { "Content-Type": "application/json" });
res.end(JSON.stringify(msgs));
return;
}
res.writeHead(404);
res.end("not found");
});
server.listen(0, "127.0.0.1", () => {
const addr = server.address();
resolve({ server, url: `http://127.0.0.1:${addr.port}` });
});
server.on("error", reject);
});
}
it("should replay tool calls from URL endpoints", async () => {
const dir = mkdtempSync(join(os.tmpdir(), "replay-test-url-"));
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
const fixture = {
sessions: [
{ id: "ses_url_001", directory: wt, updatedAt: "2026-01-01T00:00:00Z" },
{ id: "ses_url_002", directory: wt, updatedAt: "2026-01-01T00:01:00Z" },
],
messagesBySession: {
"ses_url_001": [
{
parts: [
{
type: "tool",
tool: "task",
callID: "call_1",
state: {
status: "completed",
input: { prompt: "do X", category: "quick", subagent_type: "oh-my-claudecode:oracle" },
},
},
],
},
],
"ses_url_002": [
{
parts: [
{
type: "tool",
tool: "bash",
callID: "call_2",
state: {
status: "completed",
input: { command: "git push origin main" },
},
},
{
type: "tool",
tool: "bash",
callID: "call_3",
state: {
status: "completed",
input: { command: "curl localhost:8080/health" },
},
},
],
},
],
},
};
const { server, url } = await createServer(fixture);
try {
const { stdout } = await execFileAsync(
"node",
[
join(__dirname, "guardrails-replay.mjs"),
"--url", url,
"--directory", dir,
"--assume-in-scope", wt,
"--limit", "10",
],
{
cwd: __dirname,
timeout: 15000,
},
);
// Verify per-rule counts
assert.ok(stdout.includes("task_banned_agent"), "Should report task_banned_agent from URL replay");
assert.ok(stdout.includes("bash_banned"), "Should report bash_banned from URL replay");
assert.ok(stdout.includes("bash_protected_port"), "Should report bash_protected_port from URL replay");
// Verify counts match fixture expectations
const taskBanLine = stdout.split("\n").find((l) => l.includes("task_banned_agent"));
assert.ok(taskBanLine.includes("1"), `task_banned_agent should have 1 block, got: ${taskBanLine}`);
const bashBanLine = stdout.split("\n").find((l) => l.includes("bash_banned"));
assert.ok(bashBanLine.includes("1"), `bash_banned should have 1 block, got: ${bashBanLine}`);
const portLine = stdout.split("\n").find((l) => l.includes("bash_protected_port"));
assert.ok(portLine.includes("1"), `bash_protected_port should have 1 block, got: ${portLine}`);
} finally {
server.close();
}
});
it("should handle empty sessions from URL", async () => {
const dir = mkdtempSync(join(os.tmpdir(), "replay-test-url-empty-"));
const server = http.createServer((req, res) => {
res.writeHead(200, { "Content-Type": "application/json" });
res.end("[]");
});
await new Promise((resolve) => {
server.listen(0, "127.0.0.1", resolve);
});
try {
const { stdout } = await execFileAsync(
"node",
[
join(__dirname, "guardrails-replay.mjs"),
"--url", `http://127.0.0.1:${server.address().port}`,
"--directory", dir,
],
{
cwd: __dirname,
timeout: 15000,
},
);
// Should succeed with output containing the table headers
assert.ok(stdout.includes("Rule"));
} finally {
server.close();
}
});
it("should use --directory for the config file path", async () => {
const dir = mkdtempSync(join(os.tmpdir(), "replay-test-dir-"));
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
// Verify the fixture-mode uses the specified directory
const output = execFileSync(
"node",
[
join(__dirname, "guardrails-replay.mjs"),
"--fixture", join(__dirname, "replay-fixture.json"),
"--directory", dir,
"--assume-in-scope", wt,
],
{
cwd: __dirname,
encoding: "utf-8",
},
);
assert.ok(output.includes("bash_banned"), "Should work with --directory");
});
});
// ---------------------------------------------------------------------------
// R19 — defect t (redirect exemption) + defect u (temp dir cleanup) + R15 tests
// ---------------------------------------------------------------------------
describe("guardrails-replay R19 tests", () => {
const R19_FIXTURE = join(__dirname, "replay-fixture-r19.json");
it("defect t: redirect exemption in checkBashMainCheckout uses .omo not base", () => {
// This is a code-level assertion: the fix must reference join(directory, ".omo")
// not `base`. We verify by reading the source.
const src = readFileSync(join(__dirname, "guardrails.js"), "utf-8");
const redirectExempt = src.match(/!_isInside\(target,\s*join\(directory,\s*".omo"\)\)/);
assert.ok(redirectExempt, "Redirect exemption should use join(directory, '.omo') not base");
});
it("defect u: replay temp omoDir is deleted after script completes", async () => {
const dir = mkdtempSync(join(os.tmpdir(), "replay-test-cleanup-"));
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
const output = execFileSync(
"node",
[
join(__dirname, "guardrails-replay.mjs"),
"--fixture", R19_FIXTURE,
"--directory", dir,
"--assume-in-scope", wt,
],
{
cwd: __dirname,
encoding: "utf-8",
},
);
assert.ok(output.includes("bash_banned"), "Should report bash_banned");
// Verify NO guardrails-replay-* dirs remain in /tmp
const tmpFiles = readdirSync(os.tmpdir());
const leftover = tmpFiles.filter((f) => f.startsWith("guardrails-replay-"));
assert.equal(leftover.length, 0, `Temp omoDir should be cleaned up, but found: ${leftover}`);
});
it("R15 #1: --directory is read-only (omo files unchanged, no new files)", () => {
const dir = mkdtempSync(join(os.tmpdir(), "replay-test-readonly-"));
const omoDir = join(dir, ".omo");
mkdirSync(omoDir, { recursive: true });
const knownBoulder = JSON.stringify({ status: "active", session_ids: ["test"], worktree_path: dir });
const knownConfig = JSON.stringify({ rules: { bash_banned: "block" }, protected_ports: [8080] });
writeFileSync(join(omoDir, "boulder.json"), knownBoulder);
writeFileSync(join(omoDir, "guardrails.json"), knownConfig);
// Record original content for byte-identity check
const originalBoulder = readFileSync(join(omoDir, "boulder.json"));
const originalConfig = readFileSync(join(omoDir, "guardrails.json"));
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
// Record existing structure before replay (includes .omo/ and wt/)
const existingFiles = new Set(readdirSync(dir, { recursive: true }));
const output = execFileSync(
"node",
[
join(__dirname, "guardrails-replay.mjs"),
"--fixture", join(__dirname, "replay-fixture.json"),
"--directory", dir,
"--assume-in-scope", wt,
],
{
cwd: __dirname,
encoding: "utf-8",
},
);
// .omo files must be byte-identical to originals
assert.strictEqual(
readFileSync(join(omoDir, "boulder.json")).toString(),
originalBoulder.toString(),
"boulder.json must be byte-identical",
);
assert.strictEqual(
readFileSync(join(omoDir, "guardrails.json")).toString(),
originalConfig.toString(),
"guardrails.json must be byte-identical",
);
// No new files were created under the directory by the replay
const afterFiles = readdirSync(dir, { recursive: true });
for (const f of afterFiles) {
assert.ok(existingFiles.has(f), `No new files: ${f} was not present before replay`);
}
});
it("R15 #2: one call per rule id lands in the right table", () => {
// Use a fixed directory so the fixture paths are predictable.
const fixedDir = "/tmp/r19-test-rules-wt";
mkdirSync(fixedDir, { recursive: true });
const wt = join(fixedDir, "wt");
mkdirSync(wt, { recursive: true });
try {
const output = execFileSync(
"node",
[
join(__dirname, "guardrails-replay.mjs"),
"--fixture", R19_FIXTURE,
"--directory", fixedDir,
"--assume-in-scope", wt,
],
{
cwd: __dirname,
encoding: "utf-8",
},
);
// Always table: bash_banned + bash_protected_port
const alwaysSection = output.split("Always-Scope Rules")[1] || "";
const firstSectionEnd = alwaysSection.indexOf("(B)");
const alwaysBlock = firstSectionEnd > 0 ? alwaysSection.slice(0, firstSectionEnd) : alwaysSection;
assert.ok(alwaysBlock.includes("bash_banned"), "Always table should contain bash_banned");
assert.ok(alwaysBlock.includes("bash_protected_port"), "Always table should contain bash_protected_port");
// (B) table: scoped rules that fire from the fixture.
// plan_tick_gate requires a plan file with ticks on disk (not available in replay).
const scopedSection = output.split("(B) Scoped Rules")[1] || "";
const scopedRules = ["task_needs_agent", "task_banned_agent", "task_worktree_line",
"write_outside_worktree"];
for (const ruleId of scopedRules) {
assert.ok(scopedSection.includes(ruleId), `(B) table should contain ${ruleId}`);
}
} finally {
rmSync(fixedDir, { recursive: true, force: true });
}
});
it("R15 #3: per-session scope — two sessions, different worktrees", async () => {
const wtA = mkdtempSync(join(os.tmpdir(), "replay-wtA-"));
const wtB = mkdtempSync(join(os.tmpdir(), "replay-wtB-"));
// Two sessions: each with a WORKTREE line naming its own worktree
// Both in-scope via --assume-in-scope
const fixture = [
{
sessionID: "ses_scope_a",
callID: "call_a_1",
tool: "task",
args: {
category: "quick",
prompt: `WORKTREE: ${wtA}. cd there first; never edit under /home/alee/Sources/6krrt/.\\nAnalyze.`,
},
},
{
sessionID: "ses_scope_a",
callID: "call_a_2",
tool: "bash",
args: { command: "git push origin foo" },
},
{
sessionID: "ses_scope_b",
callID: "call_b_1",
tool: "task",
args: {
category: "quick",
prompt: `WORKTREE: ${wtB}. cd there first; never edit under /home/alee/Sources/6krrt/.\\nAnalyze.`,
},
},
{
sessionID: "ses_scope_b",
callID: "call_b_2",
tool: "bash",
args: { command: "git push origin bar" },
},
];
const fixturePath = join(os.tmpdir(), "replay-fixture-scope.json");
writeFileSync(fixturePath, JSON.stringify(fixture));
try {
const dir = mkdtempSync(join(os.tmpdir(), "replay-test-scope-"));
try {
const output = execFileSync(
"node",
[
join(__dirname, "guardrails-replay.mjs"),
"--fixture", fixturePath,
"--directory", dir,
"--assume-in-scope", wtA,
],
{
cwd: __dirname,
encoding: "utf-8",
},
);
// Neither session's bash calls should be blocked — both are in-scope
// (assume-in-scope makes all calls in-scope for bash_banned check)
const bashBanLine = output.split("\n").find((l) => l.includes("bash_banned"));
if (bashBanLine) {
const match = bashBanLine.match(/\| (\d+)/);
if (match) {
const blocked = parseInt(match[1], 10);
// At most 2 blocks (one per session's bash call)
assert.ok(blocked <= 2, `Expected at most 2 blocks, got ${blocked}`);
}
}
} finally {
rmSync(dir, { recursive: true, force: true });
}
} finally {
rmSync(wtA, { recursive: true, force: true });
rmSync(wtB, { recursive: true, force: true });
try { rmSync(fixturePath, { force: true }); } catch { /* ignore */ }
}
});
});

View File

@@ -0,0 +1,829 @@
/**
* Contract tests for guardrails.js
*
* Drives the plugin with opencode's real hook shapes: (input, output) where
* input has {tool, sessionID, callID} and output has {args}. Builds a client
* whose session.get returns the real key set, writes a real .omo/guardrails.json
* (based on guardrails.example.json) and real .omo/boulder.json into a temp dir.
*
* Acceptance probe — every rule at default, real hook shape.
* Per-rule — one positive (blocks) and one negative (allows) case each.
* Only the tick gate test may use the real scripts/verify_commit.py.
*/
import { describe, it } from "node:test";
import assert from "node:assert/strict";
import {
copyFileSync,
writeFileSync,
readFileSync,
existsSync,
mkdirSync,
} from "node:fs";
import { join } from "node:path";
import { mkdtempSync } from "node:fs";
import os from "node:os";
import { execFile, execFileSync } from "node:child_process";
import { promisify } from "node:util";
import { fileURLToPath } from "node:url";
import { Guardrails } from "./guardrails.js";
const execFileAsync = promisify(execFile);
const __dirname = fileURLToPath(new URL(".", import.meta.url));
// ---------------------------------------------------------------------------
// Constants — real guardrails.example.json content (Design C defaults)
// ---------------------------------------------------------------------------
/** Mirrors the default guardrails.example.json shipped in deploy/. */
const EXAMPLE_CONFIG = {
rules: {
task_needs_agent: "block",
task_banned_agent: "block",
task_worktree_line: "block",
write_outside_worktree: "block",
bash_main_checkout: "block",
bash_banned: "block",
bash_protected_port: "block",
plan_tick_gate: "block",
},
log: ".omo/guardrails.log",
protected_ports: [8080],
tick_gate: {
cmd: [
"python3",
"{directory}/scripts/verify_commit.py",
"--repo",
"{worktree}",
"--require-clean",
"HEAD",
],
timeout_s: 900,
},
};
const SCRIPT_DIR = join(__dirname, "..", "..", "scripts");
/** Full path to python3 — required by tick_gate because guardrails checks
* existsSync on resolvedArgs[0], which must be a file path, not a PATH command. */
const PYTHON3 = execFileSync("which", ["python3"], { encoding: "utf-8" }).trim();
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
function newDir() {
return mkdtempSync(join(os.tmpdir(), "guardrails-contract-"));
}
function writeConfig(dir, configObj) {
const cfgDir = join(dir, ".omo");
mkdirSync(cfgDir, { recursive: true });
writeFileSync(join(cfgDir, "guardrails.json"), JSON.stringify(configObj));
return cfgDir;
}
function writeBoulder(dir, boulderObj) {
const boulderDir = join(dir, ".omo");
if (!existsSync(boulderDir)) {
mkdirSync(boulderDir, { recursive: true });
}
writeFileSync(join(boulderDir, "boulder.json"), JSON.stringify(boulderObj));
}
/**
* Build a client whose session.get returns the real key set.
* The guardrails plugin reads result.data.parentID for scope walk.
*/
function realClient(extraData) {
return {
session: {
get: async ({ path }) => ({
data: {
agent: "Sisyphus-ultraworker",
cost: 0.12,
directory: "/home/test/project",
id: path?.id ?? "ses_contract",
model: "gpt-5.6",
path: "/home/test/project/main",
projectID: "6krrt",
slug: "6krrt",
summary: "Refactor router dispatch",
time: "2026-10-04T12:00:00Z",
title: "R6 — contract test",
tokens: 4821,
version: "2.0.0",
...extraData,
},
}),
},
};
}
/**
* Drive the hook with opencode's real shape:
* await hooks["tool.execute.before"]({tool, sessionID, callID}, {args})
*
* All tests use this function — never build input.args.
* The {args} object is always on the OUTPUT parameter.
*/
async function hookDrive(hooks, tool, sessionID, args) {
return hooks["tool.execute.before"](
{ tool, sessionID, callID: "call_contract" },
{ args },
);
}
// ---------------------------------------------------------------------------
// Loader contract
// ---------------------------------------------------------------------------
describe("loader contract", () => {
it("every named export from guardrails.js is a function", async () => {
const mod = await import("./guardrails.js");
for (const [name, value] of Object.entries(mod)) {
assert.equal(
typeof value,
"function",
`export "${name}" must be a function`,
);
}
});
});
// ===========================================================================
// Acceptance probes — every rule at default (block), real hook shape
// ===========================================================================
describe("Acceptance probes (every rule at default, real hook shape)", () => {
const SESSION = "ses_accept";
it("task with no category, subagent_type or task_id → BLOCK", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null }),
directory: dir,
});
await assert.rejects(
hookDrive(hooks, "task", SESSION, {}),
{ message: /task\/call_omo_agent needs category or subagent_type \(or task_id for resume\)/ },
);
});
it("task with subagent_type oh-my-claudecode:writer → BLOCK", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
await assert.rejects(
hookDrive(hooks, "task", SESSION, {
category: "quick",
subagent_type: "oh-my-claudecode:writer",
prompt: "hello",
}),
{ message: /subagent_type must not start with oh-my-claudecode/ },
);
});
it("bash git push origin x → BLOCK (bash_banned)", async () => {
const dir = newDir();
writeConfig(dir, EXAMPLE_CONFIG);
const hooks = await Guardrails({ client: realClient(), directory: dir });
await assert.rejects(
hookDrive(hooks, "bash", SESSION, { command: "git push origin x" }),
{ message: /bash\/banned: blocked/ },
);
});
it("bash git stash → BLOCK (bash_banned)", async () => {
const dir = newDir();
writeConfig(dir, EXAMPLE_CONFIG);
const hooks = await Guardrails({ client: realClient(), directory: dir });
await assert.rejects(
hookDrive(hooks, "bash", SESSION, { command: "git stash" }),
{ message: /bash\/banned: blocked/ },
);
});
it("bash curl -s localhost:8080/health → BLOCK (bash_protected_port)", async () => {
const dir = newDir();
writeConfig(dir, EXAMPLE_CONFIG);
const hooks = await Guardrails({ client: realClient(), directory: dir });
await assert.rejects(
hookDrive(hooks, "bash", SESSION, {
command: "curl -s localhost:8080/health",
}),
{ message: /bash\/protected_port: blocked/ },
);
});
it("bash ls → ALLOW", async () => {
const dir = newDir();
writeConfig(dir, EXAMPLE_CONFIG);
const hooks = await Guardrails({ client: realClient(), directory: dir });
await hookDrive(hooks, "bash", SESSION, { command: "ls" });
// No throw → ALLOW
});
it("task with category:quick and correct WORKTREE line → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
// Test multiple valid forms of the WORKTREE line
const validLines = [
`WORKTREE: ${wt}. cd there first; never edit under ${dir}/.`,
`WORKTREE: ${wt}. cd there`,
`WORKTREE: ${wt} -- cd there`,
`WORKTREE: ${wt}`,
];
for (const line of validLines) {
const prompt = line + "\nRefactor this function";
await hookDrive(hooks, "task", SESSION, {
category: "quick",
prompt,
});
}
});
it("task with dotted path in WORKTREE line → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt.v2");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
const correctLine = `WORKTREE: ${wt}. cd there first; never edit under ${dir}/.`;
await hookDrive(hooks, "task", SESSION, {
category: "quick",
prompt: correctLine + "\nRefactor this function",
});
});
it("task with no WORKTREE line → REWRITE (verify using hookDrive as a wrapper)", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
const args = { category: "quick", prompt: "hello" };
// hookDrive passes { args } as output, so the hook mutates args.prompt
await hooks["tool.execute.before"](
{ tool: "task", sessionID: SESSION, callID: "call_rewrite" },
{ args },
);
const expectedLine = `WORKTREE: ${wt}. cd there first; never edit under ${dir}/.`;
assert.equal(args.prompt, expectedLine + "\nhello");
});
it("task with WORKTREE line and warn mode → ALLOW and LOG", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, { ...EXAMPLE_CONFIG, rules: { task_worktree_line: "warn" } });
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
const badLine = `WORKTREE: /wrong/path. cd there first.`;
await hookDrive(hooks, "task", SESSION, {
category: "quick",
prompt: badLine + "\nhello",
});
const logContent = readFileSync(join(dir, ".omo/guardrails.log"), "utf-8");
assert.ok(logContent.includes("task_worktree_line"));
});
});
// ===========================================================================
// Per-rule positive (blocks) and negative (allows)
// ===========================================================================
describe("task_needs_agent (R1)", () => {
const SESSION = "ses_tna";
it("positive: bare task without category/subagent_type/task_id → BLOCK", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null }),
directory: dir,
});
await assert.rejects(
hookDrive(hooks, "task", SESSION, {}),
{ message: /task\/call_omo_agent needs category or subagent_type/ },
);
});
it("negative: task with category only → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
await hookDrive(hooks, "task", SESSION, {
category: "quick",
prompt: "hello",
});
});
it("negative: task with subagent_type only → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
await hookDrive(hooks, "task", SESSION, {
subagent_type: "explore",
prompt: "hello",
});
});
it("negative: task with task_id only → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
await hookDrive(hooks, "task", SESSION, {
task_id: "task_123",
prompt: "hello",
});
});
it("negative: task with category + subagent_type → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
await hookDrive(hooks, "task", SESSION, {
category: "quick",
subagent_type: "explore",
prompt: "hello",
});
});
});
describe("task_banned_agent (R2)", () => {
const SESSION = "ses_tba";
it("positive: oh-my-claudecode:writer subagent → BLOCK", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
await assert.rejects(
hookDrive(hooks, "task", SESSION, {
category: "quick",
subagent_type: "oh-my-claudecode:writer",
prompt: "hello",
}),
{ message: /subagent_type must not start with oh-my-claudecode/ },
);
});
it("negative: normal subagent_type → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
await hookDrive(hooks, "task", SESSION, {
category: "quick",
subagent_type: "build",
prompt: "hello",
});
});
});
describe("task_worktree_line (R3)", () => {
const SESSION = "ses_twl";
it("positive: WORKTREE line with wrong path → BLOCK", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
const badLine = `WORKTREE: /some/other/path. cd there first; never edit under /some/other/.`;
await assert.rejects(
hookDrive(hooks, "task", SESSION, {
category: "quick",
subagent_type: "explore",
prompt: badLine + "\nhello",
}),
{ message: /WORKTREE line path.*does not match/ },
);
});
it("negative: correct WORKTREE line in prompt → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null, id: SESSION }),
directory: dir,
});
const dirPath = wt.slice(0, wt.lastIndexOf("/")) || wt;
const correctLine =
`WORKTREE: ${wt}. cd there first; never edit under ${dirPath}/.`;
await hookDrive(hooks, "task", SESSION, {
category: "quick",
subagent_type: "build",
prompt: correctLine + "\nRefactor this function",
});
});
});
describe("write_outside_worktree (R4)", () => {
const SESSION = "ses_wow";
it("positive: write to directory outside worktree → BLOCK", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null }),
directory: dir,
});
await assert.rejects(
hookDrive(hooks, "write", SESSION, {
filePath: join(dir, "main.py"),
content: "hello",
}),
{ message: /write\/outside_worktree/ },
);
});
it("negative: write inside worktree → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null }),
directory: dir,
});
await hookDrive(hooks, "write", SESSION, {
filePath: join(wt, "a.py"),
content: "hello",
});
});
});
describe("bash_main_checkout (R5a)", () => {
const SESSION = "ses_bmc";
it("positive: git operation in main checkout → BLOCK", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null }),
directory: dir,
});
await assert.rejects(
hookDrive(hooks, "bash", SESSION, { command: "git add ." }),
{ message: /bash\/main_checkout/ },
);
});
it("negative: git -C worktree → ALLOW", async () => {
const dir = newDir();
const wt = join(dir, "wt");
mkdirSync(wt, { recursive: true });
writeConfig(dir, EXAMPLE_CONFIG);
writeBoulder(dir, {
status: "active",
worktree_path: wt,
session_ids: [`opencode:${SESSION}`],
});
const hooks = await Guardrails({
client: realClient({ parentID: null }),
directory: dir,
});
await hookDrive(hooks, "bash", SESSION, {
command: `git -C ${wt} add .`,
});
});
});
describe("bash_banned (R5b)", () => {
const SESSION = "ses_bb";
it("positive: banned git push → BLOCK", async () => {
const dir = newDir();
writeConfig(dir, EXAMPLE_CONFIG);
const hooks = await Guardrails({ client: realClient(), directory: dir });
await assert.rejects(
hookDrive(hooks, "bash", SESSION, { command: "git push origin main" }),
{ message: /bash\/banned: blocked/ },
);
});
it("negative: safe command ls → ALLOW", async () => {
const dir = newDir();
writeConfig(dir, EXAMPLE_CONFIG);
const hooks = await Guardrails({ client: realClient(), directory: dir });
await hookDrive(hooks, "bash", SESSION, { command: "ls -la" });
});
});
describe("bash_protected_port (R5c)", () => {
const SESSION = "ses_bpp";
it("positive: curl to localhost:8080 → BLOCK", async () => {
const dir = newDir();
writeConfig(dir, EXAMPLE_CONFIG);
const hooks = await Guardrails({ client: realClient(), directory: dir });
await assert.rejects(
hookDrive(hooks, "bash", SESSION, {
command: "curl -s http://127.0.0.1:8080/health",
}),
{ message: /bash\/protected_port/ },
);
});
it("negative: curl to non-protected port → ALLOW", async () => {
const dir = newDir();
writeConfig(dir, EXAMPLE_CONFIG);
const hooks = await Guardrails({ client: realClient(), directory: dir });
await hookDrive(hooks, "bash", SESSION, {
command: "curl -s http://127.0.0.1:9999/health",
});
});
});
// ===========================================================================
// Tick gate — real scripts/verify_commit.py (only test that may use it)
// ===========================================================================
describe("plan_tick_gate with real verify_commit.py", () => {
const SESSION = "ses_ptg";
/**
* Set up a minimal git repo in wtDir with a commit and a plan file.
*/
async function setupTickRepo(wtDir, planPath) {
await execFileAsync("git", ["-C", wtDir, "init"]);
await execFileAsync("git", ["-C", wtDir, "config", "user.email", "test@test.com"]);
await execFileAsync("git", ["-C", wtDir, "config", "user.name", "Test"]);
writeFileSync(join(wtDir, "README.md"), "# Test repo");
await execFileAsync("git", ["-C", wtDir, "add", "README.md"]);
await execFileAsync("git", ["-C", wtDir, "commit", "-m", "init"]);
mkdirSync(join(wtDir, ".omo"), { recursive: true });
writeFileSync(planPath, "- [ ] 1. do it");
// Commit plan so HEAD has a valid commit tree with it
await execFileAsync("git", ["-C", wtDir, "add", ".omo/PLAN.md"]);
await execFileAsync("git", ["-C", wtDir, "commit", "-m", "add plan"]);
}
it("allows tick when verify_commit passes (clean tree)", async () => {
const dir = newDir();
const wtDir = join(dir, "wt");
mkdirSync(wtDir, { recursive: true });
// Copy verify_commit.py so the default tick gate command resolves
const scriptsDest = join(dir, "scripts");
mkdirSync(scriptsDest, { recursive: true });
copyFileSync(
join(SCRIPT_DIR, "verify_commit.py"),
join(scriptsDest, "verify_commit.py"),
);
const planPath = join(wtDir, ".omo", "PLAN.md");
await setupTickRepo(wtDir, planPath);
// Temporarily override tick_gate to point at our copied script
writeConfig(dir, {
...EXAMPLE_CONFIG,
tick_gate: {
cmd: [
PYTHON3,
"{directory}/scripts/verify_commit.py",
"--repo",
"{worktree}",
"--require-clean",
"HEAD",
],
timeout_s: 30,
},
});
writeBoulder(dir, {
status: "active",
worktree_path: wtDir,
session_ids: [`opencode:${SESSION}`],
active_plan: planPath,
});
const hooks = await Guardrails({
client: realClient({ parentID: null }),
directory: dir,
});
// Tick the todo: - [ ] 1. do it → - [x] 1. do it
await hookDrive(hooks, "write", SESSION, {
filePath: planPath,
oldString: "- [ ] 1. do it",
newString: "- [x] 1. do it",
});
// No throw → ALLOW
});
it("blocks tick when verify_commit fails (dirty tree)", async () => {
const dir = newDir();
const wtDir = join(dir, "wt");
mkdirSync(wtDir, { recursive: true });
const scriptsDest = join(dir, "scripts");
mkdirSync(scriptsDest, { recursive: true });
copyFileSync(
join(SCRIPT_DIR, "verify_commit.py"),
join(scriptsDest, "verify_commit.py"),
);
const planPath = join(wtDir, ".omo", "PLAN.md");
await setupTickRepo(wtDir, planPath);
// Make the tree dirty (modify a tracked file without committing)
writeFileSync(join(wtDir, "README.md"), "# Dirty");
writeConfig(dir, {
...EXAMPLE_CONFIG,
tick_gate: {
cmd: [
PYTHON3,
"{directory}/scripts/verify_commit.py",
"--repo",
"{worktree}",
"--require-clean",
"HEAD",
],
timeout_s: 30,
},
});
writeBoulder(dir, {
status: "active",
worktree_path: wtDir,
session_ids: [`opencode:${SESSION}`],
active_plan: planPath,
});
const hooks = await Guardrails({
client: realClient({ parentID: null }),
directory: dir,
});
await assert.rejects(
hookDrive(hooks, "write", SESSION, {
filePath: planPath,
oldString: "- [ ] 1. do it",
newString: "- [x] 1. do it",
}),
{ message: /plan_tick_gate/ },
);
});
});

View File

@@ -0,0 +1,25 @@
{
"rules": {
"task_needs_agent": "block",
"task_banned_agent": "block",
"task_worktree_line": "block",
"write_outside_worktree": "block",
"bash_main_checkout": "block",
"bash_banned": "block",
"bash_protected_port": "block",
"plan_tick_gate": "block"
},
"log": ".omo/guardrails.log",
"protected_ports": [8080],
"tick_gate": {
"cmd": [
"python3",
"{directory}/scripts/verify_commit.py",
"--repo",
"{worktree}",
"--require-clean",
"HEAD"
],
"timeout_s": 900
}
}

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,64 @@
[
{
"sessionID": "ses_r19_001",
"callID": "call_r19_001",
"tool": "task",
"args": {
"prompt": "Analyze this codebase."
}
},
{
"sessionID": "ses_r19_002",
"callID": "call_r19_002",
"tool": "task",
"args": {
"category": "quick",
"subagent_type": "oh-my-claudecode:oracle",
"prompt": "Run analysis."
}
},
{
"sessionID": "ses_r19_003",
"callID": "call_r19_003",
"tool": "task",
"args": {
"category": "quick",
"prompt": "WORKTREE: /home/alee/Sources/6krrt-worktrees/agent-guardrails. cd there first; never edit under /home/alee/Sources/6krrt/.\\nAnalyze this codebase."
}
},
{
"sessionID": "ses_r19_004",
"callID": "call_r19_004",
"tool": "bash",
"args": {
"command": "git push origin main"
}
},
{
"sessionID": "ses_r19_005",
"callID": "call_r19_005",
"tool": "bash",
"args": {
"command": "curl -s localhost:8080/health"
}
},
{
"sessionID": "ses_r19_007",
"callID": "call_r19_007",
"tool": "edit",
"args": {
"filePath": "/tmp/r19-test-rules-wt/src/test.py",
"oldString": "old",
"newString": "new"
}
},
{
"sessionID": "ses_r19_008",
"callID": "call_r19_008",
"tool": "task",
"args": {
"category": "plan",
"prompt": "tick"
}
}
]

View File

@@ -0,0 +1,89 @@
[
{
"sessionID": "opencode:12345",
"callID": "call1",
"tool": "task",
"args": {
"category": "quick",
"prompt": "WORKTREE: /home/alee/Sources/6krrt-worktrees/agent-guardrails. cd there first; never edit under /home/alee/Sources/6krrt/.\\nRefactor this code to improve performance."
}
},
{
"sessionID": "opencode:12346",
"callID": "call2",
"tool": "task",
"args": {
"category": "quick",
"subagent_type": "oh-my-claudecode:writer",
"prompt": "Write a function to calculate the factorial of a number."
}
},
{
"sessionID": "opencode:12347",
"callID": "call3",
"tool": "task",
"args": {
"prompt": "Summarize this document."
}
},
{
"sessionID": "opencode:12348",
"callID": "call4",
"tool": "bash",
"args": {
"command": "git push origin x"
}
},
{
"sessionID": "opencode:12349",
"callID": "call5",
"tool": "bash",
"args": {
"command": "git stash"
}
},
{
"sessionID": "opencode:12350",
"callID": "call6",
"tool": "bash",
"args": {
"command": "curl -s localhost:8080/health"
}
},
{
"sessionID": "opencode:12351",
"callID": "call7",
"tool": "bash",
"args": {
"command": "echo ok; git push"
}
},
{
"sessionID": "opencode:12352",
"callID": "call8",
"tool": "bash",
"args": {
"command": "ls -la"
}
},
{
"sessionID": "opencode:12353",
"callID": "call9",
"tool": "edit",
"args": {
"filePath": "/home/alee/Sources/6krrt/src/foo.py",
"oldString": "old",
"newString": "new"
}
},
{
"sessionID": "opencode:12354",
"callID": "call10",
"tool": "edit",
"args": {
"filePath": "/home/alee/Sources/6krrt-worktrees/agent-guardrails/src/bar.py",
"oldString": "old",
"newString": "new"
}
}
]

48
docs/agent-guardrails.md Normal file
View File

@@ -0,0 +1,48 @@
# Agent guardrails
An opencode plugin (`deploy/opencode-plugin/guardrails.js`) that intercepts
tool.execute calls and blocks or warns on patterns that have caused expensive
failures on this repo.
## Install
```
command cp -f deploy/opencode-plugin/guardrails.js ~/.config/opencode/plugins/
cp -n deploy/opencode-plugin/guardrails.example.json .omo/guardrails.json
```
Restart opencode. Check `~/.local/share/opencode/log/opencode.log` for
`failed to load plugin`.
## Uninstall
Delete `.omo/guardrails.json` (plugin goes inert) or remove
`guardrails.js` from `~/.config/opencode/plugins/`.
## Rule modes
Each rule can be `"block"` (default), `"warn"` (logs only, allows the
call), or `"off"` (no check). Edit in `.omo/guardrails.json`.
## Rules
| Rule | Blocks | Evidence |
|---|---|---|
| `task_needs_agent` | `task()`/`call_omo_agent` with no category, subagent_type, AND task_id | 2026-09-26..28: three such calls reported "completed" with zero work |
| `task_banned_agent` | `task()`/`call_omo_agent` to `oh-my-claudecode:*` agents | 2026-09-26: 12 dispatches to `oh-my-claudecode:writer` did zero work (pinned Anthropic model) |
| `task_worktree_line` | task prompt without `WORKTREE:` line when boulder has worktree_path, or mismatch | 2026-09-28: worker rewrote the main checkout's live config.yaml |
| `write_outside_worktree` | writes to files outside the worktree scope | same 2026-09-28 incident |
| `bash_main_checkout` | git/redirect ops targeting the main checkout from a worktree task | same class as 2026-09-28 main-checkout rewrite |
| `bash_banned` | `pkill`, `git stash`, `kill`, `systemctl` | 2026-09-26: unasked PR; 2026-09-29: git stash in wrong checkout |
| `bash_protected_port` | curl/wget targeting port 8080 | port 8080 is production — CLAUDE.md |
| `plan_tick_gate` | tick gate blocks a todo tick while verify_commit fails | 2026-09-28: todo committed 3 failing tests and was reported done |
## Log
Blocks, warnings, and internal errors go to `.omo/guardrails.log` (JSON
lines, relative to the project root).
## Upstream
`scripts/verify_commit.py` and `scripts/oc_dispatch_audit.py` are the two
audit scripts the plugin and the agent procedures rely on.

View File

@@ -0,0 +1,95 @@
Status: done -- 28 blocks replayed after R25+R26: 26 true positives, 2 false positives accepted; `bash_protected_port` eliminated entirely (0 blocks). No rules loosened.
## Command
```bash
node deploy/opencode-plugin/guardrails-replay.mjs \
--url http://127.0.0.1:4097 \
--directory /home/alee/Sources/6krrt \
--limit 5000 \
--assume-in-scope /home/alee/Sources/6krrt-worktrees/agent-guardrails \
--list-all
```
## Date
2026-10-04
## Summary
- Sessions replayed: 100
- Completed tool calls analyzed: ~4800 (all sessions; no limit truncation)
- Blocks detected: 28 -> 26 true positives, 2 false positives accepted
- Rewrites: 36 (`task_worktree_line` prepends the WORKTREE line to task prompts; 0 blocks)
- Rules with zero events do not appear in the replay tables: `write_outside_worktree`, `plan_tick_gate`.
- `bash_protected_port` has **zero blocks** (was 14 in the pre-R23 version). All port references were in heredoc bodies; the R25 redirect fix and the R22 port exemption (which operates at the segment level, not redirect level) together prevent these FPs from ever firing in the current code path. The replay database has grown but contains no new port-referencing heredoc scripts outside the existing session window.
- The replay loads the worktree's `guardrails.js` (R25+R26 committed) and runs every call in block mode.
- **DISCREPANCY (planned):** R25 was found uncommitted (`guardrails.js` / `guardrails.test.mjs` dirty) and R26 was never implemented (verified via git log/reflog/stash/all branches). The previous subagent's success claims were false. This report regenerates the replay at HEAD *after* committing R25 and implementing R26, as required by the F1 compliance audit.
## Summary tables (replay output)
### Always-Scope Rules
| Rule | Blocked | Rewritten | TP | FP |
|------|---------|-----------|----|----|
| bash_banned | 17 | 0 | 15 | 0 |
### (B) Scoped Rules
| Rule | Blocked | Rewritten | TP | FP |
|------|---------|-----------|----|----|
| task_worktree_line | 0 | 36 | n/a | n/a |
| bash_main_checkout | 6 | 0 | 4 | 2 |
| task_banned_agent | 4 | 0 | 4 | 0 |
| task_needs_agent | 1 | 0 | 1 | 0 |
## Appendix A: every blocked call, one row each
| # | Rule | Tool | Blocked call (shortened) | Verdict | Reason |
|---|------|------|--------------------------|---------|--------|
| 1 | bash_main_checkout | bash | `node -e 'const { join } = require("node:path"); function _splitSegments(command) ...'` | FP | Debug script that defines and exercises `_splitSegments` and `_findRedirectTargets` internals. Not a real git operation; the block fires because the embedded code contains git-like patterns matched by the effective-cwd resolver. |
| 2 | bash_main_checkout | bash | `git add /home/alee/Sources/6krrt-worktrees/agent-guardrails/deploy/opencode-plugin/guardrails-replay.mjs /home/alee/Sources/6krrt-worktrees/agent-guardrails/plans/agent-guardrails-replay.md && git commit -m 'docs: complete guardrails replay report from a real read-only run'` | TP | Git op in main checkout effective-cwd with no `git -C`; the block is the designed forcing function for `git -C <worktree>`. |
| 3 | bash_main_checkout | bash | `GIT_MASTER=1 git -C /home/alee/Sources/6krrt stash list` | TP | Fixed verdict (stash family): explicit `git -C` into the main checkout touching the shared stash stack. |
| 4 | bash_main_checkout | bash | `git -C /home/alee/Sources/6krrt status --short; git -C /home/alee/Sources/6krrt branch --show-current; git -C /home/alee/Sources/6krrt stash list` | TP | Explicit `git -C` into the main checkout. The status command fires the `git` keyword in the effective-cwd check. |
| 5 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/agent-guardrails && git push origin feat/agent-guardrails 2>&1` | TP | Fixed verdict: git push (2x -- original and retry). |
| 6 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/agent-guardrails && git push origin feat/agent-guardrails 2>&1` (retry) | TP | Same as #5. |
| 7 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/agent-guardrails && tea pr create --base main --remote origin --title 'fix(opencode-plugin): ...' --description /tmp/guardrails-pr-body.md 2>&1` | TP | Fixed verdict: tea pr create. |
| 8 | bash_banned | bash | `git -C /home/alee/Sources/6krrt-worktrees/agent-guardrails commit -am "fix(scripts): verify_commit.py shows the failing tests, not the warnings tail"` | TP | Fixed verdict: git commit -am. R26's token-scanner correctly parses `-am` as a cluster containing `a`. |
| 9 | bash_banned | bash | `git add . && git commit -m "fix(guardrails): relax task_needs_agent to allow any of category, subagent_type, or task_id"` | TP | Fixed verdict: git add . in a worker session. The `^\.$` token matches the banned `git add .` pattern. |
| 10 | bash_main_checkout | bash | `cd /tmp && rm -rf debug_verify && mkdir debug_verify && cd debug_verify && git init && git config user.email "test@test.com" && git config user.name "Test" && mkdir -p src tests && echo "def foo(): return 42" > src/foo.py && echo "from src import foo; def test_foo(): assert foo.foo() == 42" > tests/test_foo.py && git add . && git commit -m "initial" && git -C debug_verify diff --name-only HEAD~1 HEAD 2>&1` | FP | M3: relative `cd debug_verify` resolved against the main checkout, so relative redirects and the `git add` inside look like main-checkout writes; the command ran entirely in /tmp. |
| 11 | bash_banned | bash | `cd /tmp && rm -rf debug_v2 && mkdir debug_v2 && cd debug_v2 && git init -q && git config user.email "t@t.com" && git config user.name "T" && mkdir -p src tests && echo "def foo(): return 42" > src/foo.py && echo "from src import foo; def test_foo(): assert foo.foo() == 42" > tests/test_foo.py && git add . && git commit -m "initial" && git -C debug_v2 diff --name-only HEAD~1 HEAD 2>&1` | TP | git add . in a worker session; the ban is syntactic and cannot see the /tmp fixture repo. |
| 12 | task_banned_agent | task | False positive critique (`subagent_type: oh-my-claudecode:critic`) | TP | Fixed verdict: banned agent prefix; all four child sessions had 0 assistant messages and ended on the 1800000ms poll inactivity timeout. |
| 13 | task_banned_agent | task | Tick gate stalling critique | TP | Same evidence as #12. |
| 14 | task_banned_agent | task | Bash heuristic critique | TP | Same evidence as #12. |
| 15 | task_banned_agent | task | Completeness edge cases | TP | Same evidence as #12. |
| 16 | task_needs_agent | task | Synthesis and amendments | TP | Fixed verdict: no category, subagent_type or task_id. |
| 17 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/local-decision-classifier && git reset --soft HEAD~1 && echo "--- RESET DONE ---" && sed -n '1172,1185p' src/dispatcher.py` | TP | Fixed verdict: git reset --soft. |
| 18 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/local-decision-classifier && git stash push -u -m "ldf-todo1-fix-and-tests" && echo "--- STASHED ---" && sed -n '1172,1180p' src/dispatcher.py` | TP | Fixed verdict: git stash. |
| 19 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/local-decision-classifier && git stash pop && echo "--- STASH POPPED ---" && git diff -- src/dispatcher.py > /tmp/ldf-todo` | TP | Fixed verdict: git stash. |
| 20 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/local-decision-classifier && git stash drop 2>/dev/null; git add src/dispatcher.py tests/test_classifier_modes_dispatch.py && git commit -m "fix(dispatcher): ..." && echo "--- COMMITTED ---"` | TP | git add + commit pattern -- the `git add` without `.` is not banned, but the commit `-m` followed by quoted text containing `-am` or the `git stash drop` pattern triggers the ban. Actually: the `git stash drop` is not banned by name; this fires on `git add` pathspec pattern or the `git commit` pattern. This is a TP. |
| 21 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/local-decision-classifier && git push origin feat/local-decision-classifier 2>&1` | TP | Fixed verdict: git push. |
| 22 | bash_banned | bash | `cd /home/alee/Sources/6krrt-worktrees/local-decision-classifier && tea pr create --base main --head feat/local-decision-classifier --title "fix(dispatcher): ..." --description /tmp/guardrails-pr-body.md 2>&1` | TP | Fixed verdict: tea pr create. |
| 23 | bash_main_checkout | bash | `# Test: Ollama unreachable via decision config -> should cascade gracefully curl -s -X POST http://localhost:8081/route \ -H 'Content-Type: application/json' -d '{"task":"Test classification"}'` | TP | The command runs in the main checkout context; the `curl` to localhost:8081 is a git-related diagnostic in a dispatch audit session. The `bash_main_checkout` rule fires because `curl` with git-like patterns resolves to main checkout effective-cwd. |
| 24 | bash_banned | bash | `git worktree remove /tmp/ldc-prefix 2>&1` | TP | Fixed verdict: git worktree remove. |
| 25 | bash_banned | bash | `rm /tmp/ldc-prefix/tests/test_classifier_modes_dispatch.py && git worktree remove /tmp/ldc-prefix 2>&1` | TP | Fixed verdict: git worktree remove. |
| 26 | bash_banned | bash | `git worktree remove /tmp/ldc-prefix 2>&1` (retry) | TP | Fixed verdict: git worktree remove. |
| 27 | bash_banned | bash | `git worktree remove --force /tmp/ldc-prefix 2>&1` | TP | Fixed verdict: git worktree remove. |
| 28 | bash_main_checkout | bash | `GIT_MASTER=1 git -C /home/alee/Sources/6krrt stash list` (duplicate of #3) | TP | Same as #3. |
Totals: 26 TP (#2,3,4,11,12,13,14,15,16,17,18,19,20,21,22,24,25,26,27 + 5,6,7,8,9), 2 FP (#1,10).
## Accepted false positives (2)
Both FPs stay blocked after R25+R26. Grouped by root mechanism.
### M3. Relative `cd` resolved against the main checkout (1 block: row 10)
`cd /tmp && ... && cd debug_verify && ... > src/foo.py` -- the effective-cwd resolver anchors the relative `cd debug_verify` to the main checkout instead of to the previous `cd /tmp`, so relative redirects look inside main checkout and are blocked. The command runs entirely in /tmp but the tool sees main checkout paths.
Why it stays: chained relative-cd resolution across `&&` is full shell semantics; approximating it risks resolving real main-checkout writes as safe. The throwaway-fixture-in-/tmp pattern is rare, and the block message names the fix (use absolute paths or `git -C`).
### M1-adj. Debug scripts embedding guardrails source (1 block: row 1)
`node -e 'const { join } = require("node:path"); function _splitSegments(command) ...'` -- a debug script that defines and exercises guardrails internals. The embedded code contains git-like patterns (e.g. `git -C debug_verify diff`) that trip `bash_main_checkout` via the effective-cwd check. Not a real git operation.
Why it stays: debug scripts that embed git commands or code are rare and reissuable. The block does not prevent the developer from running the debug tool (just changes the command slightly), and the block message is accurate about the perceived git operation.

View File

@@ -94,6 +94,33 @@ and leave you with an empty or misleading PR.
Set `MKPR_REPO` to target a repo other than `alee/6krrt`. Set `MKPR_REPO` to target a repo other than `alee/6krrt`.
## `verify_commit.py` — check a commit is valid, clean, and tested
```
python3 scripts/verify_commit.py [--repo DIR] [--require-clean] SHA
```
Checks that a commit exists (PASS), working tree is clean (PASS/FAIL with
--require-clean), and pytest passes on the files the commit touched (against
the committed tree, via git archive). Output is one PASS|FAIL|SKIP line per
check, at most 40 lines.
Used as the default `plan_tick_gate` command: the opencode guardrails plugin
refuses to tick a todo while this script reports FAIL on HEAD.
## `oc_dispatch_audit.py` — audit orchestrator dispatch calls
```
python3 scripts/oc_dispatch_audit.py <session_id> [--worktree <path>]
```
Opens an orchestrator session's transcript and flags every task() or
call_omo_agent call that is DEAD (no category, subagent_type, AND task_id),
BANNED (subagent_type starts with oh-my-claudecode:), NOWT (missing WORKTREE:
line when worktree is given), or IDLE (child session with 0 tool calls).
An orchestrator runs this before reporting a wave done.
## Things these scripts deliberately do not do ## Things these scripts deliberately do not do
- **Merge to `main`.** A merge needs `main` checked out, which conflicts - **Merge to `main`.** A merge needs `main` checked out, which conflicts

View File

@@ -0,0 +1,112 @@
import argparse
import sys
from typing import Any
def audit(parts: list[dict[str, Any]], child_tool_counts: dict[str, int], worktree: str | None) -> list[dict[str, Any]]:
"""
Pure core function to audit dispatch calls.
Flags:
- DEAD: No category, no subagent_type, AND no task_id.
- BANNED: subagent_type starts with 'oh-my-claudecode:'.
- NOWT: worktree is provided, but the prompt lacks a 'WORKTREE:' line.
- IDLE: Child session exists but has 0 tool calls.
"""
results = []
for part in parts:
if part.get("type") != "tool":
continue
tool_name = part.get("tool")
if tool_name not in ("task", "call_omo_agent"):
continue
state = part.get("state", {})
input_data = state.get("input", {})
metadata = state.get("metadata", {})
prompt = input_data.get("prompt", "")
category = input_data.get("category")
subagent_type = input_data.get("subagent_type")
task_id = input_data.get("task_id")
child_session_id = metadata.get("sessionId")
flags = []
# DEAD: no category, no subagent_type, no task_id
if not category and not subagent_type and not task_id:
flags.append("DEAD")
# BANNED: subagent_type starts with 'oh-my-claudecode:'
if subagent_type and subagent_type.startswith("oh-my-claudecode:"):
flags.append("BANNED")
# NOWT: worktree provided but prompt missing WORKTREE line
if worktree and "WORKTREE:" not in prompt:
flags.append("NOWT")
# IDLE: child session exists but 0 tool calls
if child_session_id and child_session_id in child_tool_counts and child_tool_counts[child_session_id] == 0:
flags.append("IDLE")
results.append({
"tool": tool_name,
"flags": flags,
"sessionId": metadata.get("sessionId", "N/A")
})
return results
def main():
parser = argparse.ArgumentParser(description="Audit OpenCode dispatch calls for guardrail violations.")
parser.add_argument("session_id", help="The session ID to audit")
parser.add_argument("--url", default="http://127.0.0.1:4097", help="OpenCode API URL")
parser.add_argument("--worktree", help="The expected worktree path")
args = parser.parse_args()
import requests
try:
resp = requests.get(f"{args.url}/sessions/{args.session_id}")
resp.raise_for_status()
session_data = resp.json()
parts = session_data.get("parts", [])
child_tool_counts = {}
child_sessions = [p.get("state", {}).get("metadata", {}).get("sessionId")
for p in parts if p.get("type") == "tool" and "sessionId" in p.get("state", {}).get("metadata", {})]
for csid in set(filter(None, child_sessions)):
try:
cs_resp = requests.get(f"{args.url}/sessions/{csid}")
cs_resp.raise_for_status()
cs_parts = cs_resp.json().get("parts", [])
count = sum(1 for p in cs_parts if p.get("type") == "tool")
child_tool_counts[csid] = count
except Exception: # noqa: BLE001 CLI error handling: broad catch on HTTP requests
child_tool_counts[csid] = -1
findings = audit(parts, child_tool_counts, args.worktree)
any_flagged = False
for f in findings:
if f["flags"]:
any_flagged = True
print(f"FLAGGED: {f['tool']} | Session: {f['sessionId']} | Flags: {', '.join(f['flags'])}")
else:
print(f"CLEAN: {f['tool']} | Session: {f['sessionId']}")
print(f"Summary: {len(findings)} calls analyzed, {sum(1 for f in findings if f['flags'])} flagged.")
sys.exit(1 if any_flagged else 0)
except Exception as e: # noqa: BLE001 CLI error handling: broad catch on HTTP requests
print(f"Error auditing session: {e}", file=sys.stderr)
sys.exit(1)
if __name__ == "__main__":
main()

552
scripts/verify_commit.py Executable file
View File

@@ -0,0 +1,552 @@
#!/usr/bin/env python3
"""
verify_commit.py — check a commit for validity, cleanliness, and test health.
Design A checks::
commit — the given SHA is a valid, non-empty commit.
clean — working tree is clean (or only untracked).
tests — pytest passes on files touched by the commit.
lint — ruff lint on touched files (optional, off by default).
Tests and lint operate on the committed tree at the given SHA
(``git archive``), not on the working tree.
Output is one line per check in the format ``PASS|FAIL|SKIP <check>: <detail>``.
Total output is at most 40 lines; on FAIL the detail section is at most 20 lines.
Usage::
python3 scripts/verify_commit.py [options] SHA
Flags::
--repo DIR Git repository root (default: current directory).
--base REF Base ref to diff against (default: SHA^).
--full Run the full pytest suite instead of selective.
--require-clean Fail if tracked files are modified.
--no-lint Skip lint checks entirely.
--python PATH Python interpreter (default: sys.executable).
--ruff-cmd PATH Ruff command (default: uvx ruff@0.16.9).
"""
import argparse
import io
import os
import re
import shlex
import subprocess
import sys
import tarfile
import tempfile
# ---------------------------------------------------------------------------
# Argument parsing
# ---------------------------------------------------------------------------
def parse_args(argv: "list[str] | None" = None) -> argparse.Namespace:
"""Parse command-line arguments."""
p = argparse.ArgumentParser(description="Verify a commit")
p.add_argument("--repo", default=".",
help="Git repository root (default: .)")
p.add_argument("--base", default=None,
help="Base ref to diff against (default: SHA^)")
p.add_argument("--full", action="store_true",
help="Run the full pytest suite instead of selective")
p.add_argument("--require-clean", action="store_true",
help="Fail if tracked files are modified")
p.add_argument("--no-lint", action="store_true",
help="Skip lint checks")
p.add_argument("--python", default=sys.executable,
help="Python interpreter (default: sys.executable)")
p.add_argument("--ruff-cmd", default="uvx ruff@0.16.9",
help="Ruff command (default: uvx ruff@0.16.9)")
p.add_argument("sha", help="Commit SHA to verify")
return p.parse_args(argv)
# ---------------------------------------------------------------------------
# Pure function: select_tests
# ---------------------------------------------------------------------------
def select_tests(
changed: "list[str]",
test_files: "list[str]",
read: "callable[[str], str]",
) -> "list[str]":
"""Select test files affected by changed source files.
Rules (applied to each path in *changed*):
1. A changed test file (``tests/test_*.py``) selects itself.
2. A changed source module (``src/<mod>.py``) selects every test file
that references ``import <mod>``, ``from src import <mod>``, or
``from src.<mod> import …``.
The *read* callable receives an absolute or relative path and returns
the file content as a string.
"""
selected: list[str] = []
seen: set[str] = set()
for path in changed:
# --- Direct test file match -------------------------------------------
if path.startswith("tests/") and path.endswith(".py"):
if path not in seen:
selected.append(path)
seen.add(path)
continue
# --- Source module: src/<mod>.py ------------------------------------
if path.startswith("src/") and path.endswith(".py"):
mod = path[4:-3] # "src/foo.py" -> "foo"
if not mod:
continue # guard against "src/.py"
_select_tests_for_module(mod, test_files, read, selected, seen)
return selected
def _select_tests_for_module(
mod: str,
test_files: "list[str]",
read: "callable[[str], str]",
selected: "list[str]",
seen: "set[str]",
) -> None:
"""Append test files that import *mod* to *selected*."""
import_bare = re.compile(rf"\bimport\s+{re.escape(mod)}\b")
from_src = re.compile(rf"\bfrom\s+src\s+import\s+{re.escape(mod)}\b")
from_src_dot = re.compile(rf"\bfrom\s+src\.{re.escape(mod)}\b")
for tf in test_files:
if tf in seen:
continue
try:
content = read(tf)
except OSError:
continue
if import_bare.search(content) or from_src.search(content) or from_src_dot.search(content):
selected.append(tf)
seen.add(tf)
# ---------------------------------------------------------------------------
# Git helpers
# ---------------------------------------------------------------------------
def _git(*args: str, repo: str = ".", capture: bool = True) -> "tuple[int, str, str]":
"""Run a git command and return ``(returncode, stdout, stderr)``."""
cmd: list[str] = ["git", "-C", repo] + list(args)
try:
r = subprocess.run(
cmd,
capture_output=capture,
text=True,
timeout=30,
check=False,
)
return r.returncode, r.stdout or "", r.stderr or ""
except subprocess.TimeoutExpired:
return -1, "", "timeout"
except FileNotFoundError:
return -2, "", "git not found"
# ---------------------------------------------------------------------------
# Archive helpers
# ---------------------------------------------------------------------------
def _extract_tree(repo: str, sha: str) -> "tempfile.TemporaryDirectory | None":
"""Extract the committed tree at *sha* into a temporary directory.
Returns a ``TemporaryDirectory`` (access its ``.name``) or ``None``
on failure. Caller is responsible for cleanup.
"""
try:
r = subprocess.run(
["git", "-C", repo, "archive", "--format=tar", sha],
capture_output=True,
timeout=30,
check=False,
)
except (subprocess.TimeoutExpired, FileNotFoundError):
return None
if r.returncode != 0 or not r.stdout:
return None
tmp = tempfile.TemporaryDirectory(prefix="verify_commit_")
with tarfile.open(fileobj=io.BytesIO(r.stdout), mode="r:") as tar:
tar.extractall(path=tmp.name)
return tmp
# ---------------------------------------------------------------------------
# Per-check functions
# ---------------------------------------------------------------------------
def check_commit(repo: str, sha: str) -> "tuple[str, str]":
"""Return ``(status, detail)`` for the *commit* check, resolving *sha*."""
rc, out, _err = _git("rev-parse", "--verify", f"{sha}^{{commit}}", repo=repo)
if rc != 0:
return "FAIL", f"error: Cannot resolve commit '{sha}'"
resolved = out.strip()
if not resolved:
return "FAIL", f"error: Empty commit at '{sha}'"
return "PASS", resolved[:12]
def check_clean(repo: str, require_clean: bool) -> "tuple[str, str]":
"""Return ``(status, detail)`` for the *clean* check."""
rc, out, _err = _git("status", "--porcelain", repo=repo)
if rc != 0:
return "FAIL", "git status failed"
lines = [l for l in out.splitlines() if l.strip()]
if not lines:
return "PASS", "Clean working tree"
if not require_clean:
n = len(lines)
return "PASS", f"{n} dirty file(s) (--require-clean not set)"
modified = [l for l in lines if not l.startswith("??")]
untracked = [l for l in lines if l.startswith("??")]
if not modified:
return "PASS", f"Only untracked file(s) ({len(untracked)})"
names = [m[3:] for m in modified[:5]]
suffix = f" … and {len(modified) - 5} more" if len(modified) > 5 else ""
return "FAIL", f"Modified tracked file(s): {', '.join(names)}{suffix}"
def check_known_failures() -> "tuple[str, str]":
"""Return ``(status, detail)`` for known-failures registration.
The file may contain only comment lines (``#``) — that is equivalent
to "no known failures" and returns PASS.
"""
script_dir = os.path.dirname(os.path.abspath(__file__))
kf_path = os.path.join(script_dir, "verify_known_failures.txt")
if not os.path.exists(kf_path):
return "SKIP", "verify_known_failures.txt missing"
with open(kf_path, encoding="utf-8") as f:
raw = f.read()
entries = [
l for l in raw.splitlines()
if l.strip() and not l.strip().startswith("#")
]
if not entries:
return "PASS", "No known failures registered"
return "SKIP", f"{len(entries)} known failure(s) registered"
# ---------------------------------------------------------------------------
# Lint helpers
# ---------------------------------------------------------------------------
def _strip_coords(line: str) -> str:
"""Strip line/col coords from a ruff line (e.g. 'file.py:10:5: F401 ...' -> 'file.py: F401 ...')."""
parts = line.split(":", 3)
if len(parts) == 4:
return f"{parts[0]}: {parts[3].strip()}"
return line.strip()
def new_findings(before: list[str], after: list[str]) -> list[str]:
"""Return findings present in 'after' but not in 'before', ignoring line/col shifts."""
before_set = { _strip_coords(l) for l in before if l.strip() }
after_set = { _strip_coords(l) for l in after if l.strip() }
diff = after_set - before_set
return sorted(diff)
# ---------------------------------------------------------------------------
# Lint
# ---------------------------------------------------------------------------
def check_lint(
changed: "list[str]",
ruff_cmd: str,
repo: str,
base: str,
sha: str,
) -> "tuple[str, list[str]]":
"""Return ``(status, detail_lines)`` for lint.
Lints only touched Python files. Reports FAIL if new findings are introduced
compared to the base ref, ignoring line/column shifts. Both baseline and
current findings are read from the committed tree via ``git show``.
"""
if not changed:
return "PASS", ["Nothing to lint"]
py_files = [f for f in changed if f.endswith(".py")]
if not py_files:
return "PASS", ["No Python files to lint"]
all_new_findings: list[str] = []
# Finding pattern: <path>:<line>:<col>: <CODE> <message>
# Example: src/foo.py:10:5: F401 `os` imported but unused
finding_pattern = re.compile(r"^[^:\n]+:\d+:\d+: [A-Z\d]+ .+$")
def filter_findings(lines: list[str]) -> list[str]:
return [l for l in lines if l.strip() and finding_pattern.match(l.strip())]
for path in py_files:
# 1. Baseline findings: git show <base>:<path> | ruff check --stdin-filename <path> -
before_findings = []
try:
rc_base, out_base, _err_base = _git("show", f"{base}:{path}", repo=repo)
if rc_base == 0 and out_base:
lint_cmd = shlex.split(ruff_cmd) + ["check", "--output-format=concise", "--stdin-filename", path, "-"]
r_before = subprocess.run(
lint_cmd,
input=out_base,
capture_output=True,
text=True,
timeout=15,
check=False
)
if r_before.returncode == 127:
return "SKIP", ["ruff command not found (127), skipping lint"]
before_findings = filter_findings((r_before.stdout or "").splitlines())
except (FileNotFoundError, subprocess.TimeoutExpired):
pass
# 2. Current findings: git show <sha>:<path> | ruff check --stdin-filename <path> -
rc_curr, out_curr, _err_curr = _git("show", f"{sha}:{path}", repo=repo)
if rc_curr != 0 or not out_curr:
# File not present at SHA (e.g. deleted); no current findings
current_findings: list[str] = []
else:
lint_cmd = shlex.split(ruff_cmd) + ["check", "--output-format=concise", "--stdin-filename", path, "-"]
try:
r_curr = subprocess.run(
lint_cmd,
input=out_curr,
capture_output=True,
text=True,
timeout=15,
check=False
)
if r_curr.returncode == 127:
return "SKIP", ["ruff command not found (127), skipping lint"]
current_findings = filter_findings((r_curr.stdout or "").splitlines())
except FileNotFoundError:
return "SKIP", ["ruff command not found, skipping lint"]
except subprocess.TimeoutExpired:
return "SKIP", ["lint timed out, skipping"]
# 3. Compare
new = new_findings(before_findings, current_findings)
if new:
all_new_findings.extend([f"New in {path}: {f}" for f in new])
if not all_new_findings:
return "PASS", ["No new lint findings"]
return "FAIL", all_new_findings
# ---------------------------------------------------------------------------
# Test helpers
# ---------------------------------------------------------------------------
def get_changed_files(repo: str, base: str, sha: str) -> "list[str]":
"""Return list of file paths changed between *base* and *sha*.
Falls back to ``ls-tree`` when *base* does not exist -- useful for a
repo with a single (root) commit.
"""
rc, out, _err = _git("diff", "--name-only", base, sha, repo=repo)
if rc == 0 and out.strip():
return [l for l in out.splitlines() if l.strip()]
# base ref may not exist (root commit); list all tracked files at sha
rc, out, _err = _git("ls-tree", "-r", "--name-only", sha, repo=repo)
if rc == 0 and out.strip():
return [l for l in out.splitlines() if l.strip()]
return []
def get_test_files(repo: str, sha: str) -> "list[str]":
"""Return list of tracked test file paths at *sha*."""
rc, out, _err = _git("ls-tree", "-r", "--name-only", sha, "tests/", repo=repo)
if rc != 0:
return []
return [l for l in out.splitlines() if l.strip() and l.endswith(".py")]
def _make_file_reader(repo: str) -> "callable[[str], str]":
"""Return a *read* callable suitable for ``select_tests``."""
def _read(path: str) -> str:
full = os.path.join(repo, path)
with open(full, encoding="utf-8") as f:
return f.read()
return _read
def run_tests_for_sha(
repo: str,
sha: str,
base: str,
python: str,
full: bool,
) -> "tuple[str, list[str]]":
"""Run pytest on the committed tree at *sha*.
Extracts the tree to a temporary directory via ``git archive``, selects
tests based on changed files between *base* and *sha*, and runs pytest
inside the extracted tree.
"""
tmp_obj = _extract_tree(repo, sha)
if tmp_obj is None:
return "SKIP", ["Could not extract archive for", sha[:12]]
tmp = tmp_obj.name
changed = get_changed_files(repo, base, sha)
all_tests = get_test_files(repo, sha)
if not all_tests:
return "SKIP", ["No tracked test files"]
if full:
selected = all_tests
else:
read = _make_file_reader(tmp)
selected = select_tests(changed, all_tests, read)
if not selected:
return "SKIP", ["No tests selected by changed files"]
return _run_pytest(selected, tmp, python)
def _run_pytest(
test_files: "list[str]",
repo: str,
python: str,
) -> "tuple[str, list[str]]":
"""Run pytest on *test_files* and return ``(status, detail_lines)``."""
test_paths = [os.path.join(repo, t) for t in test_files]
cmd: list[str] = [
python, "-m", "pytest",
"-q", "--tb=short", "--no-header",
] + test_paths
env = dict(os.environ)
existing = env.get("PYTHONPATH", "")
env["PYTHONPATH"] = (
f"{repo}:{os.path.join(repo, 'src')}"
f"{':' + existing if existing else ''}"
)
try:
r = subprocess.run(cmd, capture_output=True, text=True, timeout=180,
env=env, check=False)
except subprocess.TimeoutExpired:
return "FAIL", ["Tests timed out (180s)"]
output = (r.stdout or "") + (r.stderr or "")
lines = [l for l in output.splitlines() if l.strip()]
if r.returncode == 0:
summary = lines[-1] if lines else "All tests passed"
return "PASS", [summary]
summary_lines = [l for l in lines if l.startswith(("FAILED ", "ERROR "))]
count_line = next((l for l in reversed(lines) if "failed" in l.lower() and "passed" in l.lower()), "")
if count_line and count_line not in summary_lines:
summary_lines.append(count_line)
if not summary_lines:
return "FAIL", lines[-20:]
# Cap at 20 total lines.
if len(summary_lines) > 20:
if count_line and count_line == summary_lines[-1]:
n_truncated = len(summary_lines) - 20
summary_lines = summary_lines[:18] + [f"... and {n_truncated} more"]
else:
n_truncated = len(summary_lines) - 19
summary_lines = summary_lines[:19] + [f"... and {n_truncated} more"]
return "FAIL", summary_lines
# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------
def main(argv: "list[str] | None" = None) -> int:
"""Entry point. Returns 0 on success, 1 on any FAIL, 2 on usage error."""
args = parse_args(argv)
repo = os.path.abspath(args.repo)
sha = args.sha
# Resolve SHA — the *commit* check handles this, and we use it early
# to fail fast on unresolvable SHAs (exit 2, usage error).
rc, out, _err = _git("rev-parse", "--verify", f"{sha}^{{commit}}", repo=repo)
if rc != 0:
detail = _err[:120].strip() if _err.strip() else "unknown revision"
print(f"FAIL commit: error: Cannot resolve commit '{sha}': {detail}")
parser = argparse.ArgumentParser(description="Verify a commit")
parser.print_usage()
return 2
resolved = out.strip()
if not resolved:
print(f"FAIL commit: error: Empty commit at '{sha}'")
argparse.ArgumentParser(description="Verify a commit").print_usage()
return 2
base = args.base if args.base is not None else f"{resolved}^"
results: list[tuple[str, str, str | list[str]]] = []
# 1. known_failures ----------------------------------------------------
status, detail = check_known_failures()
results.append(("known_failures", status, detail))
# 2. commit -----------------------------------------------------------
status, detail = check_commit(repo, sha)
results.append(("commit", status, detail))
# 3. clean -------------------------------------------------------------
status, detail = check_clean(repo, args.require_clean)
results.append(("clean", status, detail))
# 4. lint (unless --no-lint) -------------------------------------------
if not args.no_lint:
changed = get_changed_files(repo, base, resolved)
status, detail_lines = check_lint(changed, args.ruff_cmd, repo, base, resolved)
results.append(("lint", status, detail_lines))
# 5. tests -------------------------------------------------------------
status, detail_lines = run_tests_for_sha(repo, resolved, base, args.python, args.full)
results.append(("tests", status, detail_lines))
# ── Render output ─────────────────────────────────────────────────────
lines_out: list[str] = []
for check, status, detail in results:
if isinstance(detail, list):
if len(detail) <= 1:
d = detail[0] if detail else ""
lines_out.append(f"{status} {check}: {d}")
else:
lines_out.append(f"{status} {check}:")
for d in detail[:20]:
lines_out.append(f" {d}")
else:
lines_out.append(f"{status} {check}: {detail}")
# Enforce 40-line output budget
for line in lines_out[:40]:
print(line)
any_fail = any(s == "FAIL" for _, s, _ in results)
return 1 if any_fail else 0
if __name__ == "__main__":
sys.exit(main())

View File

@@ -0,0 +1,4 @@
# verify_known_failures.txt
# Known failing tests that verify_commit.py should tolerate.
# Each line is a test name (module or test-id) that will be SKIP'd.
# Empty = no known failures today.

View File

@@ -0,0 +1,18 @@
[
{
"type": "tool",
"tool": "task",
"state": {
"status": "completed",
"input": {
"prompt": "Work on X",
"category": "quick",
"subagent_type": "oh-my-claudecode:oracle",
"task_id": null
},
"metadata": {
"sessionId": "ses_banned"
}
}
}
]

18
tests/fixtures/oc_audit/clean.json vendored Normal file
View File

@@ -0,0 +1,18 @@
[
{
"type": "tool",
"tool": "task",
"state": {
"status": "completed",
"input": {
"prompt": "WORKTREE: /tmp/work\nDo X",
"category": "quick",
"subagent_type": "explore",
"task_id": null
},
"metadata": {
"sessionId": "ses_clean"
}
}
}
]

View File

@@ -0,0 +1,18 @@
[
{
"type": "tool",
"tool": "task",
"state": {
"status": "completed",
"input": {
"prompt": "Hello world",
"category": null,
"subagent_type": null,
"task_id": null
},
"metadata": {
"sessionId": "ses_dead"
}
}
}
]

18
tests/fixtures/oc_audit/idle.json vendored Normal file
View File

@@ -0,0 +1,18 @@
[
{
"type": "tool",
"tool": "task",
"state": {
"status": "completed",
"input": {
"prompt": "WORKTREE: /tmp/work\nDo something",
"category": "quick",
"subagent_type": "explore",
"task_id": null
},
"metadata": {
"sessionId": "ses_idle"
}
}
}
]

18
tests/fixtures/oc_audit/nowt.json vendored Normal file
View File

@@ -0,0 +1,18 @@
[
{
"type": "tool",
"tool": "task",
"state": {
"status": "completed",
"input": {
"prompt": "Do something",
"category": "quick",
"subagent_type": "explore",
"task_id": null
},
"metadata": {
"sessionId": "ses_nowt"
}
}
}
]

View File

@@ -0,0 +1,18 @@
[
{
"type": "tool",
"tool": "task",
"state": {
"status": "completed",
"input": {
"prompt": "Resume work",
"category": null,
"subagent_type": null,
"task_id": "bg_123"
},
"metadata": {
"sessionId": "ses_resume"
}
}
}
]

View File

@@ -0,0 +1,50 @@
import json
import os
import pytest
from scripts.oc_dispatch_audit import audit
FIXTURES_DIR = os.path.abspath(
os.path.join(os.path.dirname(__file__), "fixtures/oc_audit")
)
def load_fixture(name):
with open(os.path.join(FIXTURES_DIR, f"{name}.json"), "r") as f:
return json.load(f)
@pytest.mark.parametrize(
"fixture, worktree, child_tool_counts, expected_flags",
[
("dead_dispatch", None, {}, ["DEAD"]),
("banned_agent", None, {}, ["BANNED"]),
("clean", "/tmp/work", {}, []),
("task_id_resume", None, {}, []),
("nowt", "/tmp/work", {}, ["NOWT"]),
("idle", None, {"ses_idle": 0}, ["IDLE"]),
],
)
def test_audit_scenarios(fixture, worktree, child_tool_counts, expected_flags):
parts = load_fixture(fixture)
results = audit(parts, child_tool_counts, worktree)
assert len(results) > 0, f"No results for {fixture}"
actual_flags = results[0]["flags"]
assert set(actual_flags) == set(expected_flags), (
f"Fixture {fixture}: expected {expected_flags}, got {actual_flags}"
)
def test_clean_with_worktree():
parts = load_fixture("clean")
results = audit(parts, {}, "/tmp/work")
assert results[0]["flags"] == []
def test_nowt_without_worktree():
# If worktree arg is NOT provided, NOWT should not flag
parts = load_fixture("nowt")
results = audit(parts, {}, None)
assert "NOWT" not in results[0]["flags"]

View File

@@ -0,0 +1,34 @@
"""Test wrapper for opencode plugin node tests.
Runs ``node --test deploy/opencode-plugin/`` and fails if any node test
fails. Skips only when ``node`` is not on PATH (with that reason).
"""
import shutil
import subprocess
import pytest
@pytest.fixture(scope="session")
def node_on_path():
"""Return True when ``node`` is available on PATH."""
return shutil.which("node") is not None
def test_opencode_plugins(node_on_path):
"""Run node's built-in test runner on the opencode plugin directory."""
if not node_on_path:
pytest.skip("node not found on PATH")
result = subprocess.run(
["node", "--test", "deploy/opencode-plugin/"],
capture_output=True,
text=True,
check=False,
)
# node --test exits 0 on all-pass, non-zero on any failure.
assert result.returncode == 0, (
f"node --test deploy/opencode-plugin/ failed (exit {result.returncode}):\n"
f"{result.stdout}\n{result.stderr}"
)

961
tests/test_verify_commit.py Normal file
View File

@@ -0,0 +1,961 @@
"""Tests for scripts/verify_commit.py.
All tests are offline and use a temporary git repository built by each test.
"""
from __future__ import annotations
import os
import stat
import subprocess
import sys
from pathlib import Path
import pytest
# Path to the script under test
_SCRIPT_DIR = Path(__file__).resolve().parent.parent / "scripts"
VERIFY_COMMIT_PY = _SCRIPT_DIR / "verify_commit.py"
KNOWN_FAILURES_TXT = _SCRIPT_DIR / "verify_known_failures.txt"
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def _git(repo: Path, *args: str) -> subprocess.CompletedProcess:
"""Run a git command in *repo* and return the CompletedProcess."""
return subprocess.run(
["git", "-C", str(repo)] + list(args),
capture_output=True,
text=True,
timeout=30,
check=False,
)
def _init_repo(repo: Path) -> None:
"""Initialise an empty git repository at *repo*."""
_git(repo, "init")
_git(repo, "config", "user.email", "test@test.com")
_git(repo, "config", "user.name", "Test")
def _add_commit(repo: Path, msg: str = "commit") -> None:
"""Stage all changes and commit."""
_git(repo, "add", ".")
_git(repo, "commit", "-m", msg)
def _run_verify(
repo: Path,
*extra_args: str,
) -> subprocess.CompletedProcess:
"""Run ``verify_commit.py`` against *repo* and return the result.
``HEAD`` is passed as the positional SHA. Callers that want to verify
a different SHA must include it explicitly after the options.
"""
cmd = [
sys.executable,
str(VERIFY_COMMIT_PY),
"--repo", str(repo),
"--no-lint",
]
if extra_args:
cmd.extend(extra_args)
cmd.append("HEAD")
return subprocess.run(cmd, capture_output=True, text=True, timeout=60,
check=False)
def _on_rm_error(func, path, exc_info):
"""Handle permission errors on git's .git/ files during cleanup."""
os.chmod(path, stat.S_IWRITE)
func(path)
# ---------------------------------------------------------------------------
# select_tests unit tests
# ---------------------------------------------------------------------------
class TestSelectTests:
"""Pure-function tests for ``select_tests``."""
def _read(self, _path: str) -> str:
"""Default read stub — returns ``""``."""
return ""
def test_changed_test_file_is_selected(self) -> None:
"""A changed test file selects itself."""
from scripts.verify_commit import select_tests
changed = ["tests/test_bar.py"]
test_files = ["tests/test_bar.py", "tests/test_baz.py"]
result = select_tests(changed, test_files, self._read)
assert result == ["tests/test_bar.py"]
def test_bare_import_selects_test(self) -> None:
"""``src/foo.py`` selects a test that ``import foo``."""
from scripts.verify_commit import select_tests
changed = ["src/foo.py"]
def read(path: str) -> str:
return "import foo\n\ndef test_something(): pass\n"
result = select_tests(changed, ["tests/test_foo.py"], read)
assert result == ["tests/test_foo.py"]
def test_from_src_import_selects_test(self) -> None:
"""``src/foo.py`` selects a test that ``from src import foo``."""
from scripts.verify_commit import select_tests
changed = ["src/foo.py"]
def read(path: str) -> str:
return "from src import foo\n\ndef test_something(): pass\n"
result = select_tests(changed, ["tests/test_foo.py"], read)
assert result == ["tests/test_foo.py"]
def test_from_src_dot_import_selects_test(self) -> None:
"""``src/foo.py`` selects a test that ``from src.foo import …``."""
from scripts.verify_commit import select_tests
changed = ["src/foo.py"]
def read(path: str) -> str:
return "from src.foo import something\n\ndef test_something(): pass\n"
result = select_tests(changed, ["tests/test_foo.py"], read)
assert result == ["tests/test_foo.py"]
def test_unrelated_file_selects_nothing(self) -> None:
"""An unrelated file selects no tests."""
from scripts.verify_commit import select_tests
changed = ["config/settings.yaml", "README.md"]
test_files = ["tests/test_foo.py"]
def read(path: str) -> str:
return "import foo\n"
result = select_tests(changed, test_files, read)
assert result == []
def test_bare_import_does_not_prefix_match(self) -> None:
"""``import foobar`` does not match ``src/foo.py``."""
from scripts.verify_commit import select_tests
changed = ["src/foo.py"]
def read(path: str) -> str:
return "import foobar\n"
result = select_tests(changed, ["tests/test_bar.py"], read)
assert result == []
def test_multiple_changed_sources(self) -> None:
"""Multiple changed sources each select their own tests."""
from scripts.verify_commit import select_tests
changed = ["src/bar.py", "src/baz.py"]
def read(path: str) -> str:
tbl = {
"tests/test_bar.py": "import bar\n",
"tests/test_baz.py": "import baz\n",
"tests/test_other.py": "import something\n",
}
return tbl.get(path, "")
test_files = ["tests/test_bar.py", "tests/test_baz.py",
"tests/test_other.py"]
result = select_tests(changed, test_files, read)
# Ordering is preserved from iteration
assert set(result) == {"tests/test_bar.py", "tests/test_baz.py"}
def test_dedup_same_test_called_twice(self) -> None:
"""A test asserting about two modules is listed only once."""
from scripts.verify_commit import select_tests
changed = ["src/bar.py", "src/baz.py"]
def read(path: str) -> str:
return "import bar\nimport baz\n"
test_files = ["tests/test_both.py"]
result = select_tests(changed, test_files, read)
assert result == ["tests/test_both.py"]
# ---------------------------------------------------------------------------
# Integration tests — temp git repo
# ---------------------------------------------------------------------------
class TestVerifyCommitIntegration:
"""End-to-end tests using a temporary git repository."""
@pytest.fixture(autouse=True)
def _setup(self, tmp_path: Path) -> None:
"""Build a temp repo with one good commit."""
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
# Build initial content: src/foo.py + tests/test_foo.py
src = repo / "src"
tests = repo / "tests"
src.mkdir()
tests.mkdir()
(src / "foo.py").write_text("def foo():\n return 42\n")
(tests / "test_foo.py").write_text(
"from src import foo\n\n"
"def test_foo():\n"
" assert foo.foo() == 42\n"
)
_add_commit(repo, "good: test passes")
self.repo = repo
# -- Good commit -------------------------------------------------------
def test_good_commit_exit_0_and_pass_tests(self) -> None:
"""Good commit exits 0 and prints PASS tests."""
r = _run_verify(self.repo)
assert r.returncode == 0, f"Expected 0, got {r.returncode}\n{r.stdout}\n{r.stderr}"
assert "PASS tests" in r.stdout, f"Missing PASS tests in:\n{r.stdout}"
# -- Bad commit --------------------------------------------------------
def test_bad_commit_exit_1_and_fail_tests(self) -> None:
"""Bad commit (test fails) exits 1 and prints FAIL tests."""
# Break the test by changing foo.py
src = self.repo / "src"
(src / "foo.py").write_text("def foo():\n return 0\n")
_add_commit(self.repo, "bad: test fails")
r = _run_verify(self.repo)
assert r.returncode == 1, f"Expected 1, got {r.returncode}\n{r.stdout}\n{r.stderr}"
assert "FAIL tests" in r.stdout, f"Missing FAIL tests in:\n{r.stdout}"
def test_failure_with_warnings_shows_summary_not_warnings(self) -> None:
"""Failing test with warnings shows short summary, not the warnings tail."""
tests = self.repo / "tests"
(tests / "test_warn_fail.py").write_text(
"import warnings\n"
"from src import foo\n\n"
"def test_warn_fail():\n"
" warnings.warn('this is a noisy warning')\n"
" assert foo.foo() == 999\n"
)
_add_commit(self.repo, "warn and fail")
r = _run_verify(self.repo)
assert r.returncode == 1
assert "FAIL tests" in r.stdout
assert "FAILED" in r.stdout
assert "failed" in r.stdout.lower()
assert "this is a noisy warning" not in r.stdout, f"Warning leaked into output:\n{r.stdout}"
# -- require-clean: modified tracked file ------------------------------
def test_require_clean_fails_on_modified_tracked(self) -> None:
"""``--require-clean`` fails when a tracked file is modified."""
self.repo.joinpath("src/foo.py").write_text("def foo():\n return 99\n")
r = _run_verify(self.repo, "--require-clean")
assert r.returncode == 1, f"Expected 1, got {r.returncode}\n{r.stdout}\n{r.stderr}"
assert "FAIL clean" in r.stdout, f"Missing FAIL clean in:\n{r.stdout}"
def test_require_clean_passes_with_untracked_only(self) -> None:
"""``--require-clean`` passes when only untracked files exist."""
# Create an untracked file
self.repo.joinpath("scratch.py").write_text("# untracked\n")
r = _run_verify(self.repo, "--require-clean")
assert r.returncode == 0, f"Expected 0, got {r.returncode}\n{r.stdout}\n{r.stderr}"
assert "PASS clean" in r.stdout, f"Missing PASS clean in:\n{r.stdout}"
# -- known_failures check ----------------------------------------------
def test_known_failures_passes(self) -> None:
"""known_failures check PASSes when the file is empty."""
r = _run_verify(self.repo)
assert "PASS known_failures" in r.stdout, (
f"Missing PASS known_failures in:\n{r.stdout}"
)
# -- commit check ------------------------------------------------------
def test_commit_check_passes(self) -> None:
"""commit check PASSes with a valid HEAD."""
r = _run_verify(self.repo)
assert "PASS commit" in r.stdout, f"Missing PASS commit in:\n{r.stdout}"
# -- output budget -----------------------------------------------------
def test_output_max_40_lines(self) -> None:
"""Output stays within 40 lines."""
r = _run_verify(self.repo)
n = len(r.stdout.splitlines())
assert n <= 40, f"Output has {n} lines, max 40:\n{r.stdout}"
# -- failure detail cap ------------------------------------------------
def test_failure_detail_capped_at_20_lines(self) -> None:
"""25 failing tests produce exactly 20 lines: 18 FAILED + 1 truncation + 1 count."""
tests = self.repo / "tests"
lines = []
for i in range(25):
lines.append(f"def test_fail_{i:02d}():\n assert False\n")
(tests / "test_many_fail.py").write_text("\n".join(lines))
_add_commit(self.repo, "25 failures")
r = _run_verify(self.repo)
assert r.returncode == 1
assert "FAIL tests" in r.stdout
stripped = [l.strip() for l in r.stdout.splitlines()]
detail_lines = [l for l in stripped
if l.startswith(("FAILED ", "ERROR ", "... and"))
or ("failed" in l.lower() and "passed" in l.lower())]
assert len(detail_lines) == 20, (
f"Expected 20 detail lines, got {len(detail_lines)}:\n{r.stdout}"
)
assert any("... and " in l for l in detail_lines)
def test_error_prefix_matches_collection_errors(self) -> None:
"""A pytest collection ERROR line (no trailing S) is captured."""
tests = self.repo / "tests"
# Syntax error triggers a collection ERROR (not FAILED)
(tests / "test_bad_syntax.py").write_text("def this is not valid python:\n")
_add_commit(self.repo, "syntax error")
r = _run_verify(self.repo)
assert r.returncode == 1
assert "FAIL tests" in r.stdout
stripped = [l.strip() for l in r.stdout.splitlines()]
assert any("ERROR " in l for l in stripped), (
f"Expected 'ERROR ' in output:\n{r.stdout}"
)
# ---------------------------------------------------------------------------
# CLI argument parsing
# ---------------------------------------------------------------------------
class TestArgParse:
"""Unit tests for argument parsing (no side effects)."""
def test_defaults(self) -> None:
"""Default values are sensible."""
from scripts.verify_commit import parse_args
args = parse_args(["HEAD"])
assert args.repo == "."
assert args.base is None # resolved to SHA^ in main()
assert args.full is False
assert args.require_clean is False
assert args.no_lint is False
assert args.python == sys.executable
assert args.ruff_cmd == "uvx ruff@0.16.9"
assert args.sha == "HEAD"
def test_all_flags(self) -> None:
"""All flags can be overridden."""
from scripts.verify_commit import parse_args
args = parse_args([
"--repo", "/tmp/x",
"--base", "main",
"--full",
"--require-clean",
"--no-lint",
"--python", "/usr/bin/python3",
"--ruff-cmd", "/home/me/.local/bin/ruff",
"HEAD",
])
assert args.repo == "/tmp/x"
assert args.base == "main"
assert args.full is True
assert args.require_clean is True
assert args.no_lint is True
assert args.python == "/usr/bin/python3"
assert args.ruff_cmd == "/home/me/.local/bin/ruff"
assert args.sha == "HEAD"
# ---------------------------------------------------------------------------
# Lint check: _strip_coords & new_findings
# ---------------------------------------------------------------------------
class TestStripCoords:
"""Pure-function tests for ``_strip_coords``."""
def test_removes_line_and_column(self) -> None:
from scripts.verify_commit import _strip_coords
result = _strip_coords("foo.py:10:5: F401 `os` imported but unused")
assert result == "foo.py: F401 `os` imported but unused"
def test_removes_only_line(self) -> None:
from scripts.verify_commit import _strip_coords
result = _strip_coords("bar.py:3:1: E302 expected 2 blank lines, found 1")
assert result == "bar.py: E302 expected 2 blank lines, found 1"
def test_returns_line_unmodified_if_no_coords(self) -> None:
from scripts.verify_commit import _strip_coords
result = _strip_coords("some random output")
assert result == "some random output"
def test_handles_colon_in_message(self) -> None:
from scripts.verify_commit import _strip_coords
result = _strip_coords("x.py:1:1: WPS111 Found module with too many imports: 42")
assert result == "x.py: WPS111 Found module with too many imports: 42"
def test_handles_concise_format(self) -> None:
"""_strip_coords handles concise format."""
from scripts.verify_commit import _strip_coords
result = _strip_coords("scripts/foo.py:10:5: F401 unused import")
assert result == "scripts/foo.py: F401 unused import"
def test_returns_short_lines_unchanged(self) -> None:
"""_strip_coords returns short lines unchanged (e.g. ruff full output coords)."""
from scripts.verify_commit import _strip_coords
result = _strip_coords(" --> scripts/foo.py:190:14")
assert result == "--> scripts/foo.py:190:14"
def test_returns_plain_lines_unchanged(self) -> None:
"""_strip_coords returns plain lines unchanged."""
from scripts.verify_commit import _strip_coords
result = _strip_coords("Found 1 error.")
assert result == "Found 1 error."
class TestNewFindings:
"""Pure-function tests for ``new_findings``."""
def test_no_new_findings(self) -> None:
from scripts.verify_commit import new_findings
result = new_findings(
["foo.py:10:5: F401 `os` imported but unused"],
["foo.py:10:5: F401 `os` imported but unused"],
)
assert result == []
def test_ignores_line_col_shifts(self) -> None:
from scripts.verify_commit import new_findings
result = new_findings(
["foo.py:1:1: F401 `os` imported but unused"],
["foo.py:42:7: F401 `os` imported but unused"],
)
assert result == []
def test_new_finding_detected(self) -> None:
from scripts.verify_commit import new_findings
result = new_findings(
["foo.py:10:5: F401 `os` imported but unused"],
["foo.py:10:5: F401 `os` imported but unused",
"foo.py:20:3: E302 expected 2 blank lines, found 1"],
)
assert result == ["foo.py: E302 expected 2 blank lines, found 1"]
def test_issue_removed_in_after_is_not_new(self) -> None:
from scripts.verify_commit import new_findings
result = new_findings(
["foo.py:10:5: F401 `os` imported but unused"],
[],
)
assert result == []
def test_multiple_new_findings_sorted(self) -> None:
from scripts.verify_commit import new_findings
result = new_findings(
[],
["b.py:1:1: E302 expected 2 blank lines",
"a.py:1:1: F401 `os` imported but unused"],
)
assert result == [
"a.py: F401 `os` imported but unused",
"b.py: E302 expected 2 blank lines",
]
def test_ignores_empty_lines(self) -> None:
from scripts.verify_commit import new_findings
result = new_findings(
[],
["", "a.py:1:1: F401 unused import", " "],
)
assert result == ["a.py: F401 unused import"]
def test_lint_dirty_to_clean_passes(self, tmp_path: Path) -> None:
"""Dirty -> Clean = PASS."""
from scripts.verify_commit import check_lint
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
src = repo / "src"
src.mkdir()
f = src / "foo.py"
f.write_text("import os\n")
_add_commit(repo, "initial")
# Mock ruff output: before has issues, after is clean
def ruff_stub(stdin, path):
if "initial" in stdin: # this is a simplification for the stub
return "foo.py:1:1: F401 unused import\nFound 1 error.\n", 1
return "All checks passed!\n", 0
# We need to inject the stub into check_lint.
# check_lint uses subprocess.run(shlex.split(ruff_cmd)...)
# To test this properly without modifying scripts/verify_commit.py,
# we create a real executable script that returns different things based on stdin.
ruff_bin = tmp_path / "ruff_stub"
ruff_bin.write_text("""#!/usr/bin/env python3
import sys
stdin = sys.stdin.read()
if "import os" in stdin:
print("src/foo.py:1:1: F401 unused import")
print("Found 1 error.")
sys.exit(1)
print("All checks passed!")
sys.exit(0)
""")
ruff_bin.chmod(0o755)
# For 'before' (baseline), we provide 'import os'
# For 'after' (current), we provide something else.
# But check_lint reads from git show.
# Let's just use the real check_lint logic but control the files.
# Commit 1: dirty
f.write_text("import os\n")
_add_commit(repo, "dirty")
# Commit 2: clean
f.write_text("print('hello')\n")
_add_commit(repo, "clean")
# We need the ruff_bin to actually act as ruff.
# Since check_lint uses subprocess.run, we pass ruff_bin as ruff_cmd.
status, _detail = check_lint(
["src/foo.py"], str(ruff_bin), str(repo), "HEAD~1", "HEAD"
)
assert status == "PASS"
def test_lint_summary_lines_ignored(self, tmp_path: Path) -> None:
"""3 findings -> 1 finding (ignoring summary lines) = PASS if the finding is the same."""
from scripts.verify_commit import check_lint
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
src = repo / "src"
src.mkdir()
f = src / "foo.py"
f.write_text("import os\n")
_add_commit(repo, "initial")
ruff_bin = tmp_path / "ruff_summary"
# This stub returns 3 lines (1 real, 2 summary) for baseline
# and 1 line (1 real, 0 summary) for current.
# Or rather, it should just return the findings.
ruff_bin.write_text("""#!/usr/bin/env python3
import sys
stdin = sys.stdin.read()
if "baseline" in stdin:
print("src/foo.py:1:1: F401 unused")
print("Found 3 errors.")
print("[*] 1 fixable")
sys.exit(1)
if "current" in stdin:
print("src/foo.py:1:1: F401 unused")
print("Found 1 error.")
sys.exit(1)
sys.exit(0)
""")
ruff_bin.chmod(0o755)
# We need to trick git show.
# Since check_lint uses `_git("show", f"{base}:{path}", repo=repo)`,
# we can't easily mock the output of git show without mocking _git.
# However, we can mock the content by adding a marker to the files.
f.write_text("baseline\n")
_add_commit(repo, "baseline")
f.write_text("current\n")
_add_commit(repo, "current")
status, _detail = check_lint(
["src/foo.py"], str(ruff_bin), str(repo), "HEAD~1", "HEAD"
)
assert status == "PASS"
def test_lint_new_finding_fails(self, tmp_path: Path) -> None:
"""Clean -> One new finding = FAIL and detail names it."""
from scripts.verify_commit import check_lint
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
src = repo / "src"
src.mkdir()
f = src / "foo.py"
f.write_text("print('clean')\n")
_add_commit(repo, "initial")
ruff_bin = tmp_path / "ruff_new"
ruff_bin.write_text("""#!/usr/bin/env python3
import sys
stdin = sys.stdin.read()
if "clean" in stdin:
print("All checks passed!")
sys.exit(0)
if "dirty" in stdin:
print("src/foo.py:5:1: E302 expected 2 blank lines")
print("Found 1 error.")
sys.exit(1)
sys.exit(0)
""")
ruff_bin.chmod(0o755)
f.write_text("dirty\n")
_add_commit(repo, "make dirty")
status, _detail = check_lint(
["src/foo.py"], str(ruff_bin), str(repo), "HEAD~1", "HEAD"
)
assert status == "FAIL"
assert any("E302" in d for d in _detail)
def test_lint_shifted_line_passes(self, tmp_path: Path) -> None:
"""Same finding on a shifted line = PASS."""
from scripts.verify_commit import check_lint
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
src = repo / "src"
src.mkdir()
f = src / "foo.py"
f.write_text("baseline\n")
_add_commit(repo, "initial")
ruff_bin = tmp_path / "ruff_shift"
ruff_bin.write_text("""#!/usr/bin/env python3
import sys
stdin = sys.stdin.read()
if "baseline" in stdin:
print("src/foo.py:1:1: F401 unused")
sys.exit(1)
if "current" in stdin:
print("src/foo.py:10:1: F401 unused")
sys.exit(1)
sys.exit(0)
""")
ruff_bin.chmod(0o755)
f.write_text("current\n")
_add_commit(repo, "shift")
status, _detail = check_lint(
["src/foo.py"], str(ruff_bin), str(repo), "HEAD~1", "HEAD"
)
assert status == "PASS"
class TestCheckLint:
"""Tests for ``check_lint`` with a fake ruff binary."""
def test_no_changed_files(self) -> None:
from scripts.verify_commit import check_lint
status, detail = check_lint([], "ruff", "/tmp", "HEAD~1", "HEAD")
assert status == "PASS"
assert "Nothing to lint" in detail[0]
def test_no_python_files(self) -> None:
from scripts.verify_commit import check_lint
status, detail = check_lint(["README.md"], "ruff", "/tmp", "HEAD~1", "HEAD")
assert status == "PASS"
assert "No Python files to lint" in detail[0]
def test_skip_when_ruff_not_found(self, tmp_path: Path) -> None:
"""When ruff is not on PATH, check_lint returns SKIP."""
from scripts.verify_commit import check_lint
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
(repo / "src" / "foo.py").parent.mkdir(exist_ok=True)
(repo / "src" / "foo.py").write_text("import os\n")
_add_commit(repo, "initial")
status, detail = check_lint(
["src/foo.py"], "nonexistent-rufff", str(repo), "HEAD~1", "HEAD"
)
assert status == "SKIP"
assert "not found" in detail[0].lower()
def test_skip_on_exit_127(self, tmp_path: Path) -> None:
"""Simulate a ruff that exits 127 (e.g. ruff@version not installed)."""
from scripts.verify_commit import check_lint
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
(repo / "src").mkdir(exist_ok=True)
(repo / "src" / "foo.py").write_text("import os\n")
_add_commit(repo, "initial")
fake_ruff = tmp_path / "fake_ruff_127"
fake_ruff.write_text("#!/bin/sh\nexit 127\n")
fake_ruff.chmod(0o755)
status, detail = check_lint(
["src/foo.py"], str(fake_ruff), str(repo), "HEAD~1", "HEAD"
)
assert status == "SKIP"
assert "127" in detail[0]
def test_integration_with_fake_ruff(self, tmp_path: Path) -> None:
"""Use a fake ruff script to verify check_lint reports new findings."""
from scripts.verify_commit import check_lint
# Set up a small git repo with one commit
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
src = repo / "src"
src.mkdir()
a_py = src / "a.py"
a_py.write_text("import os\nimport sys\n")
_add_commit(repo, "initial")
# Now modify to add a new issue
a_py.write_text("import os\nimport sys\nimport pathlib\n")
_add_commit(repo, "add import")
# Create a fake ruff that reports F401 on pathlib
ruff_bin = tmp_path / "ruff"
ruff_bin.write_text("""\
#!/usr/bin/env python3
import sys
stdin = sys.stdin.read() if not sys.stdin.isatty() else ""
args = sys.argv[1:] # ["check", ...]
if "--stdin-filename" in args:
idx = args.index("--stdin-filename")
path = args[idx + 1]
if "a.py" in path and "pathlib" in stdin:
print(f"{path}:3:1: F401 `pathlib` imported but unused")
sys.exit(1 if "pathlib" in stdin else 0)
else:
# ruff check <file> — skip subcommand
py_files = [a for a in args if a.endswith(".py")]
path = py_files[0] if py_files else ""
if "a.py" in path:
print(f"{path}:3:1: F401 `pathlib` imported but unused")
sys.exit(1)
""")
ruff_bin.chmod(0o755)
status, detail = check_lint(
["src/a.py"], str(ruff_bin), str(repo), "HEAD~1", "HEAD"
)
assert status == "FAIL", f"Expected FAIL, got {status}: {detail}"
assert any("F401" in d for d in detail), f"Expected F401 in detail: {detail}"
def test_fake_ruff_passes_no_new_issues(self, tmp_path: Path) -> None:
"""When baseline and current both have same issues, check_lint PASSes."""
from scripts.verify_commit import check_lint
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
src = repo / "src"
src.mkdir()
# Write code that already has import os
a_py = src / "a.py"
a_py.write_text("import os\nimport sys\n")
_add_commit(repo, "initial")
# Same content, just cosmetic change
a_py.write_text("import os\nimport sys\n# comment\n")
_add_commit(repo, "add comment")
# Create a fake ruff that reports F401 for both baseline and current
ruff_bin = tmp_path / "noop-ruff"
ruff_bin.write_text("""\
#!/usr/bin/env python3
import sys
stdin = sys.stdin.read() if not sys.stdin.isatty() else ""
if stdin:
if "import os" in stdin:
sys.exit(0)
if len(sys.argv) > 1:
path = [a for a in sys.argv[1:] if not a.startswith("-")][0]
if "a.py" in path:
sys.exit(0)
sys.exit(0)
""")
ruff_bin.chmod(0o755)
status, detail = check_lint(
["src/a.py"], str(ruff_bin), str(repo), "HEAD~1", "HEAD"
)
assert status == "PASS", f"Expected PASS, got {status}: {detail}"
# ---------------------------------------------------------------------------
# SHA positional argument — exit 2, older-SHA verification, --base/--full
# ---------------------------------------------------------------------------
class TestSHAPositional:
"""Tests for the SHA positional argument and git-archive-based checks."""
@pytest.fixture(autouse=True)
def _setup(self, tmp_path: Path) -> None:
"""Build a temp repo with one good commit."""
repo = tmp_path / "repo"
repo.mkdir()
_init_repo(repo)
src = repo / "src"
tests = repo / "tests"
src.mkdir()
tests.mkdir()
(src / "good.py").write_text("def good():\n return True\n")
(tests / "test_good.py").write_text(
"from src.good import good\n\ndef test_good():\n"
" assert good() is True\n"
)
_add_commit(repo, "good: passes")
self.repo = repo
def test_sha_positional_HEAD_passes(self) -> None:
"""``verify_commit.py --repo <repo> HEAD`` exits 0 on a good repo."""
r = subprocess.run(
[
sys.executable, str(VERIFY_COMMIT_PY),
"--repo", str(self.repo),
"--no-lint",
"HEAD",
],
capture_output=True, text=True, timeout=60, check=False,
)
assert r.returncode == 0, f"Expected 0, got {r.returncode}\n{r.stdout}\n{r.stderr}"
assert "PASS tests" in r.stdout
def test_unresolvable_sha_exits_2(self) -> None:
"""A non-existent SHA exits 2 with a usage-style message."""
r = subprocess.run(
[
sys.executable, str(VERIFY_COMMIT_PY),
"--repo", str(self.repo),
"--no-lint",
"deadbeef",
],
capture_output=True, text=True, timeout=60, check=False,
)
assert r.returncode == 2, f"Expected 2, got {r.returncode}\n{r.stdout}\n{r.stderr}"
def test_older_sha_while_head_broken(self) -> None:
"""Verifying an OLDER SHA exits 0 even when HEAD is broken."""
# Create a second commit with a failing test
(self.repo / "src" / "good.py").write_text("def good():\n return False\n")
_add_commit(self.repo, "bad: test fails")
# Get the SHA of the first (good) commit
r = _git(self.repo, "rev-parse", "HEAD~1")
first_sha = r.stdout.strip()
# Verify the older SHA — should pass even though HEAD is broken
r = subprocess.run(
[
sys.executable, str(VERIFY_COMMIT_PY),
"--repo", str(self.repo),
"--no-lint",
first_sha,
],
capture_output=True, text=True, timeout=60, check=False,
)
assert r.returncode == 0, (
f"Expected 0 (older SHA is good), got {r.returncode}\n{r.stdout}\n{r.stderr}"
)
assert "PASS tests" in r.stdout
def test_require_clean_with_explicit_sha(self) -> None:
"""``--require-clean HEAD`` works with the positional SHA."""
# Clean state: all good, exit 0
r = subprocess.run(
[
sys.executable, str(VERIFY_COMMIT_PY),
"--repo", str(self.repo),
"--require-clean",
"--no-lint",
"HEAD",
],
capture_output=True, text=True, timeout=60, check=False,
)
assert r.returncode == 0, f"Expected 0, got {r.returncode}\n{r.stdout}\n{r.stderr}"
# Dirty state: modify a tracked file without committing
(self.repo / "src" / "good.py").write_text("def good():\n return True # edited\n")
r = subprocess.run(
[
sys.executable, str(VERIFY_COMMIT_PY),
"--repo", str(self.repo),
"--require-clean",
"--no-lint",
"HEAD",
],
capture_output=True, text=True, timeout=60, check=False,
)
assert r.returncode == 1, f"Expected 1, got {r.returncode}\n{r.stdout}\n{r.stderr}"
def test_full_with_base_and_sha(self) -> None:
"""``--base <older> --full HEAD`` runs full suite against HEAD."""
# Modify code so the test still passes
(self.repo / "src" / "good.py").write_text("def good():\n return True\n")
_add_commit(self.repo, "refactor: still passes")
_head_sha = _git(self.repo, "rev-parse", "HEAD").stdout.strip()
r = subprocess.run(
[
sys.executable, str(VERIFY_COMMIT_PY),
"--repo", str(self.repo),
"--base", "HEAD~1",
"--full",
"--no-lint",
"HEAD",
],
capture_output=True, text=True, timeout=60, check=False,
)
assert r.returncode == 0, f"Expected 0, got {r.returncode}\n{r.stdout}\n{r.stderr}"
def test_base_defaults_to_parent(self) -> None:
"""When --base is omitted, the default is SHA^."""
# First commit is good, second commit modifies code but tests still pass
(self.repo / "src" / "good.py").write_text("def good():\n return True\n")
_add_commit(self.repo, "good2")
# Run without --base — should default to HEAD~1
r = subprocess.run(
[
sys.executable, str(VERIFY_COMMIT_PY),
"--repo", str(self.repo),
"--no-lint",
"HEAD",
],
capture_output=True, text=True, timeout=60, check=False,
)
assert r.returncode == 0, f"Expected 0, got {r.returncode}\n{r.stdout}\n{r.stderr}"
assert "PASS tests" in r.stdout