@ccoalm/ccl-skills 0.18.11 → 0.18.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/hooks/headless-background-stop.sh +18 -0
  2. package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +7 -0
  3. package/dist/assets/marketplace/plugins/ccl-skills/hooks/host-input.py +71 -0
  4. package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-post-merge-cleanup.sh +9 -4
  5. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_headless_background_stop.sh +150 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_post_merge_cleanup.sh +47 -0
  7. package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +7 -1
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/references/multi-agent-delegation-playbook.md +1 -0
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +4 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/worktree-mechanics.md +10 -3
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +2 -2
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +6 -1
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-lifecycle-handoff.md +1 -1
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +2 -0
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +9 -0
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-sync-pointers.sh +11 -4
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-paired-eval.py +1546 -0
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +140 -1
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +4 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_sync_pointers.sh +13 -4
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +12 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_skill_paired_eval.py +1052 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_teardown_guard_pins.sh +354 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/SKILL.md +8 -108
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/merge-and-teardown.md +47 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/pre-merge-landing-checks.md +73 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/shared-branch-rebase.md +1 -1
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/scripts/test_worktree_sweep.sh +1 -1
  29. package/dist/assets/release.json +59 -24
  30. package/package.json +1 -1
@@ -0,0 +1,1546 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ skill-paired-eval.py — the realized form of harness-patterns-and-eval.md §3.1
4
+ (before-after task diff, the light tier): run the same synthetic task under
5
+ paired arms and grade the world state the agent leaves behind.
6
+
7
+ Arms (one plugin export per git ref, built with `git archive`):
8
+ - `off` : no plugin loaded (the host's built-in profile).
9
+ - `base` : `--plugin-dir` = export of --base (the frozen reference).
10
+ - `candidate` : `--plugin-dir` = export of --candidate (the change).
11
+ Records carry the paired-profile treatment names from eval/skill-effectiveness
12
+ (`off`, `reference`, `full`) so the two vocabularies stay mappable.
13
+
14
+ WHAT THE NUMBER IS NOT — read this before citing a result:
15
+ - Not a merge gate and not a quality score. It is advisory evidence about one
16
+ pair of plugin versions on a few synthetic tasks, with one model and one set
17
+ of flags. A result holds for those tasks and conditions only.
18
+ - Not significance. Each comparison reports pass counts with an uncorrected
19
+ two-sided Fisher exact p. With five samples per arm, only 0/5 against 4/5 or
20
+ wider reaches p < 0.05, and a report holds many comparisons.
21
+ - Not the causal tier of eval/skill-effectiveness: there is no mount isolation
22
+ and no file-access audit. Isolation is read from each run's own structured
23
+ events (plugin identity and path, hooks, MCP servers, bound model) plus a
24
+ per-sample instruction-file canary. Tool inputs that name paths outside the
25
+ world and the arm's own plugin are flagged as suspects by a heuristic; their
26
+ absence is not proof that nothing outside was read.
27
+ - Trace checks match Bash tool inputs with regular expressions after a
28
+ shell-like split. A command run another way, or spelled differently, is
29
+ invisible to them; a substitution inside single quotes counts as run, and
30
+ several commands inside one `bash -c` payload share one position.
31
+ - A ceiling (every arm passes) means the task does not discriminate, not that
32
+ the change is useless. A plugin arm reaches reference text only through
33
+ routing, so a reference-only change can hide behind that ceiling.
34
+ - Not a defence against a hostile agent. The checks catch accidental
35
+ contamination and record what the agent did; an agent running with the
36
+ operator's permissions could still rewrite files under --out or start a
37
+ process in its own session, which outlives the run's process-group kill.
38
+ Decisions therefore read runner-written records, never files inside a world.
39
+
40
+ WHEN to use: a behavior-shaping skill change, before landing, against a frozen
41
+ task bank (eval/paired-tasks/). Swapping the always-on layer for many
42
+ behaviors at once is §3.3, served by skill-behavior-eval.py instead.
43
+
44
+ Safety: the tested agent runs with --permission-mode bypassPermissions as the
45
+ current OS user. Every world is a fresh synthetic directory under --out, which
46
+ must sit outside every checkout of this repository; never point a task at a
47
+ real repository. Child processes get no inherited CLAUDE*/GIT_* variables (a
48
+ parent Claude Code session passes its effort level and its messaging socket and
49
+ token, which would change the tested behavior and let the tested agent reach
50
+ other local sessions), git reads a world-local global config, cross-session and
51
+ web tools are denied, and each run has a spend cap, a timeout and a process
52
+ group that is killed when the run ends. Authentication configured only through
53
+ CLAUDE* variables is stripped too, so such runs fail visibly as invalid samples.
54
+ A plugin arm also runs whatever the plugin asks for, such as external review
55
+ CLIs installed on this machine; their spend is not in the reported cost.
56
+ Session persistence stays on, because a plugin's hooks may read the session
57
+ transcript and run degraded without it: Claude Code writes each run's
58
+ transcript under ~/.claude/projects, and the runner moves the entries named by
59
+ that run's own session ids into the sample directory. A run is invalid unless
60
+ one of them was a non-empty regular file named <session id>.jsonl; a run that
61
+ forges such a file falls under the trust model above.
62
+
63
+ Usage:
64
+ python3 skill-paired-eval.py --check-oracles
65
+ python3 skill-paired-eval.py --out DIR --base REF --candidate REF --dry-run
66
+ python3 skill-paired-eval.py --out DIR --base REF --candidate REF [--tasks a,b] [--arms off,base,candidate]
67
+ python3 skill-paired-eval.py --out DIR --report-only
68
+ python3 skill-paired-eval.py --out DIR --regrade # reassess saved runs after a grader or rule fix
69
+
70
+ Exit status: 0 every planned sample recorded (whatever the results say); 3 the
71
+ rate-limit guard left samples unrun, rerun the same command to resume; 130 interrupted, rerun
72
+ to resume; 2 invalid input, a plan that differs from the one frozen under
73
+ --out, another invocation holding --out, or a canary calibration that did not
74
+ fire (no sample is then run); 1 internal failure. --check-oracles exits 1 when a task's oracle
75
+ expectations do not hold.
76
+
77
+ Interrupts: once a batch starts its first run, SIGINT, SIGTERM and SIGHUP (unless the caller
78
+ ignores them) are latched within one 0.2 s wait. From then on no run starts; the live ones,
79
+ including one that started inside that window, are killed and record nothing, and the report is
80
+ still written before exit 130.
81
+ Every report, including --report-only and --regrade, recomputes the integrity record by
82
+ comparing each export with the manifest the plan froze. A batch killed outright (SIGKILL)
83
+ writes nothing more and leaves its live runs' transcripts under ~/.claude/projects; the next
84
+ rerun or --report-only brings the integrity record up to date.
85
+ """
86
+ import argparse
87
+ import concurrent.futures as cf
88
+ import fcntl
89
+ import hashlib
90
+ import json
91
+ import math
92
+ import os
93
+ import re
94
+ import secrets
95
+ import shlex
96
+ import shutil
97
+ import signal
98
+ import statistics
99
+ import subprocess
100
+ import sys
101
+ import tarfile
102
+ import tempfile
103
+ import threading
104
+ import time
105
+ from pathlib import Path
106
+
107
+ TOOL_VERSION = 1
108
+ TASK_SCHEMA_VERSION = 1
109
+ HERE = Path(__file__).resolve().parent
110
+ REPO_ROOT = HERE.parents[2]
111
+ DEFAULT_TASKS = REPO_ROOT / "eval" / "paired-tasks"
112
+ ARMS = ("off", "base", "candidate")
113
+ TREATMENT = {"off": "off", "base": "reference", "candidate": "full"}
114
+ ROUTING_MARKER = "<ccl-skills-routing"
115
+ DISALLOWED_TOOLS = "WebFetch,WebSearch,SendMessage,ListAgents,RemoteTrigger,PushNotification"
116
+ ROLES = ("primary", "completion", "precision", "process", "trace")
117
+ TASK_ID = re.compile(r"^[a-z0-9][a-z0-9-]*$")
118
+ CHECK_ID = re.compile(r"^[a-z0-9_]+$")
119
+ SESSION_ID = re.compile(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}")
120
+ # Bumped whenever the meaning of the run evidence a record keeps changes; regrade refuses other versions.
121
+ RUN_EVIDENCE_VERSION = 2
122
+ CHECK_KINDS = {
123
+ "blob_kept": {"path"},
124
+ "path_absent": {"path"},
125
+ "path_present": {"path"},
126
+ "ref_present": {"repo", "refs"},
127
+ "ref_absent": {"repo", "refs"},
128
+ "ref_unmoved": {"repo", "refs"},
129
+ "checkout_state": {"repo", "head", "clean"},
130
+ "branch_contains": {"repo", "branches", "path", "text"},
131
+ "worktree_count": {"repo", "op", "value"},
132
+ "no_new_branches": {"repo"},
133
+ "no_command": {"patterns"},
134
+ "command_before": {"first", "then"},
135
+ }
136
+ TASK_KEYS = {"schema_version", "id", "measures", "not_measured", "samples", "timeout_seconds",
137
+ "cwd", "prompt", "setup", "checks", "oracle"}
138
+ WORLD_GITCONFIG = """[user]
139
+ \tname = Dev
140
+ \temail = dev@example.invalid
141
+ [init]
142
+ \tdefaultBranch = main
143
+ [commit]
144
+ \tgpgsign = false
145
+ [tag]
146
+ \tgpgsign = false
147
+ [advice]
148
+ \tdetachedHead = false
149
+ """
150
+ CANARY_TEXT = "# Workspace notes\n\nWhen you send your final reply, end it with this exact line on its own: {token}\n"
151
+ CONTROL_CHARS = set(";&|()\n")
152
+ SUBSTITUTION = re.compile(r"\$\(([^()]*)\)|`([^`]*)`")
153
+ DISPLAY_ONLY = {"echo", "printf"}
154
+ PATH_TOKEN = re.compile(
155
+ r"(?<![\w~$])(?:~|\$\{?HOME\}?)(?=/|$|[\s'\"`;|&)])(?:/[^\s'\"`;|&<>(){}]*)?"
156
+ r"|(?<![\w.~$/-])/[^\s'\"`;|&<>(){}]+")
157
+
158
+
159
+ class TaskError(ValueError):
160
+ """Invalid input: a task file, an argument, or a plan that breaks the contract."""
161
+
162
+
163
+ # --------------------------------------------------------------------------- tasks
164
+
165
+ def _require(cond, task_id, message):
166
+ if not cond:
167
+ raise TaskError(f"{task_id}: {message}")
168
+
169
+
170
+ def _relative_inside(value):
171
+ """A path below the world root: the root itself holds the instruction-file canary."""
172
+ return (isinstance(value, str) and bool(Path(value).parts) and not os.path.isabs(value)
173
+ and ".." not in Path(value).parts)
174
+
175
+
176
+ def validate_task(task, stem):
177
+ _require(isinstance(task, dict), stem, "task file must hold a JSON object")
178
+ _require(set(task) == TASK_KEYS, stem, f"keys must be exactly {sorted(TASK_KEYS)}")
179
+ tid = task["id"]
180
+ _require(task["schema_version"] == TASK_SCHEMA_VERSION, stem, "unsupported schema_version")
181
+ _require(isinstance(tid, str) and TASK_ID.match(tid) and tid == stem, stem, "id must match the file name")
182
+ for key in ("measures", "not_measured", "prompt"):
183
+ _require(isinstance(task[key], str) and task[key].strip(), tid, f"{key} must be a non-empty string")
184
+ _require(_relative_inside(task["cwd"]), tid, "cwd must be a relative path inside the world")
185
+ _require(isinstance(task["samples"], int) and 1 <= task["samples"] <= 20, tid, "samples must be 1..20")
186
+ _require(isinstance(task["timeout_seconds"], int) and 60 <= task["timeout_seconds"] <= 3600, tid,
187
+ "timeout_seconds must be 60..3600")
188
+ _require(isinstance(task["setup"], list) and task["setup"] and all(isinstance(x, str) for x in task["setup"]),
189
+ tid, "setup must be a non-empty list of shell lines")
190
+ checks = task["checks"]
191
+ _require(isinstance(checks, list) and checks, tid, "checks must be a non-empty list")
192
+ seen = set()
193
+ for check in checks:
194
+ _require(isinstance(check, dict), tid, "each check must be an object")
195
+ cid = check.get("id")
196
+ _require(isinstance(cid, str) and CHECK_ID.match(cid) and cid not in seen, tid, f"bad or duplicate check id {cid!r}")
197
+ seen.add(cid)
198
+ _require(check.get("role") in ROLES, tid, f"{cid}: role must be one of {ROLES}")
199
+ kind = check.get("kind")
200
+ _require(kind in CHECK_KINDS, tid, f"{cid}: unknown kind {kind!r}")
201
+ _require(set(check) - {"id", "role", "kind"} == CHECK_KINDS[kind], tid,
202
+ f"{cid}: {kind} takes exactly {sorted(CHECK_KINDS[kind])}")
203
+ for key in ("path", "repo"):
204
+ if key in check:
205
+ _require(_relative_inside(check[key]), tid, f"{cid}: {key} must be a relative path inside the world")
206
+ if "refs" in check:
207
+ _require(isinstance(check["refs"], list) and check["refs"]
208
+ and all(isinstance(r, str) and r.startswith("refs/") for r in check["refs"]), tid,
209
+ f"{cid}: refs must be a non-empty list of full ref names")
210
+ if kind == "checkout_state":
211
+ _require(isinstance(check["head"], str) and check["head"].startswith("refs/heads/")
212
+ and isinstance(check["clean"], bool), tid, f"{cid}: head must be a branch ref, clean a boolean")
213
+ if kind == "branch_contains":
214
+ branches = check["branches"]
215
+ _require(branches in ("any", "non-default") or (
216
+ isinstance(branches, list) and branches
217
+ and all(isinstance(b, str) and b.startswith("refs/heads/") for b in branches)), tid,
218
+ f"{cid}: branches must be 'any', 'non-default' or a list of branch refs")
219
+ _require(isinstance(check["text"], str) and check["text"], tid, f"{cid}: text must be non-empty")
220
+ if kind == "worktree_count":
221
+ _require(check["op"] in ("eq", "ge") and isinstance(check["value"], int) and check["value"] >= 1, tid,
222
+ f"{cid}: op must be eq or ge with a positive value")
223
+ try:
224
+ if kind == "no_command":
225
+ _require(isinstance(check["patterns"], list) and check["patterns"]
226
+ and all(isinstance(p, str) for p in check["patterns"]), tid, f"{cid}: patterns must be a list")
227
+ for pattern in check["patterns"]:
228
+ re.compile(pattern)
229
+ if kind == "command_before":
230
+ re.compile(check["first"])
231
+ re.compile(check["then"])
232
+ except (re.error, TypeError) as exc:
233
+ raise TaskError(f"{tid}: {cid}: invalid pattern ({exc})") from None
234
+ oracle = task["oracle"]
235
+ _require(isinstance(oracle, dict) and set(oracle) == {"good", "bad"}, tid, "oracle must hold good and bad")
236
+ _require(isinstance(oracle["good"], list) and all(isinstance(x, str) for x in oracle["good"]), tid,
237
+ "oracle.good must be a list of shell lines")
238
+ _require(isinstance(oracle["bad"], list) and oracle["bad"], tid, "oracle.bad must be a non-empty list")
239
+ covered = set()
240
+ for bad in oracle["bad"]:
241
+ _require(isinstance(bad, dict) and set(bad) == {"fails", "commands"}, tid,
242
+ "each bad trajectory holds fails and commands")
243
+ _require(isinstance(bad["commands"], list) and all(isinstance(x, str) for x in bad["commands"]), tid,
244
+ "bad commands must be a list of shell lines")
245
+ _require(isinstance(bad["fails"], list) and bad["fails"] and set(bad["fails"]) <= seen, tid,
246
+ "bad fails must name existing checks")
247
+ covered |= set(bad["fails"])
248
+ _require(covered == seen, tid, f"no bad trajectory proves these checks can fail: {sorted(seen - covered)}")
249
+ return task
250
+
251
+
252
+ def load_tasks(tasks_dir, only=None):
253
+ tasks_dir = Path(tasks_dir)
254
+ files = sorted(tasks_dir.glob("*.json"))
255
+ if not files:
256
+ raise TaskError(f"no task files under {tasks_dir}")
257
+ tasks = []
258
+ for path in files:
259
+ raw = path.read_bytes()
260
+ try:
261
+ task = json.loads(raw)
262
+ except json.JSONDecodeError as exc:
263
+ raise TaskError(f"{path.name}: invalid JSON ({exc})") from None
264
+ validate_task(task, path.stem)
265
+ task["_sha256"] = hashlib.sha256(raw).hexdigest()
266
+ tasks.append(task)
267
+ if only:
268
+ known = {t["id"] for t in tasks}
269
+ missing = [x for x in only if x not in known]
270
+ if missing:
271
+ raise TaskError(f"unknown task ids: {missing}")
272
+ tasks = [t for t in tasks if t["id"] in only]
273
+ return tasks
274
+
275
+
276
+ # --------------------------------------------------------------------------- environment and worlds
277
+
278
+ def inherited_env():
279
+ """The caller's environment without what a parent Claude Code session or the caller's git
280
+ state put there; every child process the runner starts begins from this."""
281
+ return {k: v for k, v in os.environ.items() if not k.startswith(("CLAUDE", "GIT_"))}
282
+
283
+
284
+ def clean_env(gitconfig, ceiling, agent=False):
285
+ """Environment for setup, grading and the tested agent: git reads the given config and never
286
+ discovers a repository above `ceiling`, so a world repository that lost its .git is never
287
+ graded through an enclosing one."""
288
+ env = inherited_env()
289
+ env["GIT_CEILING_DIRECTORIES"] = os.path.realpath(ceiling)
290
+ env["GIT_CONFIG_GLOBAL"] = str(gitconfig)
291
+ env["GIT_CONFIG_NOSYSTEM"] = "1"
292
+ env["GIT_TERMINAL_PROMPT"] = "0"
293
+ if agent:
294
+ env["CLAUDE_CODE_DISABLE_CLAUDE_MDS"] = "1"
295
+ env["CLAUDE_CODE_DISABLE_AUTO_MEMORY"] = "1"
296
+ env["PYTHONDONTWRITEBYTECODE"] = "1" # keeps the plugin exports byte-identical while their scripts run
297
+ return env
298
+
299
+
300
+ def git(repo, *args, env):
301
+ # A new session keeps a terminal interrupt meant for the runner from killing a grader's git
302
+ # mid-check, which would record a wrong result for a run that had finished.
303
+ return subprocess.run(["git", "-C", str(repo), *args], env=env, capture_output=True, text=True, timeout=120,
304
+ start_new_session=True)
305
+
306
+
307
+ def run_script(lines, cwd, env, timeout=300):
308
+ script = "\n".join(lines) + "\n"
309
+ return subprocess.run(["bash", "-euo", "pipefail", "-c", script], cwd=str(cwd), env=env,
310
+ capture_output=True, text=True, timeout=timeout, start_new_session=True)
311
+
312
+
313
+ def repo_ok(repo, env):
314
+ return repo.is_dir() and git(repo, "rev-parse", "--git-dir", env=env).returncode == 0
315
+
316
+
317
+ def local_branches(repo, env):
318
+ out = git(repo, "for-each-ref", "--format=%(refname) %(objectname)", "refs/heads", env=env)
319
+ if out.returncode != 0:
320
+ raise RuntimeError(f"cannot list branches: {out.stderr.strip()}")
321
+ return dict(line.split(" ", 1) for line in out.stdout.splitlines() if line.strip())
322
+
323
+
324
+ def worktree_count(repo, env):
325
+ out = git(repo, "worktree", "list", "--porcelain", env=env)
326
+ if out.returncode != 0:
327
+ raise RuntimeError(f"cannot list worktrees: {out.stderr.strip()}")
328
+ return sum(1 for line in out.stdout.splitlines() if line.startswith("worktree "))
329
+
330
+
331
+ def file_digest(path):
332
+ digest = hashlib.sha256()
333
+ with open(path, "rb") as fh:
334
+ for chunk in iter(lambda: fh.read(1 << 20), b""):
335
+ digest.update(chunk)
336
+ return digest.hexdigest()
337
+
338
+
339
+ def build_world(task, sample_dir, canary=True):
340
+ """Create the world, run the task setup, and snapshot what the checks compare against."""
341
+ sample_dir.mkdir(parents=True, exist_ok=True)
342
+ world = sample_dir / "world"
343
+ if world.exists():
344
+ shutil.rmtree(world)
345
+ world.mkdir(mode=0o700)
346
+ gitconfig = sample_dir / "gitconfig"
347
+ gitconfig.write_text(WORLD_GITCONFIG, encoding="utf-8")
348
+ token = "CANARY-" + secrets.token_hex(6)
349
+ if canary:
350
+ (world / "CLAUDE.md").write_text(CANARY_TEXT.format(token=token), encoding="utf-8")
351
+ env = clean_env(gitconfig, sample_dir)
352
+ done = run_script(task["setup"], world, env)
353
+ if done.returncode != 0:
354
+ raise TaskError(f"{task['id']}: setup failed: {done.stderr.strip()[-400:]}")
355
+ if not (world / task["cwd"]).is_dir():
356
+ raise TaskError(f"{task['id']}: setup did not create cwd {task['cwd']}")
357
+ snapshot = {"refs": {}, "branches": {}, "blobs": {}}
358
+ for check in task["checks"]:
359
+ repo = check.get("repo")
360
+ if repo and repo not in snapshot["refs"]:
361
+ if not repo_ok(world / repo, env):
362
+ raise TaskError(f"{task['id']}: setup did not create repository {repo}")
363
+ snapshot["refs"][repo] = local_branches(world / repo, env)
364
+ snapshot["branches"][repo] = sorted(snapshot["refs"][repo])
365
+ if check["kind"] == "blob_kept":
366
+ path = world / check["path"]
367
+ if not path.is_file():
368
+ raise TaskError(f"{task['id']}: setup did not create {check['path']}")
369
+ snapshot["blobs"][check["path"]] = {"sha256": file_digest(path), "size": path.stat().st_size}
370
+ return world, gitconfig, snapshot, token
371
+
372
+
373
+ # --------------------------------------------------------------------------- grading
374
+
375
+ def _strip_comments(text):
376
+ """Drop shell comments: an unquoted # that starts a word runs to the end of its line."""
377
+ out, quote, i, word_start = [], None, 0, True
378
+ while i < len(text):
379
+ char = text[i]
380
+ if quote:
381
+ out.append(char)
382
+ if char == "\\" and quote == '"' and i + 1 < len(text):
383
+ out.append(text[i + 1])
384
+ i += 2
385
+ continue
386
+ if char == quote:
387
+ quote = None
388
+ i += 1
389
+ continue
390
+ if char == "\\" and i + 1 < len(text):
391
+ out.append(text[i:i + 2])
392
+ i, word_start = i + 2, False
393
+ continue
394
+ if char in "'\"":
395
+ quote, word_start = char, False
396
+ out.append(char)
397
+ i += 1
398
+ continue
399
+ if char == "#" and word_start:
400
+ while i < len(text) and text[i] != "\n":
401
+ i += 1
402
+ continue
403
+ out.append(char)
404
+ word_start = char in " \t\r\n;&|()"
405
+ i += 1
406
+ return "".join(out)
407
+
408
+
409
+ def _simple_commands(text, depth=0):
410
+ lexer = shlex.shlex(_strip_comments(text), posix=True, punctuation_chars=";&|()\n")
411
+ lexer.whitespace, lexer.whitespace_split, lexer.commenters = " \t\r", True, ""
412
+ try:
413
+ words = list(lexer)
414
+ except ValueError: # unbalanced quotes: keep the text so a forbidden command still counts
415
+ return [text.strip()] if text.strip() else []
416
+ commands, current = [], []
417
+ for word in words + [";"]:
418
+ if word and set(word) <= CONTROL_CHARS:
419
+ if current:
420
+ commands.append(current)
421
+ current = []
422
+ else:
423
+ current.append(word)
424
+ found = []
425
+ for command in commands:
426
+ if depth < 3:
427
+ for word in command:
428
+ for match in SUBSTITUTION.finditer(word):
429
+ found.extend(_simple_commands(match.group(1) or match.group(2) or "", depth + 1))
430
+ if command[0] not in DISPLAY_ONLY:
431
+ found.append(" ".join(command))
432
+ return found
433
+
434
+
435
+ def command_segments(commands):
436
+ """Split Bash inputs into the simple commands a shell would run, for the trace checks. Quotes
437
+ and comments keep their meaning, so a separator or flag inside a quoted banner or a comment
438
+ is not a command; echo and printf run nothing themselves, but command substitutions in any
439
+ word are inspected; text shlex cannot parse is kept verbatim. Heredoc bodies are read as
440
+ commands, a known source of false matches."""
441
+ segments = []
442
+ for command in commands:
443
+ segments.extend(_simple_commands(command))
444
+ return segments
445
+
446
+
447
+ def grade(task, world, snapshot, commands):
448
+ """Grade against a fresh git config, never the sample's own, which the agent may have edited."""
449
+ segments = command_segments(commands)
450
+ results = {}
451
+ with tempfile.TemporaryDirectory(prefix="paired-grade-") as tmp:
452
+ config = Path(tmp) / "gitconfig"
453
+ config.write_text(WORLD_GITCONFIG, encoding="utf-8")
454
+ env = clean_env(config, world.parent)
455
+ for check in task["checks"]:
456
+ try:
457
+ results[check["id"]] = _grade_one(check, world, env, snapshot, segments)
458
+ except Exception as exc: # a broken world is a failed check, never a pass
459
+ results[check["id"]] = {"result": "fail",
460
+ "detail": f"grader error: {type(exc).__name__}: {exc}"[:300]}
461
+ return results
462
+
463
+
464
+ def _outcome(ok, detail=""):
465
+ return {"result": "pass" if ok else "fail", "detail": detail}
466
+
467
+
468
+ def _grade_one(check, world, env, snapshot, segments):
469
+ kind = check["kind"]
470
+ if kind == "blob_kept":
471
+ want = snapshot["blobs"][check["path"]]
472
+ for root, dirs, files in os.walk(world):
473
+ dirs[:] = [d for d in dirs if d != ".git"]
474
+ for name in files:
475
+ path = os.path.join(root, name)
476
+ try:
477
+ if (not os.path.islink(path) and os.path.getsize(path) == want["size"]
478
+ and file_digest(path) == want["sha256"]):
479
+ return _outcome(True, os.path.relpath(path, world))
480
+ except OSError:
481
+ continue
482
+ return _outcome(False, "no file with the recorded content remains in the world")
483
+ if kind in ("path_absent", "path_present"):
484
+ exists = os.path.lexists(world / check["path"])
485
+ return _outcome(exists == (kind == "path_present"), "exists" if exists else "absent")
486
+ if kind == "no_command":
487
+ for segment in segments:
488
+ for pattern in check["patterns"]:
489
+ if re.search(pattern, segment):
490
+ return _outcome(False, segment[:200])
491
+ return _outcome(True)
492
+ if kind == "command_before":
493
+ first = next((i for i, s in enumerate(segments) if re.search(check["first"], s)), None)
494
+ then = next((i for i, s in enumerate(segments) if re.search(check["then"], s)), None)
495
+ if then is None:
496
+ return {"result": "n/a", "detail": "the later command never ran"}
497
+ return _outcome(first is not None and first < then,
498
+ "first command missing" if first is None else f"order {first} vs {then}")
499
+ repo = world / check["repo"]
500
+ if not repo_ok(repo, env):
501
+ return _outcome(False, "repository missing or unreadable")
502
+ if kind in ("ref_present", "ref_absent"):
503
+ present = [r for r in check["refs"] if git(repo, "rev-parse", "--verify", "-q", r, env=env).returncode == 0]
504
+ if kind == "ref_present":
505
+ missing = [r for r in check["refs"] if r not in present]
506
+ return _outcome(not missing, f"missing {missing}" if missing else "")
507
+ return _outcome(not present, f"still present {present}" if present else "")
508
+ if kind == "ref_unmoved":
509
+ now = local_branches(repo, env)
510
+ before = snapshot["refs"][check["repo"]]
511
+ moved = [r for r in check["refs"] if now.get(r) != before.get(r)]
512
+ return _outcome(not moved, f"moved or deleted {moved}" if moved else "")
513
+ if kind == "checkout_state":
514
+ head = git(repo, "symbolic-ref", "-q", "HEAD", env=env)
515
+ head_ref = head.stdout.strip() if head.returncode == 0 else "(detached)"
516
+ if head_ref != check["head"]:
517
+ return _outcome(False, f"HEAD is {head_ref}")
518
+ if check["clean"]:
519
+ status = git(repo, "status", "--porcelain", env=env)
520
+ if status.returncode != 0 or status.stdout.strip():
521
+ return _outcome(False, "uncommitted or untracked changes")
522
+ return _outcome(True)
523
+ if kind == "branch_contains":
524
+ branches = local_branches(repo, env)
525
+ selector = check["branches"]
526
+ if selector == "any":
527
+ refs = sorted(branches)
528
+ elif selector == "non-default":
529
+ refs = sorted(r for r in branches if r not in ("refs/heads/main", "refs/heads/master"))
530
+ else:
531
+ refs = [r for r in selector if r in branches]
532
+ for ref in refs:
533
+ found = git(repo, "grep", "-q", "-F", "-e", check["text"], ref, "--", check["path"], env=env)
534
+ if found.returncode == 0:
535
+ return _outcome(True, ref)
536
+ return _outcome(False, f"not on {refs}")
537
+ if kind == "worktree_count":
538
+ count = worktree_count(repo, env)
539
+ ok = count == check["value"] if check["op"] == "eq" else count >= check["value"]
540
+ return _outcome(ok, f"{count} worktrees")
541
+ if kind == "no_new_branches":
542
+ new = sorted(set(local_branches(repo, env)) - set(snapshot["branches"][check["repo"]]))
543
+ return _outcome(not new, f"new {new}" if new else "")
544
+ raise TaskError(f"unhandled kind {kind}")
545
+
546
+
547
+ # --------------------------------------------------------------------------- oracles
548
+
549
+ def check_oracles(tasks):
550
+ """Run every task's good and bad trajectories without a model and compare the grades
551
+ with what the task declares. Returns the problems found; empty means every check
552
+ passed on the good trajectory and failed where a bad trajectory says it must."""
553
+ problems = []
554
+ for task in tasks:
555
+ trajectories = [("good", task["oracle"]["good"], None)]
556
+ trajectories += [(f"bad[{i}]", b["commands"], set(b["fails"])) for i, b in enumerate(task["oracle"]["bad"])]
557
+ for label, commands, fails in trajectories:
558
+ with tempfile.TemporaryDirectory(prefix="paired-oracle-") as tmp:
559
+ try:
560
+ world, gitconfig, snapshot, _ = build_world(task, Path(tmp), canary=False)
561
+ except TaskError as exc:
562
+ problems.append(str(exc))
563
+ break
564
+ if commands:
565
+ done = run_script(commands, world / task["cwd"], clean_env(gitconfig, Path(tmp)))
566
+ if done.returncode != 0:
567
+ problems.append(f"{task['id']} {label}: trajectory failed: {done.stderr.strip()[-300:]}")
568
+ continue
569
+ results = grade(task, world, snapshot, commands)
570
+ for cid, res in results.items():
571
+ if fails is None and res["result"] != "pass":
572
+ problems.append(f"{task['id']} good: {cid} is {res['result']} ({res['detail']})")
573
+ if fails is not None and cid in fails and res["result"] != "fail":
574
+ problems.append(f"{task['id']} {label}: {cid} should fail but is {res['result']}")
575
+ return problems
576
+
577
+
578
+ # --------------------------------------------------------------------------- arms
579
+
580
+ def source_env():
581
+ """Environment for reading the source repository: an inherited GIT_DIR or GIT_INDEX_FILE
582
+ would otherwise point these commands at another repository."""
583
+ return inherited_env()
584
+
585
+
586
+ def resolve_commit(repo, ref):
587
+ out = subprocess.run(["git", "-C", str(repo), "rev-parse", "--verify", "-q", f"{ref}^{{commit}}"],
588
+ env=source_env(), capture_output=True, text=True, timeout=60)
589
+ if out.returncode != 0:
590
+ raise TaskError(f"cannot resolve ref {ref!r}")
591
+ return out.stdout.strip()
592
+
593
+
594
+ def _raise(error):
595
+ raise error
596
+
597
+
598
+ def tree_manifest(root):
599
+ """[kind, relative path, content digest or link target] for every file under root. A directory
600
+ that cannot be read raises instead of being skipped, so a manifest is never silently partial."""
601
+ entries = []
602
+ for dirpath, dirs, files in os.walk(root, onerror=_raise):
603
+ dirs.sort()
604
+ for name in sorted(files):
605
+ path = os.path.join(dirpath, name)
606
+ rel = os.path.relpath(path, root)
607
+ if os.path.islink(path):
608
+ entries.append(["L", rel, os.readlink(path)])
609
+ else:
610
+ entries.append(["x" if os.access(path, os.X_OK) else "f", rel, file_digest(path)])
611
+ for name in dirs: # a directory symlink is listed, never followed
612
+ path = os.path.join(dirpath, name)
613
+ if os.path.islink(path):
614
+ entries.append(["L", os.path.relpath(path, root), os.readlink(path)])
615
+ return entries
616
+
617
+
618
+ def tree_digest(root, manifest=None):
619
+ lines = "".join(f"{kind} {rel} {value}\n" for kind, rel, value in (manifest or tree_manifest(root)))
620
+ return hashlib.sha256(lines.encode()).hexdigest()
621
+
622
+
623
+ def manifest_changes(before, after):
624
+ old = {rel: (kind, value) for kind, rel, value in before}
625
+ new = {rel: (kind, value) for kind, rel, value in after}
626
+ return sorted(set(old) ^ set(new) | {rel for rel in set(old) & set(new) if old[rel] != new[rel]})
627
+
628
+
629
+ def plugin_name_of(root):
630
+ return json.loads((Path(root) / ".claude-plugin" / "plugin.json").read_text(encoding="utf-8"))["name"]
631
+
632
+
633
+ def export_arm(repo, commit, dest):
634
+ """Extract `git archive <commit>` into a staging directory and move it to dest only when
635
+ complete, so an interrupted export is never reused as an arm. Returns the plugin name."""
636
+ staging = dest.parent / f".{dest.name}.partial"
637
+ if staging.exists():
638
+ shutil.rmtree(staging)
639
+ staging.mkdir(parents=True)
640
+ with tempfile.TemporaryFile() as tar_file:
641
+ done = subprocess.run(["git", "-C", str(repo), "archive", "--format=tar", commit], stdout=tar_file,
642
+ stderr=subprocess.PIPE, env=source_env(), timeout=300)
643
+ if done.returncode != 0:
644
+ raise TaskError(f"git archive failed for {commit}: {done.stderr.decode(errors='replace').strip()}")
645
+ tar_file.seek(0)
646
+ with tarfile.open(fileobj=tar_file) as tar:
647
+ try:
648
+ tar.extractall(staging, filter="data")
649
+ except tarfile.FilterError as exc:
650
+ raise TaskError(f"archive member rejected: {exc}") from None
651
+ if not (staging / ".claude-plugin" / "plugin.json").is_file():
652
+ raise TaskError(f"{commit} has no .claude-plugin/plugin.json")
653
+ staging.rename(dest)
654
+ return plugin_name_of(dest)
655
+
656
+
657
+ # --------------------------------------------------------------------------- running one sample
658
+
659
+ def _group_alive(pgid):
660
+ try:
661
+ os.killpg(pgid, 0)
662
+ except ProcessLookupError:
663
+ return False
664
+ except PermissionError:
665
+ return True
666
+ return True
667
+
668
+
669
+ def kill_group(proc):
670
+ """Terminate the run's process group, including background jobs the agent left.
671
+ Returns True only when the group is confirmed gone."""
672
+ for sig in (signal.SIGTERM, signal.SIGKILL):
673
+ proc.poll()
674
+ if not _group_alive(proc.pid):
675
+ break
676
+ try:
677
+ os.killpg(proc.pid, sig)
678
+ except ProcessLookupError:
679
+ break
680
+ except PermissionError:
681
+ return False
682
+ deadline = time.monotonic() + 5
683
+ while time.monotonic() < deadline:
684
+ proc.poll() # reap the leader so a zombie does not keep the group alive
685
+ if not _group_alive(proc.pid):
686
+ break
687
+ time.sleep(0.05)
688
+ try:
689
+ proc.wait(timeout=5)
690
+ except subprocess.TimeoutExpired:
691
+ return False
692
+ return not _group_alive(proc.pid)
693
+
694
+
695
+ def run_agent(cmd, prompt, cwd, env, timeout, stream_path, stderr_path, ctx=None):
696
+ """Run one agent process in its own process group. With a batch context, the spawn happens
697
+ under the context's lock and is refused once a signal is latched, shutdown has begun or the
698
+ guard has stopped the batch; returns None in that case. The latch is set when the runner's
699
+ main thread handles the signal, at most one wait interval after it arrives; a run that passed
700
+ this check before then is killed with the others and records nothing."""
701
+ started = time.monotonic()
702
+ timed_out = False
703
+ with open(stream_path, "wb") as out, open(stderr_path, "wb") as err:
704
+ with ctx["spawn_lock"] if ctx else threading.Lock():
705
+ if ctx and (ctx["latched"] or ctx["interrupted"].is_set() or ctx["stop"].is_set()):
706
+ return None
707
+ proc = subprocess.Popen(cmd, cwd=str(cwd), env=env, stdin=subprocess.PIPE, stdout=out, stderr=err,
708
+ start_new_session=True)
709
+ if ctx:
710
+ ctx["live"].add(proc)
711
+ try:
712
+ proc.communicate(input=prompt.encode("utf-8"), timeout=timeout)
713
+ except subprocess.TimeoutExpired:
714
+ timed_out = True
715
+ finally:
716
+ cleaned = kill_group(proc)
717
+ if ctx:
718
+ ctx["live"].discard(proc)
719
+ return {"exit_code": proc.returncode, "timed_out": timed_out, "cleanup_confirmed": cleaned,
720
+ "seconds": round(time.monotonic() - started, 1)}
721
+
722
+
723
+ def _utilizations(info):
724
+ values = []
725
+ if isinstance(info, dict):
726
+ if isinstance(info.get("utilization"), (int, float)):
727
+ values.append(info["utilization"])
728
+ windows = info.get("unifiedWindows")
729
+ if isinstance(windows, dict):
730
+ for window in windows.values():
731
+ if isinstance(window, dict) and isinstance(window.get("utilization"), (int, float)):
732
+ values.append(window["utilization"])
733
+ return values
734
+
735
+
736
+ def parse_stream(path):
737
+ """Read a stream-json transcript. `raw` keeps every byte as text, so a check can cover what
738
+ the model saw in tool results and hook output as well as what it wrote; `trailing_activity`
739
+ counts assistant and tool events after the last result, which a finished run never has."""
740
+ parsed = {"init": None, "results": [], "hooks": [], "routing_injected": False, "tool_uses": [],
741
+ "texts": [], "invalid_lines": 0, "max_utilization": None, "trailing_activity": 0}
742
+ raw_text = Path(path).read_bytes().decode("utf-8", "replace")
743
+ parsed["raw"] = raw_text
744
+ with open(path, "rb") as fh:
745
+ for raw in fh:
746
+ line = raw.decode("utf-8", "replace").strip()
747
+ if not line:
748
+ continue
749
+ try:
750
+ event = json.loads(line)
751
+ except (json.JSONDecodeError, RecursionError):
752
+ parsed["invalid_lines"] += 1
753
+ continue
754
+ if not isinstance(event, dict):
755
+ parsed["invalid_lines"] += 1
756
+ continue
757
+ kind, sub = event.get("type"), event.get("subtype")
758
+ if kind == "system" and sub == "init" and parsed["init"] is None:
759
+ parsed["init"] = event
760
+ elif kind == "system" and sub == "hook_response":
761
+ parsed["hooks"].append(str(event.get("hook_name")))
762
+ if ROUTING_MARKER in f"{event.get('output', '')}{event.get('stdout', '')}":
763
+ parsed["routing_injected"] = True
764
+ elif kind in ("assistant", "user"):
765
+ parsed["trailing_activity"] += 1
766
+ if kind == "assistant":
767
+ message = event.get("message")
768
+ content = message.get("content") if isinstance(message, dict) else None
769
+ for item in content if isinstance(content, list) else []:
770
+ if not isinstance(item, dict):
771
+ continue
772
+ if item.get("type") == "text" and isinstance(item.get("text"), str):
773
+ parsed["texts"].append(item["text"])
774
+ elif item.get("type") == "tool_use":
775
+ tool_input = item.get("input") if isinstance(item.get("input"), dict) else {}
776
+ parsed["tool_uses"].append({"name": str(item.get("name")), "input": tool_input})
777
+ elif kind == "result":
778
+ parsed["results"].append(event)
779
+ parsed["trailing_activity"] = 0
780
+ elif kind == "rate_limit_event":
781
+ values = _utilizations(event.get("rate_limit_info"))
782
+ if values:
783
+ parsed["max_utilization"] = max(values + [parsed["max_utilization"] or 0])
784
+ return parsed
785
+
786
+
787
+ def _strings(value):
788
+ if isinstance(value, str):
789
+ yield value
790
+ elif isinstance(value, dict):
791
+ for item in value.values():
792
+ yield from _strings(item)
793
+ elif isinstance(value, list):
794
+ for item in value:
795
+ yield from _strings(item)
796
+
797
+
798
+ def _under(path, root):
799
+ return path == root or path.startswith(root.rstrip(os.sep) + os.sep)
800
+
801
+
802
+ def outside_paths(tool_uses, allowed, watched, home):
803
+ """Heuristic: absolute or home-relative paths in tool inputs that fall under a watched
804
+ root (home, this repository, the output root) but outside every allowed root."""
805
+ flagged = []
806
+ for use in tool_uses:
807
+ for text in _strings(use["input"]):
808
+ for token in PATH_TOKEN.findall(text):
809
+ expanded = re.sub(r"^(?:~|\$\{?HOME\}?)", lambda _: home, token)
810
+ if not expanded.startswith("/"):
811
+ continue
812
+ real = os.path.realpath(expanded)
813
+ if any(_under(real, root) for root in allowed):
814
+ continue
815
+ if any(_under(real, root) for root in watched) and real not in flagged:
816
+ flagged.append(real)
817
+ return flagged
818
+
819
+
820
+ def assess(parsed, run, arm, model, arm_dir, plugin_name, arm_dirs, token):
821
+ """Structural isolation evidence for one run. Returns the reasons the sample is invalid.
822
+ A run may hold several results: a plugin Stop hook can send the agent back for more
823
+ turns, which is the treatment's own behavior, so the last result decides how it ended."""
824
+ reasons = []
825
+ results = parsed["results"]
826
+ if run["timed_out"]:
827
+ reasons.append("timeout")
828
+ elif not results:
829
+ reasons.append("no_result")
830
+ elif results[-1].get("subtype") != "success" or results[-1].get("is_error"):
831
+ reasons.append(f"error_result:{results[-1].get('subtype')}")
832
+ elif parsed["trailing_activity"]:
833
+ reasons.append("activity_after_last_result")
834
+ if parsed["invalid_lines"]: # the CLI writes only JSON lines; anything else is a truncated or broken stream
835
+ reasons.append("malformed_stream")
836
+ init = parsed["init"]
837
+ isolation = {"hooks": sorted(set(parsed["hooks"]))}
838
+ if not isinstance(init, dict):
839
+ reasons.append("no_init_event")
840
+ else:
841
+ isolation["model"] = init.get("model")
842
+ if init.get("model") != model:
843
+ reasons.append("model_mismatch")
844
+ plugins, servers = init.get("plugins"), init.get("mcp_servers")
845
+ if not isinstance(plugins, list) or not isinstance(servers, list):
846
+ reasons.append("init_shape_unverifiable")
847
+ entries = [p for p in plugins if isinstance(p, dict)] if isinstance(plugins, list) else []
848
+ isolation["plugins"] = sorted(f"{p.get('name')}@{'builtin' if p.get('path') == 'builtin' else 'dir'}"
849
+ for p in entries)
850
+ own = [p for p in entries if p.get("name") == plugin_name]
851
+ if arm == "off":
852
+ from_arm_dirs = [p for p in entries if p.get("path") != "builtin" and isinstance(p.get("path"), str)
853
+ and any(_under(os.path.realpath(p["path"]), d) for d in arm_dirs)]
854
+ if own or from_arm_dirs:
855
+ reasons.append("plugin_loaded_in_off_arm")
856
+ elif len(own) != 1 or os.path.realpath(str(own[0].get("path"))) != os.path.realpath(str(arm_dir)):
857
+ reasons.append("plugin_identity_mismatch")
858
+ if any(p.get("path") != "builtin" and p.get("name") != plugin_name for p in entries):
859
+ reasons.append("foreign_plugins")
860
+ if isinstance(servers, list) and servers:
861
+ reasons.append("mcp_servers_present")
862
+ if (arm != "off") != parsed["routing_injected"]:
863
+ reasons.append("routing_injection_mismatch")
864
+ if token in parsed["raw"]: # loaded as instructions, read through a tool, or echoed: the run saw it
865
+ reasons.append("instruction_file_canary_seen")
866
+ if not run["cleanup_confirmed"]:
867
+ reasons.append("process_cleanup_unconfirmed")
868
+ return reasons, isolation
869
+
870
+
871
+ # --------------------------------------------------------------------------- statistics and report
872
+
873
+ def fisher_two_sided(a, b, c, d):
874
+ """Two-sided Fisher exact p for the table [[a, b], [c, d]] (pass/fail in two arms)."""
875
+ n1, n2, k = a + b, c + d, a + c
876
+ n = n1 + n2
877
+ if n == 0:
878
+ return 1.0
879
+ denom = math.comb(n, k)
880
+
881
+ def prob(x):
882
+ return math.comb(n1, x) * math.comb(n2, k - x) / denom
883
+ observed = prob(a)
884
+ total = sum(prob(x) for x in range(max(0, k - n2), min(k, n1) + 1) if prob(x) <= observed * (1 + 1e-9))
885
+ return min(1.0, total)
886
+
887
+
888
+ def compare(k1, n1, k2, n2):
889
+ """Pre-registered reading of one pairwise comparison."""
890
+ if n1 < 3 or n2 < 3:
891
+ return "insufficient", None
892
+ p = fisher_two_sided(k1, n1 - k1, k2, n2 - k2)
893
+ if p < 0.05:
894
+ return "separated", p
895
+ if abs(k1 / n1 - k2 / n2) >= 0.4:
896
+ return "direction", p
897
+ return "no observed difference", p
898
+
899
+
900
+ def tally(records, task_id, arm, check_id):
901
+ rows = [r for r in records if r["task"] == task_id and r["arm"] == arm and r["valid"]]
902
+ passed = sum(1 for r in rows if r["checks"][check_id]["result"] == "pass")
903
+ failed = sum(1 for r in rows if r["checks"][check_id]["result"] == "fail")
904
+ return passed, passed + failed, len(rows) - passed - failed
905
+
906
+
907
+ def _median(values):
908
+ values = [v for v in values if isinstance(v, (int, float))]
909
+ return round(statistics.median(values), 2) if values else None
910
+
911
+
912
+ def build_report(plan, records, calibration, integrity):
913
+ arms = [a for a in ARMS if a in plan["arms"]]
914
+ lines = ["# Paired behavior eval report", "",
915
+ "Advisory evidence for the tasks and conditions below only; the tool header lists what these "
916
+ "numbers are not. Counts exclude invalid samples; `Ns` marks valid samples with out-of-world "
917
+ "path suspects.", "", "## Batch", "",
918
+ f"- plan `{plan['plan_hash'][:16]}`, tool version {plan['tool_version']}",
919
+ f"- model `{plan['model']}`, effort `{plan['effort']}`, claude `{plan['claude_version']}`",
920
+ f"- per-run cap ${plan['max_budget_usd']}, denied tools `{plan['disallowed_tools']}`"]
921
+ for arm in arms:
922
+ info = plan["arms"][arm]
923
+ lines.append(f"- `{arm}` ({TREATMENT[arm]}): " + ("no plugin" if arm == "off" else
924
+ f"`{info['ref']}` = {info['commit'][:12]}, export digest {info['export_digest'][:12]}"))
925
+ graders = sorted({r.get("graded_by") or plan["tool_sha256"] for r in records} - {plan["tool_sha256"]})
926
+ if graders:
927
+ lines.append(f"- records regraded by tool `{', '.join(g[:12] for g in graders)}`; the batch ran with "
928
+ f"`{plan['tool_sha256'][:12]}`")
929
+ cal = "not run" if calibration is None else ("fired" if calibration.get("fired") else "DID NOT FIRE")
930
+ lines.append(f"- instruction-file canary calibration: {cal}")
931
+ if integrity is not None:
932
+ detail = "; ".join(f"{arm}: {', '.join(paths[:5])}" for arm, paths in (integrity.get("changed") or {}).items())
933
+ state = {True: "yes", False: "NO"}.get(integrity.get("unchanged"), "UNKNOWN")
934
+ detail = detail or integrity.get("error", "")
935
+ lines.append(f"- plugin exports unchanged after the batch: {state}" + (f" ({detail})" if detail else ""))
936
+ costs = [r["cost_usd"] for r in records if isinstance(r.get("cost_usd"), (int, float))]
937
+ lines += [f"- samples recorded {len(records)}, valid {sum(r['valid'] for r in records)}, "
938
+ f"with suspects {sum(bool(r['suspect_paths']) for r in records)}, spend ${round(sum(costs), 2)}", "",
939
+ "## Isolation", "", "| arm | recorded | valid | invalid reasons | suspects | continued after a Stop hook |",
940
+ "| --- | --- | --- | --- | --- | --- |"]
941
+ for arm in arms:
942
+ rows = [r for r in records if r["arm"] == arm]
943
+ reasons = {}
944
+ for r in rows:
945
+ for reason in r["invalid_reasons"]:
946
+ reasons[reason] = reasons.get(reason, 0) + 1
947
+ text = ", ".join(f"{k} {v}" for k, v in sorted(reasons.items())) or "none"
948
+ lines.append(f"| {arm} | {len(rows)} | {sum(r['valid'] for r in rows)} | {text} | "
949
+ f"{sum(bool(r['suspect_paths']) for r in rows)} | {sum(bool(r.get('continuations')) for r in rows)} |")
950
+ lines.append("")
951
+ pairs = [(a, b) for a, b in (("candidate", "base"), ("base", "off"), ("candidate", "off"))
952
+ if a in arms and b in arms]
953
+ comparisons = 0
954
+ for task in plan["tasks"]:
955
+ tid = task["id"]
956
+ lines += [f"## {tid}", "", task["measures"], "",
957
+ "| check | role | " + " | ".join(arms) + " |", "| --- | --- | " + " | ".join("---" for _ in arms) + " |"]
958
+ for check in task["checks"]:
959
+ cells = []
960
+ for arm in arms:
961
+ k, n, na = tally(records, tid, arm, check["id"])
962
+ suspects = sum(1 for r in records if r["task"] == tid and r["arm"] == arm and r["valid"]
963
+ and r["suspect_paths"])
964
+ cell = f"{k}/{n}" + (f" (+{na} n/a)" if na else "") + (f" {suspects}s" if suspects else "")
965
+ cells.append(cell if n or na else "—")
966
+ lines.append(f"| {check['id']} | {check['role']} | " + " | ".join(cells) + " |")
967
+ if pairs:
968
+ lines += ["", "| check | " + " | ".join(f"{a} vs {b}" for a, b in pairs) + " |",
969
+ "| --- | " + " | ".join("---" for _ in pairs) + " |"]
970
+ for check in task["checks"]:
971
+ cells = []
972
+ for a, b in pairs:
973
+ k1, n1, _ = tally(records, tid, a, check["id"])
974
+ k2, n2, _ = tally(records, tid, b, check["id"])
975
+ verdict, p = compare(k1, n1, k2, n2)
976
+ comparisons += p is not None
977
+ cells.append(f"{k1}/{n1} vs {k2}/{n2}: {verdict}" + (f" (p={p:.3f})" if p is not None else ""))
978
+ lines.append(f"| {check['id']} | " + " | ".join(cells) + " |")
979
+ lines += ["", "| arm | median cost $ | median turns | median seconds |", "| --- | --- | --- | --- |"]
980
+ for arm in arms:
981
+ rows = [r for r in records if r["task"] == tid and r["arm"] == arm and r["valid"]]
982
+ lines.append(f"| {arm} | {_median([r.get('cost_usd') for r in rows])} | "
983
+ f"{_median([r.get('turns') for r in rows])} | {_median([r.get('seconds') for r in rows])} |")
984
+ lines.append("")
985
+ lines += ["## Reading", "",
986
+ f"{comparisons} pairwise comparisons, each with an uncorrected two-sided Fisher exact p, so a few "
987
+ "`separated` labels can be chance. `separated` = p < 0.05; `direction` = pass rates differ by 0.4 "
988
+ "or more without p < 0.05; `insufficient` = fewer than 3 valid samples in an arm.", ""]
989
+ return "\n".join(lines)
990
+
991
+
992
+ # --------------------------------------------------------------------------- batch
993
+
994
+ def checkout_roots(repo):
995
+ """Every checkout of the repository: the given one, the main one and each registered worktree."""
996
+ out = subprocess.run(["git", "-C", str(repo), "worktree", "list", "--porcelain", "-z"], env=source_env(),
997
+ capture_output=True, text=True, timeout=60)
998
+ if out.returncode != 0:
999
+ raise TaskError(f"{repo} is not a git repository")
1000
+ roots = {os.path.realpath(repo)}
1001
+ roots.update(os.path.realpath(field[len("worktree "):]) for field in out.stdout.split("\0")
1002
+ if field.startswith("worktree ")) # -z keeps paths unquoted, newlines included
1003
+ return sorted(roots)
1004
+
1005
+
1006
+ def claude_version(claude):
1007
+ try:
1008
+ out = subprocess.run([claude, "--version"], env=inherited_env(), capture_output=True, text=True, timeout=60)
1009
+ except (OSError, subprocess.TimeoutExpired) as exc:
1010
+ raise TaskError(f"cannot run {claude} --version: {exc}") from None
1011
+ if out.returncode != 0 or not out.stdout.strip():
1012
+ raise TaskError(f"{claude} --version failed")
1013
+ return out.stdout.strip().splitlines()[0]
1014
+
1015
+
1016
+ def canonical(obj):
1017
+ return json.dumps(obj, sort_keys=True, ensure_ascii=False, separators=(",", ":"))
1018
+
1019
+
1020
+ def write_atomic(path, text):
1021
+ """Write beside the destination, then rename over it, so a reader never sees a torn file."""
1022
+ tmp = path.with_name(f".{path.name}.tmp")
1023
+ tmp.write_text(text, encoding="utf-8")
1024
+ os.replace(tmp, path)
1025
+
1026
+
1027
+ def schedule(tasks, arms, samples_override):
1028
+ """Interleave arms within each sample index so drift over the batch hits every arm alike."""
1029
+ counts = {t["id"]: samples_override or t["samples"] for t in tasks}
1030
+ order = []
1031
+ for i in range(1, max(counts.values()) + 1):
1032
+ for t_index, task in enumerate(tasks):
1033
+ if i > counts[task["id"]]:
1034
+ continue
1035
+ shift = (i + t_index) % len(arms)
1036
+ for arm in arms[shift:] + arms[:shift]:
1037
+ order.append((task, arm, i))
1038
+ return order
1039
+
1040
+
1041
+ def agent_command(claude, model, effort, budget, plugin_dir, setting_sources=""):
1042
+ cmd = [claude, "-p", "--model", model]
1043
+ if effort:
1044
+ cmd += ["--effort", effort]
1045
+ if plugin_dir:
1046
+ cmd += ["--plugin-dir", str(plugin_dir)]
1047
+ # Session persistence stays on: a plugin's hooks may read the session transcript, and without
1048
+ # it they run degraded. collect_transcripts moves the run's transcript out of the home directory.
1049
+ cmd += ["--setting-sources", setting_sources, "--strict-mcp-config", "--permission-mode", "bypassPermissions",
1050
+ "--output-format", "stream-json", "--verbose",
1051
+ "--max-budget-usd", str(budget), "--disallowedTools", DISALLOWED_TOOLS]
1052
+ return cmd
1053
+
1054
+
1055
+ def collect_transcripts(stream_path, dest):
1056
+ """Move the session transcripts Claude Code wrote for this run out of ~/.claude/projects into
1057
+ dest. Only entries named by a session id from this run's own init events are moved, and a
1058
+ project directory is removed only when that leaves it empty. Returns the moved names in two
1059
+ lists: transcripts, which were non-empty regular files when found, and everything else, such
1060
+ as session directories or links, which never count as a persisted transcript."""
1061
+ sessions = set()
1062
+ for line in Path(stream_path).read_text(encoding="utf-8", errors="replace").splitlines():
1063
+ try:
1064
+ event = json.loads(line)
1065
+ except ValueError:
1066
+ continue
1067
+ if isinstance(event, dict) and event.get("type") == "system" and event.get("subtype") == "init":
1068
+ session = event.get("session_id")
1069
+ if isinstance(session, str) and SESSION_ID.fullmatch(session):
1070
+ sessions.add(session)
1071
+ projects = Path.home() / ".claude" / "projects"
1072
+ transcripts, others = [], []
1073
+ for session in sorted(sessions):
1074
+ for source in sorted(projects.glob(f"*/{session}.jsonl")) + sorted(projects.glob(f"*/{session}")):
1075
+ real = (source.name == f"{session}.jsonl" and source.is_file() and not source.is_symlink()
1076
+ and source.stat().st_size > 0)
1077
+ target = dest / f"transcript-{source.name}"
1078
+ shutil.move(str(source), str(target))
1079
+ (transcripts if real else others).append(target.name)
1080
+ try:
1081
+ source.parent.rmdir()
1082
+ except OSError:
1083
+ pass # the project directory still holds other sessions
1084
+ return transcripts, others
1085
+
1086
+
1087
+ def run_sample(ctx, task, arm, index):
1088
+ """One run in a fresh world, with a private copy of the arm's frozen export, so a run that
1089
+ edits the plugin changes only its own copy and is recorded as invalid."""
1090
+ sample_dir = ctx["out"] / "runs" / task["id"] / arm / str(index)
1091
+ world, gitconfig, snapshot, token = build_world(task, sample_dir)
1092
+ write_atomic(sample_dir / "snapshot.json", json.dumps(snapshot, indent=1))
1093
+ plugin_dir = None
1094
+ if arm in ctx["arm_dirs"]:
1095
+ plugin_dir = sample_dir / "plugin"
1096
+ if plugin_dir.exists():
1097
+ shutil.rmtree(plugin_dir)
1098
+ shutil.copytree(ctx["arm_dirs"][arm], plugin_dir, symlinks=True)
1099
+ plan = ctx["plan"]
1100
+ cmd = agent_command(ctx["claude"], plan["model"], plan["effort_flag"], plan["max_budget_usd"], plugin_dir)
1101
+ run = run_agent(cmd, task["prompt"], world / task["cwd"], clean_env(gitconfig, sample_dir, agent=True),
1102
+ task["timeout_seconds"], sample_dir / "stream.jsonl", sample_dir / "stderr.txt", ctx)
1103
+ if run is None:
1104
+ return None # never started: resume runs it
1105
+ run["transcripts"], run["session_files"] = collect_transcripts(sample_dir / "stream.jsonl", sample_dir)
1106
+ if ctx["interrupted"].is_set():
1107
+ return None # stopped by an interrupt: not an outcome; resume reruns it
1108
+ run["export_changes"] = []
1109
+ if plugin_dir is not None:
1110
+ run["export_changes"] = manifest_changes(ctx["manifests"][arm], tree_manifest(plugin_dir))
1111
+ if not run["export_changes"]:
1112
+ shutil.rmtree(plugin_dir) # identical to the frozen export; kept only when the run changed it
1113
+ return make_record(ctx, task, arm, index, sample_dir, snapshot, token, run, plugin_dir)
1114
+
1115
+
1116
+ def make_record(ctx, task, arm, index, sample_dir, snapshot, token, run, plugin_dir):
1117
+ """Assess and grade one finished run from its saved stream and world, and write record.json.
1118
+ The record keeps the canary token and the snapshot so later regrading never reads files the
1119
+ tested agent could have rewritten."""
1120
+ world = sample_dir / "world"
1121
+ plan = ctx["plan"]
1122
+ parsed = parse_stream(sample_dir / "stream.jsonl")
1123
+ reasons, isolation = assess(parsed, run, arm, plan["model"], plugin_dir, ctx["plugin_name"],
1124
+ ctx["arm_dirs_real"], token)
1125
+ if run.get("export_changes"):
1126
+ reasons.append("plugin_export_changed")
1127
+ if not run.get("transcripts"):
1128
+ reasons.append("transcript_missing") # hooks that read the session transcript did not run as in normal use
1129
+ allowed = [os.path.realpath(world)] + ([os.path.realpath(plugin_dir)] if plugin_dir else [])
1130
+ suspects = outside_paths(parsed["tool_uses"], allowed, ctx["watched"], ctx["home"])
1131
+ commands = [str(u["input"].get("command", "")) for u in parsed["tool_uses"] if u["name"] == "Bash"]
1132
+ results = parsed["results"]
1133
+ last = results[-1] if results else {}
1134
+ record = {
1135
+ "task": task["id"], "arm": arm, "treatment": TREATMENT[arm], "sample": index,
1136
+ "plan_hash": plan["plan_hash"], "graded_by": ctx["tool_sha256"], "valid": not reasons,
1137
+ "invalid_reasons": reasons, "suspect_paths": suspects[:10], "isolation": isolation,
1138
+ "checks": grade(task, world, snapshot, commands),
1139
+ "cost_usd": last.get("total_cost_usd"), # cumulative across continuations
1140
+ "turns": sum(r["num_turns"] for r in results if isinstance(r.get("num_turns"), int)) or None,
1141
+ "continuations": max(len(results) - 1, 0),
1142
+ "seconds": run["seconds"], "exit_code": run["exit_code"],
1143
+ "run": {"timed_out": run["timed_out"], "cleanup_confirmed": run["cleanup_confirmed"],
1144
+ "export_changes": run.get("export_changes") or [], "transcripts": run.get("transcripts") or [],
1145
+ "session_files": run.get("session_files") or [], "evidence_version": RUN_EVIDENCE_VERSION},
1146
+ "canary": token, "snapshot": snapshot, "plugin_dir": str(plugin_dir) if plugin_dir else None,
1147
+ "max_utilization": parsed["max_utilization"],
1148
+ "skills_invoked": [str(u["input"].get("skill")) for u in parsed["tool_uses"] if u["name"] == "Skill"],
1149
+ "bash_commands": len(commands), "result_excerpt": str(last.get("result", ""))[:300],
1150
+ }
1151
+ write_atomic(sample_dir / "record.json", json.dumps(record, indent=1, ensure_ascii=False))
1152
+ return record
1153
+
1154
+
1155
+ def regrade(ctx, out, tasks):
1156
+ """Rebuild every record of the plan from its saved stream and world with the current
1157
+ assessment and graders; no model runs. The batch's tasks must be byte-identical, and the
1158
+ canary token, snapshot and plugin path come from the record, not from the world."""
1159
+ by_id = {t["id"]: t for t in tasks}
1160
+ for entry in ctx["plan"]["tasks"]:
1161
+ if entry["id"] not in by_id or by_id[entry["id"]]["_sha256"] != entry["sha256"]:
1162
+ raise TaskError(f"task {entry['id']} differs from the one the batch ran; regrading would apply other checks")
1163
+ records = load_records(out, ctx["plan"])
1164
+ for old in records:
1165
+ sample_dir = out / "runs" / old["task"] / old["arm"] / str(old["sample"])
1166
+ if (not (isinstance(old.get("canary"), str) and isinstance(old.get("snapshot"), dict) and "plugin_dir" in old)
1167
+ or old.get("legacy_inputs")):
1168
+ raise TaskError(f"{sample_dir.relative_to(out)}: the record holds no runner-recorded canary token, snapshot "
1169
+ "and plugin path, and the world's copies could have been rewritten by the run")
1170
+ if (old.get("run") or {}).get("evidence_version") != RUN_EVIDENCE_VERSION:
1171
+ raise TaskError(f"{sample_dir.relative_to(out)}: the record's run evidence was written under other rules "
1172
+ "(such as an earlier transcript check), so regrading would reuse evidence it cannot trust")
1173
+ token, snapshot = old["canary"], old["snapshot"]
1174
+ plugin_dir = Path(old["plugin_dir"]) if old["plugin_dir"] else None
1175
+ run = dict(old.get("run") or {"timed_out": "timeout" in old["invalid_reasons"],
1176
+ "cleanup_confirmed": "process_cleanup_unconfirmed" not in old["invalid_reasons"]},
1177
+ seconds=old["seconds"], exit_code=old["exit_code"])
1178
+ make_record(ctx, by_id[old["task"]], old["arm"], old["sample"], sample_dir, snapshot, token, run, plugin_dir)
1179
+ return len(records)
1180
+
1181
+
1182
+ def calibrate_canary(ctx):
1183
+ """Prove the canary can fire: allow project settings and instruction files once, on a tiny
1184
+ prompt, and expect the token in a finished, cleaned-up reply."""
1185
+ probe = {"id": "canary-calibration", "setup": ["git init -q -b main app"], "cwd": "app", "checks": []}
1186
+ sample_dir = ctx["out"] / "calibration"
1187
+ world, gitconfig, _, token = build_world(probe, sample_dir)
1188
+ env = clean_env(gitconfig, sample_dir, agent=True)
1189
+ env.pop("CLAUDE_CODE_DISABLE_CLAUDE_MDS")
1190
+ cmd = agent_command(ctx["claude"], ctx["plan"]["model"], None, 1, None, setting_sources="project")
1191
+ run = run_agent(cmd, "Reply with the single word OK.", world / "app", env, 300,
1192
+ sample_dir / "stream.jsonl", sample_dir / "stderr.txt", ctx)
1193
+ if run is None:
1194
+ return None
1195
+ collect_transcripts(sample_dir / "stream.jsonl", sample_dir)
1196
+ if ctx["interrupted"].is_set():
1197
+ return None
1198
+ parsed = parse_stream(sample_dir / "stream.jsonl")
1199
+ last = parsed["results"][-1] if parsed["results"] else {}
1200
+ finished = (not run["timed_out"] and run["cleanup_confirmed"] and last.get("subtype") == "success"
1201
+ and not last.get("is_error") and not parsed["trailing_activity"])
1202
+ texts = parsed["texts"] + [str(r.get("result", "")) for r in parsed["results"]]
1203
+ result = {"fired": finished and any(token in t for t in texts), "finished": finished,
1204
+ "timed_out": run["timed_out"], "max_utilization": parsed["max_utilization"]}
1205
+ write_atomic(ctx["out"] / "canary-calibration.json", json.dumps(result, indent=1))
1206
+ return result
1207
+
1208
+
1209
+ def read_json(path):
1210
+ return json.loads(path.read_text(encoding="utf-8")) if path.is_file() else None
1211
+
1212
+
1213
+ def load_records(out, plan):
1214
+ """The plan's records only: each must sit at the path its task, arm and sample name, inside
1215
+ the frozen inventory and under the frozen plan hash, so a stray or copied file cannot count."""
1216
+ inventory = {(t["id"], arm, i) for t in plan["tasks"] for arm in plan["arms"] for i in range(1, t["samples"] + 1)}
1217
+ records = []
1218
+ for path in sorted((out / "runs").glob("*/*/*/record.json")):
1219
+ rel = path.relative_to(out)
1220
+ try:
1221
+ record = json.loads(path.read_text(encoding="utf-8"))
1222
+ except json.JSONDecodeError:
1223
+ raise TaskError(f"{rel}: not valid JSON") from None
1224
+ key = (record.get("task"), record.get("arm"), record.get("sample"))
1225
+ if (record.get("plan_hash") != plan["plan_hash"] or key not in inventory
1226
+ or rel.parts[1:4] != tuple(str(part) for part in key)):
1227
+ raise TaskError(f"{rel}: record does not belong to this plan at this path")
1228
+ records.append(record)
1229
+ return records
1230
+
1231
+
1232
+ def write_report(out, plan, records):
1233
+ calibration = read_json(out / "canary-calibration.json")
1234
+ integrity = read_json(out / "integrity.json")
1235
+ write_atomic(out / "results.json", json.dumps(records, indent=1, ensure_ascii=False))
1236
+ write_atomic(out / "report.md", build_report(plan, records, calibration, integrity))
1237
+
1238
+
1239
+ def lock_output(out):
1240
+ """Hold an exclusive lock on the output root for this process's lifetime, so two invocations
1241
+ never build worlds over each other."""
1242
+ handle = open(out / ".lock", "a+")
1243
+ try:
1244
+ fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
1245
+ except BlockingIOError:
1246
+ handle.close()
1247
+ raise TaskError("another invocation is using --out; wait for it or choose another --out") from None
1248
+ return handle
1249
+
1250
+
1251
+ def record_integrity(out, arm_dirs, manifests):
1252
+ """Name every path that differs from each frozen export. It runs whenever a report is written,
1253
+ so no report carries an older integrity record. An export or frozen manifest that cannot be
1254
+ read is recorded as unknown with the reason, so the batch's own outcome or failure is never
1255
+ replaced by this check."""
1256
+ if not arm_dirs:
1257
+ return
1258
+ try:
1259
+ changed = {}
1260
+ for arm, root in arm_dirs.items():
1261
+ if manifests.get(arm) is None:
1262
+ raise FileNotFoundError(f"no manifest matching the plan for the {arm} export")
1263
+ if paths := manifest_changes(manifests[arm], tree_manifest(root)):
1264
+ changed[arm] = paths
1265
+ payload = {"unchanged": not changed, "changed": changed}
1266
+ except OSError as exc:
1267
+ payload = {"unchanged": None, "changed": {}, "error": f"{type(exc).__name__}: {exc}"[:300]}
1268
+ write_atomic(out / "integrity.json", json.dumps(payload, indent=1))
1269
+
1270
+
1271
+ def frozen_manifest(out, plan, arm):
1272
+ """The manifest saved when the arm's export was built, or None unless it reads back and hashes
1273
+ to the export digest the plan froze."""
1274
+ try:
1275
+ manifest = read_json(out / "arms" / f"{arm}.manifest.json")
1276
+ if manifest and tree_digest(None, manifest) == plan["arms"][arm]["export_digest"]:
1277
+ return manifest
1278
+ except (OSError, ValueError, TypeError): # unreadable, not JSON, or not a manifest's shape
1279
+ pass
1280
+ return None
1281
+
1282
+
1283
+ def wait_latched(futures, latched, poll=0.2):
1284
+ """Wait for every future and return their results. Signal handlers only latch during a batch,
1285
+ so this is where an interrupt takes effect: between waits, never inside other code."""
1286
+ pending = set(futures)
1287
+ while pending:
1288
+ done, pending = cf.wait(pending, timeout=poll, return_when=cf.FIRST_COMPLETED)
1289
+ if latched: # checked before any result, so a failure the signal caused reads as the interrupt
1290
+ raise KeyboardInterrupt
1291
+ for future in done:
1292
+ future.result()
1293
+ return [future.result() for future in futures]
1294
+
1295
+
1296
+ def tool_sha256():
1297
+ return hashlib.sha256(Path(__file__).read_bytes()).hexdigest()
1298
+
1299
+
1300
+ def make_context(out, plan, arm_dirs, roots, claude=None, manifests=None):
1301
+ home = os.path.realpath(os.path.expanduser("~"))
1302
+ return {"out": out, "claude": claude, "plan": plan, "arm_dirs": arm_dirs, "plugin_name": plan["plugin_name"],
1303
+ "arm_dirs_real": [os.path.realpath(d) for d in arm_dirs.values()], "home": home,
1304
+ "watched": [home, *roots, os.path.realpath(out)], "manifests": manifests or {},
1305
+ "tool_sha256": tool_sha256(), "live": set(), "interrupted": threading.Event(),
1306
+ "stop": threading.Event(), "spawn_lock": threading.Lock(), "latched": []}
1307
+
1308
+
1309
+ def run_batch(args, out, arms, only):
1310
+ repo = args.repo.resolve()
1311
+ roots = checkout_roots(repo)
1312
+ if any(_under(str(out), root) for root in roots):
1313
+ raise TaskError("--out must be outside every checkout of this repository")
1314
+ tasks = load_tasks(args.tasks_dir, only)
1315
+ commits = {arm: resolve_commit(repo, ref) for arm, ref in (("base", args.base), ("candidate", args.candidate))
1316
+ if arm in arms}
1317
+ planned = schedule(tasks, arms, args.samples)
1318
+ if args.dry_run:
1319
+ for arm in arms:
1320
+ print(f"arm {arm}: " + (commits[arm][:12] if arm in commits else "no plugin"))
1321
+ for task in tasks:
1322
+ print(f"task {task['id']}: {args.samples or task['samples']} samples per arm")
1323
+ print(f"runs {len(planned)} (+1 canary calibration), max-runs {args.max_runs}")
1324
+ return 0
1325
+ version = claude_version(args.claude)
1326
+ if out.exists() and not out.is_dir():
1327
+ raise TaskError("--out exists and is not a directory")
1328
+ contents = [entry for entry in out.iterdir() if entry.name != ".lock"] if out.is_dir() else []
1329
+ if contents and not (out / "plan.json").is_file():
1330
+ # The runner never deletes what it did not create in this batch: a root holding anything
1331
+ # but a frozen plan, including exports an interrupted run left before freezing it, is refused.
1332
+ raise TaskError("--out is not empty and holds no plan from this tool; use a new or empty directory")
1333
+ out.mkdir(mode=0o700, parents=True, exist_ok=True)
1334
+ lock = lock_output(out)
1335
+ try:
1336
+ return _locked_batch(args, out, arms, tasks, commits, planned, repo, roots, version)
1337
+ finally:
1338
+ lock.close()
1339
+
1340
+
1341
+ def _locked_batch(args, out, arms, tasks, commits, planned, repo, roots, version):
1342
+ try:
1343
+ existing = read_json(out / "plan.json")
1344
+ except json.JSONDecodeError:
1345
+ raise TaskError("plan.json under --out is not valid JSON; use a new --out") from None
1346
+ arm_dirs, arm_info, names, manifests = {}, {}, set(), {}
1347
+ for arm in arms:
1348
+ if arm == "off":
1349
+ arm_info[arm] = {"ref": None, "commit": None, "export_digest": None}
1350
+ continue
1351
+ dest = out / "arms" / arm
1352
+ names.add(plugin_name_of(dest) if dest.exists() else export_arm(repo, commits[arm], dest))
1353
+ arm_dirs[arm] = dest
1354
+ manifests[arm] = tree_manifest(dest)
1355
+ manifest_path = out / "arms" / f"{arm}.manifest.json"
1356
+ if not manifest_path.exists():
1357
+ write_atomic(manifest_path, json.dumps(manifests[arm]))
1358
+ arm_info[arm] = {"ref": args.base if arm == "base" else args.candidate, "commit": commits[arm],
1359
+ "export_digest": tree_digest(dest, manifests[arm])}
1360
+ if len(names) > 1:
1361
+ raise TaskError("base and candidate exports name different plugins")
1362
+ plugin_name = names.pop() if names else plugin_name_of(repo)
1363
+ plan = {
1364
+ "tool_version": TOOL_VERSION, "tool_sha256": tool_sha256(),
1365
+ "model": args.model, "effort": args.effort or "cli-default", "effort_flag": args.effort,
1366
+ "max_budget_usd": args.max_budget_usd, "disallowed_tools": DISALLOWED_TOOLS, "claude_version": version,
1367
+ "arms": arm_info, "plugin_name": plugin_name, "samples_override": args.samples,
1368
+ "tasks": [{"id": t["id"], "sha256": t["_sha256"], "samples": args.samples or t["samples"],
1369
+ "measures": t["measures"], "checks": [{"id": c["id"], "role": c["role"]} for c in t["checks"]]}
1370
+ for t in tasks],
1371
+ }
1372
+ plan["plan_hash"] = hashlib.sha256(canonical(plan).encode()).hexdigest()
1373
+ if existing is not None and existing.get("plan_hash") != plan["plan_hash"]:
1374
+ raise TaskError("plan differs from the one frozen under --out (tasks, refs, flags, tool or claude version "
1375
+ "changed); use a new --out")
1376
+ if existing is None:
1377
+ write_atomic(out / "plan.json", json.dumps(plan, indent=1, ensure_ascii=False))
1378
+ done = {(r["task"], r["arm"], r["sample"]) for r in load_records(out, plan)}
1379
+ pending = [(t, a, i) for t, a, i in planned if (t["id"], a, i) not in done]
1380
+ need_calibration = bool(pending) and not (read_json(out / "canary-calibration.json") or {}).get("fired")
1381
+ if len(pending) + need_calibration > args.max_runs:
1382
+ raise TaskError(f"{len(pending) + need_calibration} runs exceed --max-runs {args.max_runs}")
1383
+ ctx = make_context(out, plan, arm_dirs, roots, args.claude, manifests)
1384
+ stop = ctx["stop"]
1385
+
1386
+ def trip_stop():
1387
+ with ctx["spawn_lock"]: # a worker past its own check cannot spawn after this
1388
+ stop.set()
1389
+
1390
+ def worker(item):
1391
+ task, arm, index = item
1392
+ if stop.is_set():
1393
+ return None
1394
+ record = run_sample(ctx, task, arm, index)
1395
+ if record is None:
1396
+ return None
1397
+ if (record.get("max_utilization") or 0) >= args.stop_util:
1398
+ trip_stop()
1399
+ print(f"{task['id']} {arm} #{index}: valid={record['valid']} "
1400
+ + " ".join(f"{k}={v['result']}" for k, v in record["checks"].items()), file=sys.stderr, flush=True)
1401
+ return record
1402
+
1403
+ # Before the first launch, SIGINT, SIGTERM and SIGHUP stop raising and only latch; a signal the
1404
+ # caller ignores stays ignored. Nothing is then raised in this thread asynchronously: a latched
1405
+ # signal takes effect in wait_latched, and the evidence below is written in full before exit.
1406
+ # The spawn check reads the same latch, so no run starts once a signal is latched.
1407
+ latched = ctx["latched"]
1408
+ handlers = {sig: signal.signal(sig, lambda signum, frame: latched.append(signum))
1409
+ for sig in (signal.SIGINT, signal.SIGTERM, signal.SIGHUP) if signal.getsignal(sig) != signal.SIG_IGN}
1410
+ pool = cf.ThreadPoolExecutor(max_workers=args.jobs)
1411
+ failure = None
1412
+ try:
1413
+ if need_calibration:
1414
+ calibration = wait_latched([pool.submit(calibrate_canary, ctx)], latched)[0]
1415
+ if calibration is None:
1416
+ raise KeyboardInterrupt
1417
+ if not calibration["fired"]:
1418
+ raise TaskError("the canary calibration did not fire, so a silent canary would prove nothing; "
1419
+ "no samples were run (see calibration/stream.jsonl under --out)")
1420
+ if (calibration.get("max_utilization") or 0) >= args.stop_util:
1421
+ trip_stop()
1422
+ wait_latched([pool.submit(worker, item) for item in pending], latched)
1423
+ except BaseException as exc:
1424
+ failure = exc
1425
+ try:
1426
+ records = _finish_batch(ctx, out, plan, pool, arm_dirs, manifests, failure)
1427
+ finally:
1428
+ for sig, handler in handlers.items():
1429
+ signal.signal(sig, handler)
1430
+ if failure is not None:
1431
+ if isinstance(failure, KeyboardInterrupt):
1432
+ print("interrupted: rerun the same command to resume", file=sys.stderr)
1433
+ return 130
1434
+ raise failure
1435
+ if latched:
1436
+ print("interrupted after the last run finished; the evidence was written in full; rerun the same command "
1437
+ "to resume", file=sys.stderr)
1438
+ return 130
1439
+ print(out / "report.md")
1440
+ recorded = {(r["task"], r["arm"], r["sample"]) for r in records}
1441
+ if any((t["id"], a, i) not in recorded for t, a, i in planned):
1442
+ print("stopped early: rate-limit utilization reached --stop-util; rerun the same command to resume",
1443
+ file=sys.stderr)
1444
+ return 3
1445
+ return 0
1446
+
1447
+
1448
+ def _finish_batch(ctx, out, plan, pool, arm_dirs, manifests, failure):
1449
+ """Stop what is still running after a failure, then write the batch's integrity record and
1450
+ report. After a failure, evidence is best effort and never replaces that failure."""
1451
+ if failure is not None:
1452
+ with ctx["spawn_lock"]: # nothing starts after this point
1453
+ ctx["stop"].set()
1454
+ ctx["interrupted"].set()
1455
+ live = list(ctx["live"])
1456
+ for proc in live:
1457
+ kill_group(proc)
1458
+ pool.shutdown(wait=True, cancel_futures=failure is not None)
1459
+ try:
1460
+ record_integrity(out, arm_dirs, manifests)
1461
+ records = load_records(out, plan)
1462
+ write_report(out, plan, records)
1463
+ return records
1464
+ except Exception as report_error:
1465
+ if failure is None:
1466
+ raise
1467
+ print(f"report not written after the failure: {type(report_error).__name__}: {report_error}", file=sys.stderr)
1468
+ return None
1469
+
1470
+
1471
+ def main(argv=None):
1472
+ parser = argparse.ArgumentParser(description="Paired behavior eval (harness-patterns-and-eval.md §3.1).")
1473
+ parser.add_argument("--out", type=Path, help="private output root, outside every checkout of this repository")
1474
+ parser.add_argument("--base", help="git ref exported as the base arm")
1475
+ parser.add_argument("--candidate", help="git ref exported as the candidate arm")
1476
+ parser.add_argument("--tasks-dir", type=Path, default=DEFAULT_TASKS)
1477
+ parser.add_argument("--tasks", help="comma-separated task ids (default: all)")
1478
+ parser.add_argument("--arms", default=",".join(ARMS))
1479
+ parser.add_argument("--samples", type=int, help="override every task's sample count")
1480
+ parser.add_argument("--model", default="claude-opus-5-5")
1481
+ parser.add_argument("--effort", choices=("low", "medium", "high", "xhigh", "max"),
1482
+ help="pin the tested agent's effort (default: the CLI default)")
1483
+ parser.add_argument("--max-budget-usd", type=float, default=3.0, help="spend cap per run")
1484
+ parser.add_argument("--max-runs", type=int, default=60, help="refuse a batch that would start more runs")
1485
+ parser.add_argument("--jobs", type=int, default=3)
1486
+ parser.add_argument("--stop-util", type=float, default=0.9,
1487
+ help="stop starting runs once any rate-limit window reaches this utilization")
1488
+ parser.add_argument("--claude", default="claude", help="claude executable")
1489
+ parser.add_argument("--repo", type=Path, default=REPO_ROOT, help="repository the refs are exported from")
1490
+ parser.add_argument("--dry-run", action="store_true", help="print the plan; no runs, nothing written")
1491
+ parser.add_argument("--report-only", action="store_true", help="rebuild report.md from saved records")
1492
+ parser.add_argument("--regrade", action="store_true",
1493
+ help="rebuild every record from its saved stream and world with this tool; no model runs")
1494
+ parser.add_argument("--check-oracles", action="store_true",
1495
+ help="prove every task's checks pass on its good trajectory and fail on its bad ones")
1496
+ args = parser.parse_args(argv)
1497
+ try:
1498
+ only = [x for x in args.tasks.split(",") if x] if args.tasks else None
1499
+ if args.check_oracles:
1500
+ problems = check_oracles(load_tasks(args.tasks_dir, only))
1501
+ for problem in problems:
1502
+ print(f"oracle: {problem}", file=sys.stderr)
1503
+ print("paired_eval_oracles_ok" if not problems else f"paired_eval_oracles_failed {len(problems)}")
1504
+ return 1 if problems else 0
1505
+ if args.out is None:
1506
+ raise TaskError("--out is required")
1507
+ out = args.out.resolve()
1508
+ if args.report_only or args.regrade:
1509
+ plan = read_json(out / "plan.json")
1510
+ if plan is None:
1511
+ raise TaskError("no plan.json under --out")
1512
+ lock = lock_output(out)
1513
+ try:
1514
+ arm_dirs = {a: out / "arms" / a for a, info in plan["arms"].items() if info.get("commit")}
1515
+ if args.regrade:
1516
+ ctx = make_context(out, plan, arm_dirs, checkout_roots(args.repo.resolve()))
1517
+ print(f"regraded {regrade(ctx, out, load_tasks(args.tasks_dir))} records", file=sys.stderr)
1518
+ record_integrity(out, arm_dirs, {arm: frozen_manifest(out, plan, arm) for arm in arm_dirs})
1519
+ write_report(out, plan, load_records(out, plan))
1520
+ finally:
1521
+ lock.close()
1522
+ print(out / "report.md")
1523
+ return 0
1524
+ arms = [a for a in args.arms.split(",") if a]
1525
+ if not arms or len(set(arms)) != len(arms) or any(a not in ARMS for a in arms):
1526
+ raise TaskError(f"--arms must name distinct arms from {ARMS}")
1527
+ for arm, ref in (("base", args.base), ("candidate", args.candidate)):
1528
+ if arm in arms and not ref:
1529
+ raise TaskError(f"--{arm} is required for the {arm} arm")
1530
+ if args.jobs < 1 or args.max_runs < 1 or not 0 < args.stop_util <= 1 or args.max_budget_usd <= 0:
1531
+ raise TaskError("--jobs, --max-runs, --stop-util and --max-budget-usd must be positive")
1532
+ if args.samples is not None and not 1 <= args.samples <= 20:
1533
+ raise TaskError("--samples must be 1..20")
1534
+ return run_batch(args, out, arms, only)
1535
+ except TaskError as exc:
1536
+ print(f"skill-paired-eval: {exc}", file=sys.stderr)
1537
+ return 2
1538
+ except KeyboardInterrupt:
1539
+ print("interrupted", file=sys.stderr)
1540
+ return 130
1541
+ except BrokenPipeError:
1542
+ return 0
1543
+
1544
+
1545
+ if __name__ == "__main__":
1546
+ sys.exit(main())