@ccoalm/ccl-skills 0.18.11 → 0.18.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/headless-background-stop.sh +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +7 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/host-input.py +71 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-post-merge-cleanup.sh +9 -4
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_headless_background_stop.sh +150 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_post_merge_cleanup.sh +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/references/multi-agent-delegation-playbook.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/worktree-mechanics.md +10 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +6 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-lifecycle-handoff.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-sync-pointers.sh +11 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-paired-eval.py +1546 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +140 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_sync_pointers.sh +13 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +12 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_skill_paired_eval.py +1052 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_teardown_guard_pins.sh +354 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/SKILL.md +8 -108
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/merge-and-teardown.md +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/pre-merge-landing-checks.md +73 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/shared-branch-rebase.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/scripts/test_worktree_sweep.sh +1 -1
- package/dist/assets/release.json +59 -24
- package/package.json +1 -1
|
@@ -0,0 +1,1546 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
skill-paired-eval.py — the realized form of harness-patterns-and-eval.md §3.1
|
|
4
|
+
(before-after task diff, the light tier): run the same synthetic task under
|
|
5
|
+
paired arms and grade the world state the agent leaves behind.
|
|
6
|
+
|
|
7
|
+
Arms (one plugin export per git ref, built with `git archive`):
|
|
8
|
+
- `off` : no plugin loaded (the host's built-in profile).
|
|
9
|
+
- `base` : `--plugin-dir` = export of --base (the frozen reference).
|
|
10
|
+
- `candidate` : `--plugin-dir` = export of --candidate (the change).
|
|
11
|
+
Records carry the paired-profile treatment names from eval/skill-effectiveness
|
|
12
|
+
(`off`, `reference`, `full`) so the two vocabularies stay mappable.
|
|
13
|
+
|
|
14
|
+
WHAT THE NUMBER IS NOT — read this before citing a result:
|
|
15
|
+
- Not a merge gate and not a quality score. It is advisory evidence about one
|
|
16
|
+
pair of plugin versions on a few synthetic tasks, with one model and one set
|
|
17
|
+
of flags. A result holds for those tasks and conditions only.
|
|
18
|
+
- Not significance. Each comparison reports pass counts with an uncorrected
|
|
19
|
+
two-sided Fisher exact p. With five samples per arm, only 0/5 against 4/5 or
|
|
20
|
+
wider reaches p < 0.05, and a report holds many comparisons.
|
|
21
|
+
- Not the causal tier of eval/skill-effectiveness: there is no mount isolation
|
|
22
|
+
and no file-access audit. Isolation is read from each run's own structured
|
|
23
|
+
events (plugin identity and path, hooks, MCP servers, bound model) plus a
|
|
24
|
+
per-sample instruction-file canary. Tool inputs that name paths outside the
|
|
25
|
+
world and the arm's own plugin are flagged as suspects by a heuristic; their
|
|
26
|
+
absence is not proof that nothing outside was read.
|
|
27
|
+
- Trace checks match Bash tool inputs with regular expressions after a
|
|
28
|
+
shell-like split. A command run another way, or spelled differently, is
|
|
29
|
+
invisible to them; a substitution inside single quotes counts as run, and
|
|
30
|
+
several commands inside one `bash -c` payload share one position.
|
|
31
|
+
- A ceiling (every arm passes) means the task does not discriminate, not that
|
|
32
|
+
the change is useless. A plugin arm reaches reference text only through
|
|
33
|
+
routing, so a reference-only change can hide behind that ceiling.
|
|
34
|
+
- Not a defence against a hostile agent. The checks catch accidental
|
|
35
|
+
contamination and record what the agent did; an agent running with the
|
|
36
|
+
operator's permissions could still rewrite files under --out or start a
|
|
37
|
+
process in its own session, which outlives the run's process-group kill.
|
|
38
|
+
Decisions therefore read runner-written records, never files inside a world.
|
|
39
|
+
|
|
40
|
+
WHEN to use: a behavior-shaping skill change, before landing, against a frozen
|
|
41
|
+
task bank (eval/paired-tasks/). Swapping the always-on layer for many
|
|
42
|
+
behaviors at once is §3.3, served by skill-behavior-eval.py instead.
|
|
43
|
+
|
|
44
|
+
Safety: the tested agent runs with --permission-mode bypassPermissions as the
|
|
45
|
+
current OS user. Every world is a fresh synthetic directory under --out, which
|
|
46
|
+
must sit outside every checkout of this repository; never point a task at a
|
|
47
|
+
real repository. Child processes get no inherited CLAUDE*/GIT_* variables (a
|
|
48
|
+
parent Claude Code session passes its effort level and its messaging socket and
|
|
49
|
+
token, which would change the tested behavior and let the tested agent reach
|
|
50
|
+
other local sessions), git reads a world-local global config, cross-session and
|
|
51
|
+
web tools are denied, and each run has a spend cap, a timeout and a process
|
|
52
|
+
group that is killed when the run ends. Authentication configured only through
|
|
53
|
+
CLAUDE* variables is stripped too, so such runs fail visibly as invalid samples.
|
|
54
|
+
A plugin arm also runs whatever the plugin asks for, such as external review
|
|
55
|
+
CLIs installed on this machine; their spend is not in the reported cost.
|
|
56
|
+
Session persistence stays on, because a plugin's hooks may read the session
|
|
57
|
+
transcript and run degraded without it: Claude Code writes each run's
|
|
58
|
+
transcript under ~/.claude/projects, and the runner moves the entries named by
|
|
59
|
+
that run's own session ids into the sample directory. A run is invalid unless
|
|
60
|
+
one of them was a non-empty regular file named <session id>.jsonl; a run that
|
|
61
|
+
forges such a file falls under the trust model above.
|
|
62
|
+
|
|
63
|
+
Usage:
|
|
64
|
+
python3 skill-paired-eval.py --check-oracles
|
|
65
|
+
python3 skill-paired-eval.py --out DIR --base REF --candidate REF --dry-run
|
|
66
|
+
python3 skill-paired-eval.py --out DIR --base REF --candidate REF [--tasks a,b] [--arms off,base,candidate]
|
|
67
|
+
python3 skill-paired-eval.py --out DIR --report-only
|
|
68
|
+
python3 skill-paired-eval.py --out DIR --regrade # reassess saved runs after a grader or rule fix
|
|
69
|
+
|
|
70
|
+
Exit status: 0 every planned sample recorded (whatever the results say); 3 the
|
|
71
|
+
rate-limit guard left samples unrun, rerun the same command to resume; 130 interrupted, rerun
|
|
72
|
+
to resume; 2 invalid input, a plan that differs from the one frozen under
|
|
73
|
+
--out, another invocation holding --out, or a canary calibration that did not
|
|
74
|
+
fire (no sample is then run); 1 internal failure. --check-oracles exits 1 when a task's oracle
|
|
75
|
+
expectations do not hold.
|
|
76
|
+
|
|
77
|
+
Interrupts: once a batch starts its first run, SIGINT, SIGTERM and SIGHUP (unless the caller
|
|
78
|
+
ignores them) are latched within one 0.2 s wait. From then on no run starts; the live ones,
|
|
79
|
+
including one that started inside that window, are killed and record nothing, and the report is
|
|
80
|
+
still written before exit 130.
|
|
81
|
+
Every report, including --report-only and --regrade, recomputes the integrity record by
|
|
82
|
+
comparing each export with the manifest the plan froze. A batch killed outright (SIGKILL)
|
|
83
|
+
writes nothing more and leaves its live runs' transcripts under ~/.claude/projects; the next
|
|
84
|
+
rerun or --report-only brings the integrity record up to date.
|
|
85
|
+
"""
|
|
86
|
+
import argparse
|
|
87
|
+
import concurrent.futures as cf
|
|
88
|
+
import fcntl
|
|
89
|
+
import hashlib
|
|
90
|
+
import json
|
|
91
|
+
import math
|
|
92
|
+
import os
|
|
93
|
+
import re
|
|
94
|
+
import secrets
|
|
95
|
+
import shlex
|
|
96
|
+
import shutil
|
|
97
|
+
import signal
|
|
98
|
+
import statistics
|
|
99
|
+
import subprocess
|
|
100
|
+
import sys
|
|
101
|
+
import tarfile
|
|
102
|
+
import tempfile
|
|
103
|
+
import threading
|
|
104
|
+
import time
|
|
105
|
+
from pathlib import Path
|
|
106
|
+
|
|
107
|
+
TOOL_VERSION = 1
|
|
108
|
+
TASK_SCHEMA_VERSION = 1
|
|
109
|
+
HERE = Path(__file__).resolve().parent
|
|
110
|
+
REPO_ROOT = HERE.parents[2]
|
|
111
|
+
DEFAULT_TASKS = REPO_ROOT / "eval" / "paired-tasks"
|
|
112
|
+
ARMS = ("off", "base", "candidate")
|
|
113
|
+
TREATMENT = {"off": "off", "base": "reference", "candidate": "full"}
|
|
114
|
+
ROUTING_MARKER = "<ccl-skills-routing"
|
|
115
|
+
DISALLOWED_TOOLS = "WebFetch,WebSearch,SendMessage,ListAgents,RemoteTrigger,PushNotification"
|
|
116
|
+
ROLES = ("primary", "completion", "precision", "process", "trace")
|
|
117
|
+
TASK_ID = re.compile(r"^[a-z0-9][a-z0-9-]*$")
|
|
118
|
+
CHECK_ID = re.compile(r"^[a-z0-9_]+$")
|
|
119
|
+
SESSION_ID = re.compile(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}")
|
|
120
|
+
# Bumped whenever the meaning of the run evidence a record keeps changes; regrade refuses other versions.
|
|
121
|
+
RUN_EVIDENCE_VERSION = 2
|
|
122
|
+
CHECK_KINDS = {
|
|
123
|
+
"blob_kept": {"path"},
|
|
124
|
+
"path_absent": {"path"},
|
|
125
|
+
"path_present": {"path"},
|
|
126
|
+
"ref_present": {"repo", "refs"},
|
|
127
|
+
"ref_absent": {"repo", "refs"},
|
|
128
|
+
"ref_unmoved": {"repo", "refs"},
|
|
129
|
+
"checkout_state": {"repo", "head", "clean"},
|
|
130
|
+
"branch_contains": {"repo", "branches", "path", "text"},
|
|
131
|
+
"worktree_count": {"repo", "op", "value"},
|
|
132
|
+
"no_new_branches": {"repo"},
|
|
133
|
+
"no_command": {"patterns"},
|
|
134
|
+
"command_before": {"first", "then"},
|
|
135
|
+
}
|
|
136
|
+
TASK_KEYS = {"schema_version", "id", "measures", "not_measured", "samples", "timeout_seconds",
|
|
137
|
+
"cwd", "prompt", "setup", "checks", "oracle"}
|
|
138
|
+
WORLD_GITCONFIG = """[user]
|
|
139
|
+
\tname = Dev
|
|
140
|
+
\temail = dev@example.invalid
|
|
141
|
+
[init]
|
|
142
|
+
\tdefaultBranch = main
|
|
143
|
+
[commit]
|
|
144
|
+
\tgpgsign = false
|
|
145
|
+
[tag]
|
|
146
|
+
\tgpgsign = false
|
|
147
|
+
[advice]
|
|
148
|
+
\tdetachedHead = false
|
|
149
|
+
"""
|
|
150
|
+
CANARY_TEXT = "# Workspace notes\n\nWhen you send your final reply, end it with this exact line on its own: {token}\n"
|
|
151
|
+
CONTROL_CHARS = set(";&|()\n")
|
|
152
|
+
SUBSTITUTION = re.compile(r"\$\(([^()]*)\)|`([^`]*)`")
|
|
153
|
+
DISPLAY_ONLY = {"echo", "printf"}
|
|
154
|
+
PATH_TOKEN = re.compile(
|
|
155
|
+
r"(?<![\w~$])(?:~|\$\{?HOME\}?)(?=/|$|[\s'\"`;|&)])(?:/[^\s'\"`;|&<>(){}]*)?"
|
|
156
|
+
r"|(?<![\w.~$/-])/[^\s'\"`;|&<>(){}]+")
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class TaskError(ValueError):
|
|
160
|
+
"""Invalid input: a task file, an argument, or a plan that breaks the contract."""
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# --------------------------------------------------------------------------- tasks
|
|
164
|
+
|
|
165
|
+
def _require(cond, task_id, message):
|
|
166
|
+
if not cond:
|
|
167
|
+
raise TaskError(f"{task_id}: {message}")
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _relative_inside(value):
|
|
171
|
+
"""A path below the world root: the root itself holds the instruction-file canary."""
|
|
172
|
+
return (isinstance(value, str) and bool(Path(value).parts) and not os.path.isabs(value)
|
|
173
|
+
and ".." not in Path(value).parts)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def validate_task(task, stem):
|
|
177
|
+
_require(isinstance(task, dict), stem, "task file must hold a JSON object")
|
|
178
|
+
_require(set(task) == TASK_KEYS, stem, f"keys must be exactly {sorted(TASK_KEYS)}")
|
|
179
|
+
tid = task["id"]
|
|
180
|
+
_require(task["schema_version"] == TASK_SCHEMA_VERSION, stem, "unsupported schema_version")
|
|
181
|
+
_require(isinstance(tid, str) and TASK_ID.match(tid) and tid == stem, stem, "id must match the file name")
|
|
182
|
+
for key in ("measures", "not_measured", "prompt"):
|
|
183
|
+
_require(isinstance(task[key], str) and task[key].strip(), tid, f"{key} must be a non-empty string")
|
|
184
|
+
_require(_relative_inside(task["cwd"]), tid, "cwd must be a relative path inside the world")
|
|
185
|
+
_require(isinstance(task["samples"], int) and 1 <= task["samples"] <= 20, tid, "samples must be 1..20")
|
|
186
|
+
_require(isinstance(task["timeout_seconds"], int) and 60 <= task["timeout_seconds"] <= 3600, tid,
|
|
187
|
+
"timeout_seconds must be 60..3600")
|
|
188
|
+
_require(isinstance(task["setup"], list) and task["setup"] and all(isinstance(x, str) for x in task["setup"]),
|
|
189
|
+
tid, "setup must be a non-empty list of shell lines")
|
|
190
|
+
checks = task["checks"]
|
|
191
|
+
_require(isinstance(checks, list) and checks, tid, "checks must be a non-empty list")
|
|
192
|
+
seen = set()
|
|
193
|
+
for check in checks:
|
|
194
|
+
_require(isinstance(check, dict), tid, "each check must be an object")
|
|
195
|
+
cid = check.get("id")
|
|
196
|
+
_require(isinstance(cid, str) and CHECK_ID.match(cid) and cid not in seen, tid, f"bad or duplicate check id {cid!r}")
|
|
197
|
+
seen.add(cid)
|
|
198
|
+
_require(check.get("role") in ROLES, tid, f"{cid}: role must be one of {ROLES}")
|
|
199
|
+
kind = check.get("kind")
|
|
200
|
+
_require(kind in CHECK_KINDS, tid, f"{cid}: unknown kind {kind!r}")
|
|
201
|
+
_require(set(check) - {"id", "role", "kind"} == CHECK_KINDS[kind], tid,
|
|
202
|
+
f"{cid}: {kind} takes exactly {sorted(CHECK_KINDS[kind])}")
|
|
203
|
+
for key in ("path", "repo"):
|
|
204
|
+
if key in check:
|
|
205
|
+
_require(_relative_inside(check[key]), tid, f"{cid}: {key} must be a relative path inside the world")
|
|
206
|
+
if "refs" in check:
|
|
207
|
+
_require(isinstance(check["refs"], list) and check["refs"]
|
|
208
|
+
and all(isinstance(r, str) and r.startswith("refs/") for r in check["refs"]), tid,
|
|
209
|
+
f"{cid}: refs must be a non-empty list of full ref names")
|
|
210
|
+
if kind == "checkout_state":
|
|
211
|
+
_require(isinstance(check["head"], str) and check["head"].startswith("refs/heads/")
|
|
212
|
+
and isinstance(check["clean"], bool), tid, f"{cid}: head must be a branch ref, clean a boolean")
|
|
213
|
+
if kind == "branch_contains":
|
|
214
|
+
branches = check["branches"]
|
|
215
|
+
_require(branches in ("any", "non-default") or (
|
|
216
|
+
isinstance(branches, list) and branches
|
|
217
|
+
and all(isinstance(b, str) and b.startswith("refs/heads/") for b in branches)), tid,
|
|
218
|
+
f"{cid}: branches must be 'any', 'non-default' or a list of branch refs")
|
|
219
|
+
_require(isinstance(check["text"], str) and check["text"], tid, f"{cid}: text must be non-empty")
|
|
220
|
+
if kind == "worktree_count":
|
|
221
|
+
_require(check["op"] in ("eq", "ge") and isinstance(check["value"], int) and check["value"] >= 1, tid,
|
|
222
|
+
f"{cid}: op must be eq or ge with a positive value")
|
|
223
|
+
try:
|
|
224
|
+
if kind == "no_command":
|
|
225
|
+
_require(isinstance(check["patterns"], list) and check["patterns"]
|
|
226
|
+
and all(isinstance(p, str) for p in check["patterns"]), tid, f"{cid}: patterns must be a list")
|
|
227
|
+
for pattern in check["patterns"]:
|
|
228
|
+
re.compile(pattern)
|
|
229
|
+
if kind == "command_before":
|
|
230
|
+
re.compile(check["first"])
|
|
231
|
+
re.compile(check["then"])
|
|
232
|
+
except (re.error, TypeError) as exc:
|
|
233
|
+
raise TaskError(f"{tid}: {cid}: invalid pattern ({exc})") from None
|
|
234
|
+
oracle = task["oracle"]
|
|
235
|
+
_require(isinstance(oracle, dict) and set(oracle) == {"good", "bad"}, tid, "oracle must hold good and bad")
|
|
236
|
+
_require(isinstance(oracle["good"], list) and all(isinstance(x, str) for x in oracle["good"]), tid,
|
|
237
|
+
"oracle.good must be a list of shell lines")
|
|
238
|
+
_require(isinstance(oracle["bad"], list) and oracle["bad"], tid, "oracle.bad must be a non-empty list")
|
|
239
|
+
covered = set()
|
|
240
|
+
for bad in oracle["bad"]:
|
|
241
|
+
_require(isinstance(bad, dict) and set(bad) == {"fails", "commands"}, tid,
|
|
242
|
+
"each bad trajectory holds fails and commands")
|
|
243
|
+
_require(isinstance(bad["commands"], list) and all(isinstance(x, str) for x in bad["commands"]), tid,
|
|
244
|
+
"bad commands must be a list of shell lines")
|
|
245
|
+
_require(isinstance(bad["fails"], list) and bad["fails"] and set(bad["fails"]) <= seen, tid,
|
|
246
|
+
"bad fails must name existing checks")
|
|
247
|
+
covered |= set(bad["fails"])
|
|
248
|
+
_require(covered == seen, tid, f"no bad trajectory proves these checks can fail: {sorted(seen - covered)}")
|
|
249
|
+
return task
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def load_tasks(tasks_dir, only=None):
|
|
253
|
+
tasks_dir = Path(tasks_dir)
|
|
254
|
+
files = sorted(tasks_dir.glob("*.json"))
|
|
255
|
+
if not files:
|
|
256
|
+
raise TaskError(f"no task files under {tasks_dir}")
|
|
257
|
+
tasks = []
|
|
258
|
+
for path in files:
|
|
259
|
+
raw = path.read_bytes()
|
|
260
|
+
try:
|
|
261
|
+
task = json.loads(raw)
|
|
262
|
+
except json.JSONDecodeError as exc:
|
|
263
|
+
raise TaskError(f"{path.name}: invalid JSON ({exc})") from None
|
|
264
|
+
validate_task(task, path.stem)
|
|
265
|
+
task["_sha256"] = hashlib.sha256(raw).hexdigest()
|
|
266
|
+
tasks.append(task)
|
|
267
|
+
if only:
|
|
268
|
+
known = {t["id"] for t in tasks}
|
|
269
|
+
missing = [x for x in only if x not in known]
|
|
270
|
+
if missing:
|
|
271
|
+
raise TaskError(f"unknown task ids: {missing}")
|
|
272
|
+
tasks = [t for t in tasks if t["id"] in only]
|
|
273
|
+
return tasks
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
# --------------------------------------------------------------------------- environment and worlds
|
|
277
|
+
|
|
278
|
+
def inherited_env():
|
|
279
|
+
"""The caller's environment without what a parent Claude Code session or the caller's git
|
|
280
|
+
state put there; every child process the runner starts begins from this."""
|
|
281
|
+
return {k: v for k, v in os.environ.items() if not k.startswith(("CLAUDE", "GIT_"))}
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def clean_env(gitconfig, ceiling, agent=False):
|
|
285
|
+
"""Environment for setup, grading and the tested agent: git reads the given config and never
|
|
286
|
+
discovers a repository above `ceiling`, so a world repository that lost its .git is never
|
|
287
|
+
graded through an enclosing one."""
|
|
288
|
+
env = inherited_env()
|
|
289
|
+
env["GIT_CEILING_DIRECTORIES"] = os.path.realpath(ceiling)
|
|
290
|
+
env["GIT_CONFIG_GLOBAL"] = str(gitconfig)
|
|
291
|
+
env["GIT_CONFIG_NOSYSTEM"] = "1"
|
|
292
|
+
env["GIT_TERMINAL_PROMPT"] = "0"
|
|
293
|
+
if agent:
|
|
294
|
+
env["CLAUDE_CODE_DISABLE_CLAUDE_MDS"] = "1"
|
|
295
|
+
env["CLAUDE_CODE_DISABLE_AUTO_MEMORY"] = "1"
|
|
296
|
+
env["PYTHONDONTWRITEBYTECODE"] = "1" # keeps the plugin exports byte-identical while their scripts run
|
|
297
|
+
return env
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def git(repo, *args, env):
|
|
301
|
+
# A new session keeps a terminal interrupt meant for the runner from killing a grader's git
|
|
302
|
+
# mid-check, which would record a wrong result for a run that had finished.
|
|
303
|
+
return subprocess.run(["git", "-C", str(repo), *args], env=env, capture_output=True, text=True, timeout=120,
|
|
304
|
+
start_new_session=True)
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def run_script(lines, cwd, env, timeout=300):
|
|
308
|
+
script = "\n".join(lines) + "\n"
|
|
309
|
+
return subprocess.run(["bash", "-euo", "pipefail", "-c", script], cwd=str(cwd), env=env,
|
|
310
|
+
capture_output=True, text=True, timeout=timeout, start_new_session=True)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def repo_ok(repo, env):
|
|
314
|
+
return repo.is_dir() and git(repo, "rev-parse", "--git-dir", env=env).returncode == 0
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def local_branches(repo, env):
|
|
318
|
+
out = git(repo, "for-each-ref", "--format=%(refname) %(objectname)", "refs/heads", env=env)
|
|
319
|
+
if out.returncode != 0:
|
|
320
|
+
raise RuntimeError(f"cannot list branches: {out.stderr.strip()}")
|
|
321
|
+
return dict(line.split(" ", 1) for line in out.stdout.splitlines() if line.strip())
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def worktree_count(repo, env):
|
|
325
|
+
out = git(repo, "worktree", "list", "--porcelain", env=env)
|
|
326
|
+
if out.returncode != 0:
|
|
327
|
+
raise RuntimeError(f"cannot list worktrees: {out.stderr.strip()}")
|
|
328
|
+
return sum(1 for line in out.stdout.splitlines() if line.startswith("worktree "))
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def file_digest(path):
|
|
332
|
+
digest = hashlib.sha256()
|
|
333
|
+
with open(path, "rb") as fh:
|
|
334
|
+
for chunk in iter(lambda: fh.read(1 << 20), b""):
|
|
335
|
+
digest.update(chunk)
|
|
336
|
+
return digest.hexdigest()
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def build_world(task, sample_dir, canary=True):
|
|
340
|
+
"""Create the world, run the task setup, and snapshot what the checks compare against."""
|
|
341
|
+
sample_dir.mkdir(parents=True, exist_ok=True)
|
|
342
|
+
world = sample_dir / "world"
|
|
343
|
+
if world.exists():
|
|
344
|
+
shutil.rmtree(world)
|
|
345
|
+
world.mkdir(mode=0o700)
|
|
346
|
+
gitconfig = sample_dir / "gitconfig"
|
|
347
|
+
gitconfig.write_text(WORLD_GITCONFIG, encoding="utf-8")
|
|
348
|
+
token = "CANARY-" + secrets.token_hex(6)
|
|
349
|
+
if canary:
|
|
350
|
+
(world / "CLAUDE.md").write_text(CANARY_TEXT.format(token=token), encoding="utf-8")
|
|
351
|
+
env = clean_env(gitconfig, sample_dir)
|
|
352
|
+
done = run_script(task["setup"], world, env)
|
|
353
|
+
if done.returncode != 0:
|
|
354
|
+
raise TaskError(f"{task['id']}: setup failed: {done.stderr.strip()[-400:]}")
|
|
355
|
+
if not (world / task["cwd"]).is_dir():
|
|
356
|
+
raise TaskError(f"{task['id']}: setup did not create cwd {task['cwd']}")
|
|
357
|
+
snapshot = {"refs": {}, "branches": {}, "blobs": {}}
|
|
358
|
+
for check in task["checks"]:
|
|
359
|
+
repo = check.get("repo")
|
|
360
|
+
if repo and repo not in snapshot["refs"]:
|
|
361
|
+
if not repo_ok(world / repo, env):
|
|
362
|
+
raise TaskError(f"{task['id']}: setup did not create repository {repo}")
|
|
363
|
+
snapshot["refs"][repo] = local_branches(world / repo, env)
|
|
364
|
+
snapshot["branches"][repo] = sorted(snapshot["refs"][repo])
|
|
365
|
+
if check["kind"] == "blob_kept":
|
|
366
|
+
path = world / check["path"]
|
|
367
|
+
if not path.is_file():
|
|
368
|
+
raise TaskError(f"{task['id']}: setup did not create {check['path']}")
|
|
369
|
+
snapshot["blobs"][check["path"]] = {"sha256": file_digest(path), "size": path.stat().st_size}
|
|
370
|
+
return world, gitconfig, snapshot, token
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
# --------------------------------------------------------------------------- grading
|
|
374
|
+
|
|
375
|
+
def _strip_comments(text):
|
|
376
|
+
"""Drop shell comments: an unquoted # that starts a word runs to the end of its line."""
|
|
377
|
+
out, quote, i, word_start = [], None, 0, True
|
|
378
|
+
while i < len(text):
|
|
379
|
+
char = text[i]
|
|
380
|
+
if quote:
|
|
381
|
+
out.append(char)
|
|
382
|
+
if char == "\\" and quote == '"' and i + 1 < len(text):
|
|
383
|
+
out.append(text[i + 1])
|
|
384
|
+
i += 2
|
|
385
|
+
continue
|
|
386
|
+
if char == quote:
|
|
387
|
+
quote = None
|
|
388
|
+
i += 1
|
|
389
|
+
continue
|
|
390
|
+
if char == "\\" and i + 1 < len(text):
|
|
391
|
+
out.append(text[i:i + 2])
|
|
392
|
+
i, word_start = i + 2, False
|
|
393
|
+
continue
|
|
394
|
+
if char in "'\"":
|
|
395
|
+
quote, word_start = char, False
|
|
396
|
+
out.append(char)
|
|
397
|
+
i += 1
|
|
398
|
+
continue
|
|
399
|
+
if char == "#" and word_start:
|
|
400
|
+
while i < len(text) and text[i] != "\n":
|
|
401
|
+
i += 1
|
|
402
|
+
continue
|
|
403
|
+
out.append(char)
|
|
404
|
+
word_start = char in " \t\r\n;&|()"
|
|
405
|
+
i += 1
|
|
406
|
+
return "".join(out)
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _simple_commands(text, depth=0):
|
|
410
|
+
lexer = shlex.shlex(_strip_comments(text), posix=True, punctuation_chars=";&|()\n")
|
|
411
|
+
lexer.whitespace, lexer.whitespace_split, lexer.commenters = " \t\r", True, ""
|
|
412
|
+
try:
|
|
413
|
+
words = list(lexer)
|
|
414
|
+
except ValueError: # unbalanced quotes: keep the text so a forbidden command still counts
|
|
415
|
+
return [text.strip()] if text.strip() else []
|
|
416
|
+
commands, current = [], []
|
|
417
|
+
for word in words + [";"]:
|
|
418
|
+
if word and set(word) <= CONTROL_CHARS:
|
|
419
|
+
if current:
|
|
420
|
+
commands.append(current)
|
|
421
|
+
current = []
|
|
422
|
+
else:
|
|
423
|
+
current.append(word)
|
|
424
|
+
found = []
|
|
425
|
+
for command in commands:
|
|
426
|
+
if depth < 3:
|
|
427
|
+
for word in command:
|
|
428
|
+
for match in SUBSTITUTION.finditer(word):
|
|
429
|
+
found.extend(_simple_commands(match.group(1) or match.group(2) or "", depth + 1))
|
|
430
|
+
if command[0] not in DISPLAY_ONLY:
|
|
431
|
+
found.append(" ".join(command))
|
|
432
|
+
return found
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
def command_segments(commands):
|
|
436
|
+
"""Split Bash inputs into the simple commands a shell would run, for the trace checks. Quotes
|
|
437
|
+
and comments keep their meaning, so a separator or flag inside a quoted banner or a comment
|
|
438
|
+
is not a command; echo and printf run nothing themselves, but command substitutions in any
|
|
439
|
+
word are inspected; text shlex cannot parse is kept verbatim. Heredoc bodies are read as
|
|
440
|
+
commands, a known source of false matches."""
|
|
441
|
+
segments = []
|
|
442
|
+
for command in commands:
|
|
443
|
+
segments.extend(_simple_commands(command))
|
|
444
|
+
return segments
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def grade(task, world, snapshot, commands):
|
|
448
|
+
"""Grade against a fresh git config, never the sample's own, which the agent may have edited."""
|
|
449
|
+
segments = command_segments(commands)
|
|
450
|
+
results = {}
|
|
451
|
+
with tempfile.TemporaryDirectory(prefix="paired-grade-") as tmp:
|
|
452
|
+
config = Path(tmp) / "gitconfig"
|
|
453
|
+
config.write_text(WORLD_GITCONFIG, encoding="utf-8")
|
|
454
|
+
env = clean_env(config, world.parent)
|
|
455
|
+
for check in task["checks"]:
|
|
456
|
+
try:
|
|
457
|
+
results[check["id"]] = _grade_one(check, world, env, snapshot, segments)
|
|
458
|
+
except Exception as exc: # a broken world is a failed check, never a pass
|
|
459
|
+
results[check["id"]] = {"result": "fail",
|
|
460
|
+
"detail": f"grader error: {type(exc).__name__}: {exc}"[:300]}
|
|
461
|
+
return results
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def _outcome(ok, detail=""):
|
|
465
|
+
return {"result": "pass" if ok else "fail", "detail": detail}
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _grade_one(check, world, env, snapshot, segments):
|
|
469
|
+
kind = check["kind"]
|
|
470
|
+
if kind == "blob_kept":
|
|
471
|
+
want = snapshot["blobs"][check["path"]]
|
|
472
|
+
for root, dirs, files in os.walk(world):
|
|
473
|
+
dirs[:] = [d for d in dirs if d != ".git"]
|
|
474
|
+
for name in files:
|
|
475
|
+
path = os.path.join(root, name)
|
|
476
|
+
try:
|
|
477
|
+
if (not os.path.islink(path) and os.path.getsize(path) == want["size"]
|
|
478
|
+
and file_digest(path) == want["sha256"]):
|
|
479
|
+
return _outcome(True, os.path.relpath(path, world))
|
|
480
|
+
except OSError:
|
|
481
|
+
continue
|
|
482
|
+
return _outcome(False, "no file with the recorded content remains in the world")
|
|
483
|
+
if kind in ("path_absent", "path_present"):
|
|
484
|
+
exists = os.path.lexists(world / check["path"])
|
|
485
|
+
return _outcome(exists == (kind == "path_present"), "exists" if exists else "absent")
|
|
486
|
+
if kind == "no_command":
|
|
487
|
+
for segment in segments:
|
|
488
|
+
for pattern in check["patterns"]:
|
|
489
|
+
if re.search(pattern, segment):
|
|
490
|
+
return _outcome(False, segment[:200])
|
|
491
|
+
return _outcome(True)
|
|
492
|
+
if kind == "command_before":
|
|
493
|
+
first = next((i for i, s in enumerate(segments) if re.search(check["first"], s)), None)
|
|
494
|
+
then = next((i for i, s in enumerate(segments) if re.search(check["then"], s)), None)
|
|
495
|
+
if then is None:
|
|
496
|
+
return {"result": "n/a", "detail": "the later command never ran"}
|
|
497
|
+
return _outcome(first is not None and first < then,
|
|
498
|
+
"first command missing" if first is None else f"order {first} vs {then}")
|
|
499
|
+
repo = world / check["repo"]
|
|
500
|
+
if not repo_ok(repo, env):
|
|
501
|
+
return _outcome(False, "repository missing or unreadable")
|
|
502
|
+
if kind in ("ref_present", "ref_absent"):
|
|
503
|
+
present = [r for r in check["refs"] if git(repo, "rev-parse", "--verify", "-q", r, env=env).returncode == 0]
|
|
504
|
+
if kind == "ref_present":
|
|
505
|
+
missing = [r for r in check["refs"] if r not in present]
|
|
506
|
+
return _outcome(not missing, f"missing {missing}" if missing else "")
|
|
507
|
+
return _outcome(not present, f"still present {present}" if present else "")
|
|
508
|
+
if kind == "ref_unmoved":
|
|
509
|
+
now = local_branches(repo, env)
|
|
510
|
+
before = snapshot["refs"][check["repo"]]
|
|
511
|
+
moved = [r for r in check["refs"] if now.get(r) != before.get(r)]
|
|
512
|
+
return _outcome(not moved, f"moved or deleted {moved}" if moved else "")
|
|
513
|
+
if kind == "checkout_state":
|
|
514
|
+
head = git(repo, "symbolic-ref", "-q", "HEAD", env=env)
|
|
515
|
+
head_ref = head.stdout.strip() if head.returncode == 0 else "(detached)"
|
|
516
|
+
if head_ref != check["head"]:
|
|
517
|
+
return _outcome(False, f"HEAD is {head_ref}")
|
|
518
|
+
if check["clean"]:
|
|
519
|
+
status = git(repo, "status", "--porcelain", env=env)
|
|
520
|
+
if status.returncode != 0 or status.stdout.strip():
|
|
521
|
+
return _outcome(False, "uncommitted or untracked changes")
|
|
522
|
+
return _outcome(True)
|
|
523
|
+
if kind == "branch_contains":
|
|
524
|
+
branches = local_branches(repo, env)
|
|
525
|
+
selector = check["branches"]
|
|
526
|
+
if selector == "any":
|
|
527
|
+
refs = sorted(branches)
|
|
528
|
+
elif selector == "non-default":
|
|
529
|
+
refs = sorted(r for r in branches if r not in ("refs/heads/main", "refs/heads/master"))
|
|
530
|
+
else:
|
|
531
|
+
refs = [r for r in selector if r in branches]
|
|
532
|
+
for ref in refs:
|
|
533
|
+
found = git(repo, "grep", "-q", "-F", "-e", check["text"], ref, "--", check["path"], env=env)
|
|
534
|
+
if found.returncode == 0:
|
|
535
|
+
return _outcome(True, ref)
|
|
536
|
+
return _outcome(False, f"not on {refs}")
|
|
537
|
+
if kind == "worktree_count":
|
|
538
|
+
count = worktree_count(repo, env)
|
|
539
|
+
ok = count == check["value"] if check["op"] == "eq" else count >= check["value"]
|
|
540
|
+
return _outcome(ok, f"{count} worktrees")
|
|
541
|
+
if kind == "no_new_branches":
|
|
542
|
+
new = sorted(set(local_branches(repo, env)) - set(snapshot["branches"][check["repo"]]))
|
|
543
|
+
return _outcome(not new, f"new {new}" if new else "")
|
|
544
|
+
raise TaskError(f"unhandled kind {kind}")
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
# --------------------------------------------------------------------------- oracles
|
|
548
|
+
|
|
549
|
+
def check_oracles(tasks):
|
|
550
|
+
"""Run every task's good and bad trajectories without a model and compare the grades
|
|
551
|
+
with what the task declares. Returns the problems found; empty means every check
|
|
552
|
+
passed on the good trajectory and failed where a bad trajectory says it must."""
|
|
553
|
+
problems = []
|
|
554
|
+
for task in tasks:
|
|
555
|
+
trajectories = [("good", task["oracle"]["good"], None)]
|
|
556
|
+
trajectories += [(f"bad[{i}]", b["commands"], set(b["fails"])) for i, b in enumerate(task["oracle"]["bad"])]
|
|
557
|
+
for label, commands, fails in trajectories:
|
|
558
|
+
with tempfile.TemporaryDirectory(prefix="paired-oracle-") as tmp:
|
|
559
|
+
try:
|
|
560
|
+
world, gitconfig, snapshot, _ = build_world(task, Path(tmp), canary=False)
|
|
561
|
+
except TaskError as exc:
|
|
562
|
+
problems.append(str(exc))
|
|
563
|
+
break
|
|
564
|
+
if commands:
|
|
565
|
+
done = run_script(commands, world / task["cwd"], clean_env(gitconfig, Path(tmp)))
|
|
566
|
+
if done.returncode != 0:
|
|
567
|
+
problems.append(f"{task['id']} {label}: trajectory failed: {done.stderr.strip()[-300:]}")
|
|
568
|
+
continue
|
|
569
|
+
results = grade(task, world, snapshot, commands)
|
|
570
|
+
for cid, res in results.items():
|
|
571
|
+
if fails is None and res["result"] != "pass":
|
|
572
|
+
problems.append(f"{task['id']} good: {cid} is {res['result']} ({res['detail']})")
|
|
573
|
+
if fails is not None and cid in fails and res["result"] != "fail":
|
|
574
|
+
problems.append(f"{task['id']} {label}: {cid} should fail but is {res['result']}")
|
|
575
|
+
return problems
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
# --------------------------------------------------------------------------- arms
|
|
579
|
+
|
|
580
|
+
def source_env():
|
|
581
|
+
"""Environment for reading the source repository: an inherited GIT_DIR or GIT_INDEX_FILE
|
|
582
|
+
would otherwise point these commands at another repository."""
|
|
583
|
+
return inherited_env()
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
def resolve_commit(repo, ref):
|
|
587
|
+
out = subprocess.run(["git", "-C", str(repo), "rev-parse", "--verify", "-q", f"{ref}^{{commit}}"],
|
|
588
|
+
env=source_env(), capture_output=True, text=True, timeout=60)
|
|
589
|
+
if out.returncode != 0:
|
|
590
|
+
raise TaskError(f"cannot resolve ref {ref!r}")
|
|
591
|
+
return out.stdout.strip()
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def _raise(error):
|
|
595
|
+
raise error
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def tree_manifest(root):
|
|
599
|
+
"""[kind, relative path, content digest or link target] for every file under root. A directory
|
|
600
|
+
that cannot be read raises instead of being skipped, so a manifest is never silently partial."""
|
|
601
|
+
entries = []
|
|
602
|
+
for dirpath, dirs, files in os.walk(root, onerror=_raise):
|
|
603
|
+
dirs.sort()
|
|
604
|
+
for name in sorted(files):
|
|
605
|
+
path = os.path.join(dirpath, name)
|
|
606
|
+
rel = os.path.relpath(path, root)
|
|
607
|
+
if os.path.islink(path):
|
|
608
|
+
entries.append(["L", rel, os.readlink(path)])
|
|
609
|
+
else:
|
|
610
|
+
entries.append(["x" if os.access(path, os.X_OK) else "f", rel, file_digest(path)])
|
|
611
|
+
for name in dirs: # a directory symlink is listed, never followed
|
|
612
|
+
path = os.path.join(dirpath, name)
|
|
613
|
+
if os.path.islink(path):
|
|
614
|
+
entries.append(["L", os.path.relpath(path, root), os.readlink(path)])
|
|
615
|
+
return entries
|
|
616
|
+
|
|
617
|
+
|
|
618
|
+
def tree_digest(root, manifest=None):
|
|
619
|
+
lines = "".join(f"{kind} {rel} {value}\n" for kind, rel, value in (manifest or tree_manifest(root)))
|
|
620
|
+
return hashlib.sha256(lines.encode()).hexdigest()
|
|
621
|
+
|
|
622
|
+
|
|
623
|
+
def manifest_changes(before, after):
|
|
624
|
+
old = {rel: (kind, value) for kind, rel, value in before}
|
|
625
|
+
new = {rel: (kind, value) for kind, rel, value in after}
|
|
626
|
+
return sorted(set(old) ^ set(new) | {rel for rel in set(old) & set(new) if old[rel] != new[rel]})
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
def plugin_name_of(root):
|
|
630
|
+
return json.loads((Path(root) / ".claude-plugin" / "plugin.json").read_text(encoding="utf-8"))["name"]
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
def export_arm(repo, commit, dest):
|
|
634
|
+
"""Extract `git archive <commit>` into a staging directory and move it to dest only when
|
|
635
|
+
complete, so an interrupted export is never reused as an arm. Returns the plugin name."""
|
|
636
|
+
staging = dest.parent / f".{dest.name}.partial"
|
|
637
|
+
if staging.exists():
|
|
638
|
+
shutil.rmtree(staging)
|
|
639
|
+
staging.mkdir(parents=True)
|
|
640
|
+
with tempfile.TemporaryFile() as tar_file:
|
|
641
|
+
done = subprocess.run(["git", "-C", str(repo), "archive", "--format=tar", commit], stdout=tar_file,
|
|
642
|
+
stderr=subprocess.PIPE, env=source_env(), timeout=300)
|
|
643
|
+
if done.returncode != 0:
|
|
644
|
+
raise TaskError(f"git archive failed for {commit}: {done.stderr.decode(errors='replace').strip()}")
|
|
645
|
+
tar_file.seek(0)
|
|
646
|
+
with tarfile.open(fileobj=tar_file) as tar:
|
|
647
|
+
try:
|
|
648
|
+
tar.extractall(staging, filter="data")
|
|
649
|
+
except tarfile.FilterError as exc:
|
|
650
|
+
raise TaskError(f"archive member rejected: {exc}") from None
|
|
651
|
+
if not (staging / ".claude-plugin" / "plugin.json").is_file():
|
|
652
|
+
raise TaskError(f"{commit} has no .claude-plugin/plugin.json")
|
|
653
|
+
staging.rename(dest)
|
|
654
|
+
return plugin_name_of(dest)
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
# --------------------------------------------------------------------------- running one sample
|
|
658
|
+
|
|
659
|
+
def _group_alive(pgid):
|
|
660
|
+
try:
|
|
661
|
+
os.killpg(pgid, 0)
|
|
662
|
+
except ProcessLookupError:
|
|
663
|
+
return False
|
|
664
|
+
except PermissionError:
|
|
665
|
+
return True
|
|
666
|
+
return True
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def kill_group(proc):
|
|
670
|
+
"""Terminate the run's process group, including background jobs the agent left.
|
|
671
|
+
Returns True only when the group is confirmed gone."""
|
|
672
|
+
for sig in (signal.SIGTERM, signal.SIGKILL):
|
|
673
|
+
proc.poll()
|
|
674
|
+
if not _group_alive(proc.pid):
|
|
675
|
+
break
|
|
676
|
+
try:
|
|
677
|
+
os.killpg(proc.pid, sig)
|
|
678
|
+
except ProcessLookupError:
|
|
679
|
+
break
|
|
680
|
+
except PermissionError:
|
|
681
|
+
return False
|
|
682
|
+
deadline = time.monotonic() + 5
|
|
683
|
+
while time.monotonic() < deadline:
|
|
684
|
+
proc.poll() # reap the leader so a zombie does not keep the group alive
|
|
685
|
+
if not _group_alive(proc.pid):
|
|
686
|
+
break
|
|
687
|
+
time.sleep(0.05)
|
|
688
|
+
try:
|
|
689
|
+
proc.wait(timeout=5)
|
|
690
|
+
except subprocess.TimeoutExpired:
|
|
691
|
+
return False
|
|
692
|
+
return not _group_alive(proc.pid)
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
def run_agent(cmd, prompt, cwd, env, timeout, stream_path, stderr_path, ctx=None):
|
|
696
|
+
"""Run one agent process in its own process group. With a batch context, the spawn happens
|
|
697
|
+
under the context's lock and is refused once a signal is latched, shutdown has begun or the
|
|
698
|
+
guard has stopped the batch; returns None in that case. The latch is set when the runner's
|
|
699
|
+
main thread handles the signal, at most one wait interval after it arrives; a run that passed
|
|
700
|
+
this check before then is killed with the others and records nothing."""
|
|
701
|
+
started = time.monotonic()
|
|
702
|
+
timed_out = False
|
|
703
|
+
with open(stream_path, "wb") as out, open(stderr_path, "wb") as err:
|
|
704
|
+
with ctx["spawn_lock"] if ctx else threading.Lock():
|
|
705
|
+
if ctx and (ctx["latched"] or ctx["interrupted"].is_set() or ctx["stop"].is_set()):
|
|
706
|
+
return None
|
|
707
|
+
proc = subprocess.Popen(cmd, cwd=str(cwd), env=env, stdin=subprocess.PIPE, stdout=out, stderr=err,
|
|
708
|
+
start_new_session=True)
|
|
709
|
+
if ctx:
|
|
710
|
+
ctx["live"].add(proc)
|
|
711
|
+
try:
|
|
712
|
+
proc.communicate(input=prompt.encode("utf-8"), timeout=timeout)
|
|
713
|
+
except subprocess.TimeoutExpired:
|
|
714
|
+
timed_out = True
|
|
715
|
+
finally:
|
|
716
|
+
cleaned = kill_group(proc)
|
|
717
|
+
if ctx:
|
|
718
|
+
ctx["live"].discard(proc)
|
|
719
|
+
return {"exit_code": proc.returncode, "timed_out": timed_out, "cleanup_confirmed": cleaned,
|
|
720
|
+
"seconds": round(time.monotonic() - started, 1)}
|
|
721
|
+
|
|
722
|
+
|
|
723
|
+
def _utilizations(info):
|
|
724
|
+
values = []
|
|
725
|
+
if isinstance(info, dict):
|
|
726
|
+
if isinstance(info.get("utilization"), (int, float)):
|
|
727
|
+
values.append(info["utilization"])
|
|
728
|
+
windows = info.get("unifiedWindows")
|
|
729
|
+
if isinstance(windows, dict):
|
|
730
|
+
for window in windows.values():
|
|
731
|
+
if isinstance(window, dict) and isinstance(window.get("utilization"), (int, float)):
|
|
732
|
+
values.append(window["utilization"])
|
|
733
|
+
return values
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
def parse_stream(path):
|
|
737
|
+
"""Read a stream-json transcript. `raw` keeps every byte as text, so a check can cover what
|
|
738
|
+
the model saw in tool results and hook output as well as what it wrote; `trailing_activity`
|
|
739
|
+
counts assistant and tool events after the last result, which a finished run never has."""
|
|
740
|
+
parsed = {"init": None, "results": [], "hooks": [], "routing_injected": False, "tool_uses": [],
|
|
741
|
+
"texts": [], "invalid_lines": 0, "max_utilization": None, "trailing_activity": 0}
|
|
742
|
+
raw_text = Path(path).read_bytes().decode("utf-8", "replace")
|
|
743
|
+
parsed["raw"] = raw_text
|
|
744
|
+
with open(path, "rb") as fh:
|
|
745
|
+
for raw in fh:
|
|
746
|
+
line = raw.decode("utf-8", "replace").strip()
|
|
747
|
+
if not line:
|
|
748
|
+
continue
|
|
749
|
+
try:
|
|
750
|
+
event = json.loads(line)
|
|
751
|
+
except (json.JSONDecodeError, RecursionError):
|
|
752
|
+
parsed["invalid_lines"] += 1
|
|
753
|
+
continue
|
|
754
|
+
if not isinstance(event, dict):
|
|
755
|
+
parsed["invalid_lines"] += 1
|
|
756
|
+
continue
|
|
757
|
+
kind, sub = event.get("type"), event.get("subtype")
|
|
758
|
+
if kind == "system" and sub == "init" and parsed["init"] is None:
|
|
759
|
+
parsed["init"] = event
|
|
760
|
+
elif kind == "system" and sub == "hook_response":
|
|
761
|
+
parsed["hooks"].append(str(event.get("hook_name")))
|
|
762
|
+
if ROUTING_MARKER in f"{event.get('output', '')}{event.get('stdout', '')}":
|
|
763
|
+
parsed["routing_injected"] = True
|
|
764
|
+
elif kind in ("assistant", "user"):
|
|
765
|
+
parsed["trailing_activity"] += 1
|
|
766
|
+
if kind == "assistant":
|
|
767
|
+
message = event.get("message")
|
|
768
|
+
content = message.get("content") if isinstance(message, dict) else None
|
|
769
|
+
for item in content if isinstance(content, list) else []:
|
|
770
|
+
if not isinstance(item, dict):
|
|
771
|
+
continue
|
|
772
|
+
if item.get("type") == "text" and isinstance(item.get("text"), str):
|
|
773
|
+
parsed["texts"].append(item["text"])
|
|
774
|
+
elif item.get("type") == "tool_use":
|
|
775
|
+
tool_input = item.get("input") if isinstance(item.get("input"), dict) else {}
|
|
776
|
+
parsed["tool_uses"].append({"name": str(item.get("name")), "input": tool_input})
|
|
777
|
+
elif kind == "result":
|
|
778
|
+
parsed["results"].append(event)
|
|
779
|
+
parsed["trailing_activity"] = 0
|
|
780
|
+
elif kind == "rate_limit_event":
|
|
781
|
+
values = _utilizations(event.get("rate_limit_info"))
|
|
782
|
+
if values:
|
|
783
|
+
parsed["max_utilization"] = max(values + [parsed["max_utilization"] or 0])
|
|
784
|
+
return parsed
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
def _strings(value):
|
|
788
|
+
if isinstance(value, str):
|
|
789
|
+
yield value
|
|
790
|
+
elif isinstance(value, dict):
|
|
791
|
+
for item in value.values():
|
|
792
|
+
yield from _strings(item)
|
|
793
|
+
elif isinstance(value, list):
|
|
794
|
+
for item in value:
|
|
795
|
+
yield from _strings(item)
|
|
796
|
+
|
|
797
|
+
|
|
798
|
+
def _under(path, root):
|
|
799
|
+
return path == root or path.startswith(root.rstrip(os.sep) + os.sep)
|
|
800
|
+
|
|
801
|
+
|
|
802
|
+
def outside_paths(tool_uses, allowed, watched, home):
|
|
803
|
+
"""Heuristic: absolute or home-relative paths in tool inputs that fall under a watched
|
|
804
|
+
root (home, this repository, the output root) but outside every allowed root."""
|
|
805
|
+
flagged = []
|
|
806
|
+
for use in tool_uses:
|
|
807
|
+
for text in _strings(use["input"]):
|
|
808
|
+
for token in PATH_TOKEN.findall(text):
|
|
809
|
+
expanded = re.sub(r"^(?:~|\$\{?HOME\}?)", lambda _: home, token)
|
|
810
|
+
if not expanded.startswith("/"):
|
|
811
|
+
continue
|
|
812
|
+
real = os.path.realpath(expanded)
|
|
813
|
+
if any(_under(real, root) for root in allowed):
|
|
814
|
+
continue
|
|
815
|
+
if any(_under(real, root) for root in watched) and real not in flagged:
|
|
816
|
+
flagged.append(real)
|
|
817
|
+
return flagged
|
|
818
|
+
|
|
819
|
+
|
|
820
|
+
def assess(parsed, run, arm, model, arm_dir, plugin_name, arm_dirs, token):
|
|
821
|
+
"""Structural isolation evidence for one run. Returns the reasons the sample is invalid.
|
|
822
|
+
A run may hold several results: a plugin Stop hook can send the agent back for more
|
|
823
|
+
turns, which is the treatment's own behavior, so the last result decides how it ended."""
|
|
824
|
+
reasons = []
|
|
825
|
+
results = parsed["results"]
|
|
826
|
+
if run["timed_out"]:
|
|
827
|
+
reasons.append("timeout")
|
|
828
|
+
elif not results:
|
|
829
|
+
reasons.append("no_result")
|
|
830
|
+
elif results[-1].get("subtype") != "success" or results[-1].get("is_error"):
|
|
831
|
+
reasons.append(f"error_result:{results[-1].get('subtype')}")
|
|
832
|
+
elif parsed["trailing_activity"]:
|
|
833
|
+
reasons.append("activity_after_last_result")
|
|
834
|
+
if parsed["invalid_lines"]: # the CLI writes only JSON lines; anything else is a truncated or broken stream
|
|
835
|
+
reasons.append("malformed_stream")
|
|
836
|
+
init = parsed["init"]
|
|
837
|
+
isolation = {"hooks": sorted(set(parsed["hooks"]))}
|
|
838
|
+
if not isinstance(init, dict):
|
|
839
|
+
reasons.append("no_init_event")
|
|
840
|
+
else:
|
|
841
|
+
isolation["model"] = init.get("model")
|
|
842
|
+
if init.get("model") != model:
|
|
843
|
+
reasons.append("model_mismatch")
|
|
844
|
+
plugins, servers = init.get("plugins"), init.get("mcp_servers")
|
|
845
|
+
if not isinstance(plugins, list) or not isinstance(servers, list):
|
|
846
|
+
reasons.append("init_shape_unverifiable")
|
|
847
|
+
entries = [p for p in plugins if isinstance(p, dict)] if isinstance(plugins, list) else []
|
|
848
|
+
isolation["plugins"] = sorted(f"{p.get('name')}@{'builtin' if p.get('path') == 'builtin' else 'dir'}"
|
|
849
|
+
for p in entries)
|
|
850
|
+
own = [p for p in entries if p.get("name") == plugin_name]
|
|
851
|
+
if arm == "off":
|
|
852
|
+
from_arm_dirs = [p for p in entries if p.get("path") != "builtin" and isinstance(p.get("path"), str)
|
|
853
|
+
and any(_under(os.path.realpath(p["path"]), d) for d in arm_dirs)]
|
|
854
|
+
if own or from_arm_dirs:
|
|
855
|
+
reasons.append("plugin_loaded_in_off_arm")
|
|
856
|
+
elif len(own) != 1 or os.path.realpath(str(own[0].get("path"))) != os.path.realpath(str(arm_dir)):
|
|
857
|
+
reasons.append("plugin_identity_mismatch")
|
|
858
|
+
if any(p.get("path") != "builtin" and p.get("name") != plugin_name for p in entries):
|
|
859
|
+
reasons.append("foreign_plugins")
|
|
860
|
+
if isinstance(servers, list) and servers:
|
|
861
|
+
reasons.append("mcp_servers_present")
|
|
862
|
+
if (arm != "off") != parsed["routing_injected"]:
|
|
863
|
+
reasons.append("routing_injection_mismatch")
|
|
864
|
+
if token in parsed["raw"]: # loaded as instructions, read through a tool, or echoed: the run saw it
|
|
865
|
+
reasons.append("instruction_file_canary_seen")
|
|
866
|
+
if not run["cleanup_confirmed"]:
|
|
867
|
+
reasons.append("process_cleanup_unconfirmed")
|
|
868
|
+
return reasons, isolation
|
|
869
|
+
|
|
870
|
+
|
|
871
|
+
# --------------------------------------------------------------------------- statistics and report
|
|
872
|
+
|
|
873
|
+
def fisher_two_sided(a, b, c, d):
|
|
874
|
+
"""Two-sided Fisher exact p for the table [[a, b], [c, d]] (pass/fail in two arms)."""
|
|
875
|
+
n1, n2, k = a + b, c + d, a + c
|
|
876
|
+
n = n1 + n2
|
|
877
|
+
if n == 0:
|
|
878
|
+
return 1.0
|
|
879
|
+
denom = math.comb(n, k)
|
|
880
|
+
|
|
881
|
+
def prob(x):
|
|
882
|
+
return math.comb(n1, x) * math.comb(n2, k - x) / denom
|
|
883
|
+
observed = prob(a)
|
|
884
|
+
total = sum(prob(x) for x in range(max(0, k - n2), min(k, n1) + 1) if prob(x) <= observed * (1 + 1e-9))
|
|
885
|
+
return min(1.0, total)
|
|
886
|
+
|
|
887
|
+
|
|
888
|
+
def compare(k1, n1, k2, n2):
|
|
889
|
+
"""Pre-registered reading of one pairwise comparison."""
|
|
890
|
+
if n1 < 3 or n2 < 3:
|
|
891
|
+
return "insufficient", None
|
|
892
|
+
p = fisher_two_sided(k1, n1 - k1, k2, n2 - k2)
|
|
893
|
+
if p < 0.05:
|
|
894
|
+
return "separated", p
|
|
895
|
+
if abs(k1 / n1 - k2 / n2) >= 0.4:
|
|
896
|
+
return "direction", p
|
|
897
|
+
return "no observed difference", p
|
|
898
|
+
|
|
899
|
+
|
|
900
|
+
def tally(records, task_id, arm, check_id):
|
|
901
|
+
rows = [r for r in records if r["task"] == task_id and r["arm"] == arm and r["valid"]]
|
|
902
|
+
passed = sum(1 for r in rows if r["checks"][check_id]["result"] == "pass")
|
|
903
|
+
failed = sum(1 for r in rows if r["checks"][check_id]["result"] == "fail")
|
|
904
|
+
return passed, passed + failed, len(rows) - passed - failed
|
|
905
|
+
|
|
906
|
+
|
|
907
|
+
def _median(values):
|
|
908
|
+
values = [v for v in values if isinstance(v, (int, float))]
|
|
909
|
+
return round(statistics.median(values), 2) if values else None
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
def build_report(plan, records, calibration, integrity):
|
|
913
|
+
arms = [a for a in ARMS if a in plan["arms"]]
|
|
914
|
+
lines = ["# Paired behavior eval report", "",
|
|
915
|
+
"Advisory evidence for the tasks and conditions below only; the tool header lists what these "
|
|
916
|
+
"numbers are not. Counts exclude invalid samples; `Ns` marks valid samples with out-of-world "
|
|
917
|
+
"path suspects.", "", "## Batch", "",
|
|
918
|
+
f"- plan `{plan['plan_hash'][:16]}`, tool version {plan['tool_version']}",
|
|
919
|
+
f"- model `{plan['model']}`, effort `{plan['effort']}`, claude `{plan['claude_version']}`",
|
|
920
|
+
f"- per-run cap ${plan['max_budget_usd']}, denied tools `{plan['disallowed_tools']}`"]
|
|
921
|
+
for arm in arms:
|
|
922
|
+
info = plan["arms"][arm]
|
|
923
|
+
lines.append(f"- `{arm}` ({TREATMENT[arm]}): " + ("no plugin" if arm == "off" else
|
|
924
|
+
f"`{info['ref']}` = {info['commit'][:12]}, export digest {info['export_digest'][:12]}"))
|
|
925
|
+
graders = sorted({r.get("graded_by") or plan["tool_sha256"] for r in records} - {plan["tool_sha256"]})
|
|
926
|
+
if graders:
|
|
927
|
+
lines.append(f"- records regraded by tool `{', '.join(g[:12] for g in graders)}`; the batch ran with "
|
|
928
|
+
f"`{plan['tool_sha256'][:12]}`")
|
|
929
|
+
cal = "not run" if calibration is None else ("fired" if calibration.get("fired") else "DID NOT FIRE")
|
|
930
|
+
lines.append(f"- instruction-file canary calibration: {cal}")
|
|
931
|
+
if integrity is not None:
|
|
932
|
+
detail = "; ".join(f"{arm}: {', '.join(paths[:5])}" for arm, paths in (integrity.get("changed") or {}).items())
|
|
933
|
+
state = {True: "yes", False: "NO"}.get(integrity.get("unchanged"), "UNKNOWN")
|
|
934
|
+
detail = detail or integrity.get("error", "")
|
|
935
|
+
lines.append(f"- plugin exports unchanged after the batch: {state}" + (f" ({detail})" if detail else ""))
|
|
936
|
+
costs = [r["cost_usd"] for r in records if isinstance(r.get("cost_usd"), (int, float))]
|
|
937
|
+
lines += [f"- samples recorded {len(records)}, valid {sum(r['valid'] for r in records)}, "
|
|
938
|
+
f"with suspects {sum(bool(r['suspect_paths']) for r in records)}, spend ${round(sum(costs), 2)}", "",
|
|
939
|
+
"## Isolation", "", "| arm | recorded | valid | invalid reasons | suspects | continued after a Stop hook |",
|
|
940
|
+
"| --- | --- | --- | --- | --- | --- |"]
|
|
941
|
+
for arm in arms:
|
|
942
|
+
rows = [r for r in records if r["arm"] == arm]
|
|
943
|
+
reasons = {}
|
|
944
|
+
for r in rows:
|
|
945
|
+
for reason in r["invalid_reasons"]:
|
|
946
|
+
reasons[reason] = reasons.get(reason, 0) + 1
|
|
947
|
+
text = ", ".join(f"{k} {v}" for k, v in sorted(reasons.items())) or "none"
|
|
948
|
+
lines.append(f"| {arm} | {len(rows)} | {sum(r['valid'] for r in rows)} | {text} | "
|
|
949
|
+
f"{sum(bool(r['suspect_paths']) for r in rows)} | {sum(bool(r.get('continuations')) for r in rows)} |")
|
|
950
|
+
lines.append("")
|
|
951
|
+
pairs = [(a, b) for a, b in (("candidate", "base"), ("base", "off"), ("candidate", "off"))
|
|
952
|
+
if a in arms and b in arms]
|
|
953
|
+
comparisons = 0
|
|
954
|
+
for task in plan["tasks"]:
|
|
955
|
+
tid = task["id"]
|
|
956
|
+
lines += [f"## {tid}", "", task["measures"], "",
|
|
957
|
+
"| check | role | " + " | ".join(arms) + " |", "| --- | --- | " + " | ".join("---" for _ in arms) + " |"]
|
|
958
|
+
for check in task["checks"]:
|
|
959
|
+
cells = []
|
|
960
|
+
for arm in arms:
|
|
961
|
+
k, n, na = tally(records, tid, arm, check["id"])
|
|
962
|
+
suspects = sum(1 for r in records if r["task"] == tid and r["arm"] == arm and r["valid"]
|
|
963
|
+
and r["suspect_paths"])
|
|
964
|
+
cell = f"{k}/{n}" + (f" (+{na} n/a)" if na else "") + (f" {suspects}s" if suspects else "")
|
|
965
|
+
cells.append(cell if n or na else "—")
|
|
966
|
+
lines.append(f"| {check['id']} | {check['role']} | " + " | ".join(cells) + " |")
|
|
967
|
+
if pairs:
|
|
968
|
+
lines += ["", "| check | " + " | ".join(f"{a} vs {b}" for a, b in pairs) + " |",
|
|
969
|
+
"| --- | " + " | ".join("---" for _ in pairs) + " |"]
|
|
970
|
+
for check in task["checks"]:
|
|
971
|
+
cells = []
|
|
972
|
+
for a, b in pairs:
|
|
973
|
+
k1, n1, _ = tally(records, tid, a, check["id"])
|
|
974
|
+
k2, n2, _ = tally(records, tid, b, check["id"])
|
|
975
|
+
verdict, p = compare(k1, n1, k2, n2)
|
|
976
|
+
comparisons += p is not None
|
|
977
|
+
cells.append(f"{k1}/{n1} vs {k2}/{n2}: {verdict}" + (f" (p={p:.3f})" if p is not None else ""))
|
|
978
|
+
lines.append(f"| {check['id']} | " + " | ".join(cells) + " |")
|
|
979
|
+
lines += ["", "| arm | median cost $ | median turns | median seconds |", "| --- | --- | --- | --- |"]
|
|
980
|
+
for arm in arms:
|
|
981
|
+
rows = [r for r in records if r["task"] == tid and r["arm"] == arm and r["valid"]]
|
|
982
|
+
lines.append(f"| {arm} | {_median([r.get('cost_usd') for r in rows])} | "
|
|
983
|
+
f"{_median([r.get('turns') for r in rows])} | {_median([r.get('seconds') for r in rows])} |")
|
|
984
|
+
lines.append("")
|
|
985
|
+
lines += ["## Reading", "",
|
|
986
|
+
f"{comparisons} pairwise comparisons, each with an uncorrected two-sided Fisher exact p, so a few "
|
|
987
|
+
"`separated` labels can be chance. `separated` = p < 0.05; `direction` = pass rates differ by 0.4 "
|
|
988
|
+
"or more without p < 0.05; `insufficient` = fewer than 3 valid samples in an arm.", ""]
|
|
989
|
+
return "\n".join(lines)
|
|
990
|
+
|
|
991
|
+
|
|
992
|
+
# --------------------------------------------------------------------------- batch
|
|
993
|
+
|
|
994
|
+
def checkout_roots(repo):
|
|
995
|
+
"""Every checkout of the repository: the given one, the main one and each registered worktree."""
|
|
996
|
+
out = subprocess.run(["git", "-C", str(repo), "worktree", "list", "--porcelain", "-z"], env=source_env(),
|
|
997
|
+
capture_output=True, text=True, timeout=60)
|
|
998
|
+
if out.returncode != 0:
|
|
999
|
+
raise TaskError(f"{repo} is not a git repository")
|
|
1000
|
+
roots = {os.path.realpath(repo)}
|
|
1001
|
+
roots.update(os.path.realpath(field[len("worktree "):]) for field in out.stdout.split("\0")
|
|
1002
|
+
if field.startswith("worktree ")) # -z keeps paths unquoted, newlines included
|
|
1003
|
+
return sorted(roots)
|
|
1004
|
+
|
|
1005
|
+
|
|
1006
|
+
def claude_version(claude):
|
|
1007
|
+
try:
|
|
1008
|
+
out = subprocess.run([claude, "--version"], env=inherited_env(), capture_output=True, text=True, timeout=60)
|
|
1009
|
+
except (OSError, subprocess.TimeoutExpired) as exc:
|
|
1010
|
+
raise TaskError(f"cannot run {claude} --version: {exc}") from None
|
|
1011
|
+
if out.returncode != 0 or not out.stdout.strip():
|
|
1012
|
+
raise TaskError(f"{claude} --version failed")
|
|
1013
|
+
return out.stdout.strip().splitlines()[0]
|
|
1014
|
+
|
|
1015
|
+
|
|
1016
|
+
def canonical(obj):
|
|
1017
|
+
return json.dumps(obj, sort_keys=True, ensure_ascii=False, separators=(",", ":"))
|
|
1018
|
+
|
|
1019
|
+
|
|
1020
|
+
def write_atomic(path, text):
|
|
1021
|
+
"""Write beside the destination, then rename over it, so a reader never sees a torn file."""
|
|
1022
|
+
tmp = path.with_name(f".{path.name}.tmp")
|
|
1023
|
+
tmp.write_text(text, encoding="utf-8")
|
|
1024
|
+
os.replace(tmp, path)
|
|
1025
|
+
|
|
1026
|
+
|
|
1027
|
+
def schedule(tasks, arms, samples_override):
|
|
1028
|
+
"""Interleave arms within each sample index so drift over the batch hits every arm alike."""
|
|
1029
|
+
counts = {t["id"]: samples_override or t["samples"] for t in tasks}
|
|
1030
|
+
order = []
|
|
1031
|
+
for i in range(1, max(counts.values()) + 1):
|
|
1032
|
+
for t_index, task in enumerate(tasks):
|
|
1033
|
+
if i > counts[task["id"]]:
|
|
1034
|
+
continue
|
|
1035
|
+
shift = (i + t_index) % len(arms)
|
|
1036
|
+
for arm in arms[shift:] + arms[:shift]:
|
|
1037
|
+
order.append((task, arm, i))
|
|
1038
|
+
return order
|
|
1039
|
+
|
|
1040
|
+
|
|
1041
|
+
def agent_command(claude, model, effort, budget, plugin_dir, setting_sources=""):
|
|
1042
|
+
cmd = [claude, "-p", "--model", model]
|
|
1043
|
+
if effort:
|
|
1044
|
+
cmd += ["--effort", effort]
|
|
1045
|
+
if plugin_dir:
|
|
1046
|
+
cmd += ["--plugin-dir", str(plugin_dir)]
|
|
1047
|
+
# Session persistence stays on: a plugin's hooks may read the session transcript, and without
|
|
1048
|
+
# it they run degraded. collect_transcripts moves the run's transcript out of the home directory.
|
|
1049
|
+
cmd += ["--setting-sources", setting_sources, "--strict-mcp-config", "--permission-mode", "bypassPermissions",
|
|
1050
|
+
"--output-format", "stream-json", "--verbose",
|
|
1051
|
+
"--max-budget-usd", str(budget), "--disallowedTools", DISALLOWED_TOOLS]
|
|
1052
|
+
return cmd
|
|
1053
|
+
|
|
1054
|
+
|
|
1055
|
+
def collect_transcripts(stream_path, dest):
|
|
1056
|
+
"""Move the session transcripts Claude Code wrote for this run out of ~/.claude/projects into
|
|
1057
|
+
dest. Only entries named by a session id from this run's own init events are moved, and a
|
|
1058
|
+
project directory is removed only when that leaves it empty. Returns the moved names in two
|
|
1059
|
+
lists: transcripts, which were non-empty regular files when found, and everything else, such
|
|
1060
|
+
as session directories or links, which never count as a persisted transcript."""
|
|
1061
|
+
sessions = set()
|
|
1062
|
+
for line in Path(stream_path).read_text(encoding="utf-8", errors="replace").splitlines():
|
|
1063
|
+
try:
|
|
1064
|
+
event = json.loads(line)
|
|
1065
|
+
except ValueError:
|
|
1066
|
+
continue
|
|
1067
|
+
if isinstance(event, dict) and event.get("type") == "system" and event.get("subtype") == "init":
|
|
1068
|
+
session = event.get("session_id")
|
|
1069
|
+
if isinstance(session, str) and SESSION_ID.fullmatch(session):
|
|
1070
|
+
sessions.add(session)
|
|
1071
|
+
projects = Path.home() / ".claude" / "projects"
|
|
1072
|
+
transcripts, others = [], []
|
|
1073
|
+
for session in sorted(sessions):
|
|
1074
|
+
for source in sorted(projects.glob(f"*/{session}.jsonl")) + sorted(projects.glob(f"*/{session}")):
|
|
1075
|
+
real = (source.name == f"{session}.jsonl" and source.is_file() and not source.is_symlink()
|
|
1076
|
+
and source.stat().st_size > 0)
|
|
1077
|
+
target = dest / f"transcript-{source.name}"
|
|
1078
|
+
shutil.move(str(source), str(target))
|
|
1079
|
+
(transcripts if real else others).append(target.name)
|
|
1080
|
+
try:
|
|
1081
|
+
source.parent.rmdir()
|
|
1082
|
+
except OSError:
|
|
1083
|
+
pass # the project directory still holds other sessions
|
|
1084
|
+
return transcripts, others
|
|
1085
|
+
|
|
1086
|
+
|
|
1087
|
+
def run_sample(ctx, task, arm, index):
|
|
1088
|
+
"""One run in a fresh world, with a private copy of the arm's frozen export, so a run that
|
|
1089
|
+
edits the plugin changes only its own copy and is recorded as invalid."""
|
|
1090
|
+
sample_dir = ctx["out"] / "runs" / task["id"] / arm / str(index)
|
|
1091
|
+
world, gitconfig, snapshot, token = build_world(task, sample_dir)
|
|
1092
|
+
write_atomic(sample_dir / "snapshot.json", json.dumps(snapshot, indent=1))
|
|
1093
|
+
plugin_dir = None
|
|
1094
|
+
if arm in ctx["arm_dirs"]:
|
|
1095
|
+
plugin_dir = sample_dir / "plugin"
|
|
1096
|
+
if plugin_dir.exists():
|
|
1097
|
+
shutil.rmtree(plugin_dir)
|
|
1098
|
+
shutil.copytree(ctx["arm_dirs"][arm], plugin_dir, symlinks=True)
|
|
1099
|
+
plan = ctx["plan"]
|
|
1100
|
+
cmd = agent_command(ctx["claude"], plan["model"], plan["effort_flag"], plan["max_budget_usd"], plugin_dir)
|
|
1101
|
+
run = run_agent(cmd, task["prompt"], world / task["cwd"], clean_env(gitconfig, sample_dir, agent=True),
|
|
1102
|
+
task["timeout_seconds"], sample_dir / "stream.jsonl", sample_dir / "stderr.txt", ctx)
|
|
1103
|
+
if run is None:
|
|
1104
|
+
return None # never started: resume runs it
|
|
1105
|
+
run["transcripts"], run["session_files"] = collect_transcripts(sample_dir / "stream.jsonl", sample_dir)
|
|
1106
|
+
if ctx["interrupted"].is_set():
|
|
1107
|
+
return None # stopped by an interrupt: not an outcome; resume reruns it
|
|
1108
|
+
run["export_changes"] = []
|
|
1109
|
+
if plugin_dir is not None:
|
|
1110
|
+
run["export_changes"] = manifest_changes(ctx["manifests"][arm], tree_manifest(plugin_dir))
|
|
1111
|
+
if not run["export_changes"]:
|
|
1112
|
+
shutil.rmtree(plugin_dir) # identical to the frozen export; kept only when the run changed it
|
|
1113
|
+
return make_record(ctx, task, arm, index, sample_dir, snapshot, token, run, plugin_dir)
|
|
1114
|
+
|
|
1115
|
+
|
|
1116
|
+
def make_record(ctx, task, arm, index, sample_dir, snapshot, token, run, plugin_dir):
|
|
1117
|
+
"""Assess and grade one finished run from its saved stream and world, and write record.json.
|
|
1118
|
+
The record keeps the canary token and the snapshot so later regrading never reads files the
|
|
1119
|
+
tested agent could have rewritten."""
|
|
1120
|
+
world = sample_dir / "world"
|
|
1121
|
+
plan = ctx["plan"]
|
|
1122
|
+
parsed = parse_stream(sample_dir / "stream.jsonl")
|
|
1123
|
+
reasons, isolation = assess(parsed, run, arm, plan["model"], plugin_dir, ctx["plugin_name"],
|
|
1124
|
+
ctx["arm_dirs_real"], token)
|
|
1125
|
+
if run.get("export_changes"):
|
|
1126
|
+
reasons.append("plugin_export_changed")
|
|
1127
|
+
if not run.get("transcripts"):
|
|
1128
|
+
reasons.append("transcript_missing") # hooks that read the session transcript did not run as in normal use
|
|
1129
|
+
allowed = [os.path.realpath(world)] + ([os.path.realpath(plugin_dir)] if plugin_dir else [])
|
|
1130
|
+
suspects = outside_paths(parsed["tool_uses"], allowed, ctx["watched"], ctx["home"])
|
|
1131
|
+
commands = [str(u["input"].get("command", "")) for u in parsed["tool_uses"] if u["name"] == "Bash"]
|
|
1132
|
+
results = parsed["results"]
|
|
1133
|
+
last = results[-1] if results else {}
|
|
1134
|
+
record = {
|
|
1135
|
+
"task": task["id"], "arm": arm, "treatment": TREATMENT[arm], "sample": index,
|
|
1136
|
+
"plan_hash": plan["plan_hash"], "graded_by": ctx["tool_sha256"], "valid": not reasons,
|
|
1137
|
+
"invalid_reasons": reasons, "suspect_paths": suspects[:10], "isolation": isolation,
|
|
1138
|
+
"checks": grade(task, world, snapshot, commands),
|
|
1139
|
+
"cost_usd": last.get("total_cost_usd"), # cumulative across continuations
|
|
1140
|
+
"turns": sum(r["num_turns"] for r in results if isinstance(r.get("num_turns"), int)) or None,
|
|
1141
|
+
"continuations": max(len(results) - 1, 0),
|
|
1142
|
+
"seconds": run["seconds"], "exit_code": run["exit_code"],
|
|
1143
|
+
"run": {"timed_out": run["timed_out"], "cleanup_confirmed": run["cleanup_confirmed"],
|
|
1144
|
+
"export_changes": run.get("export_changes") or [], "transcripts": run.get("transcripts") or [],
|
|
1145
|
+
"session_files": run.get("session_files") or [], "evidence_version": RUN_EVIDENCE_VERSION},
|
|
1146
|
+
"canary": token, "snapshot": snapshot, "plugin_dir": str(plugin_dir) if plugin_dir else None,
|
|
1147
|
+
"max_utilization": parsed["max_utilization"],
|
|
1148
|
+
"skills_invoked": [str(u["input"].get("skill")) for u in parsed["tool_uses"] if u["name"] == "Skill"],
|
|
1149
|
+
"bash_commands": len(commands), "result_excerpt": str(last.get("result", ""))[:300],
|
|
1150
|
+
}
|
|
1151
|
+
write_atomic(sample_dir / "record.json", json.dumps(record, indent=1, ensure_ascii=False))
|
|
1152
|
+
return record
|
|
1153
|
+
|
|
1154
|
+
|
|
1155
|
+
def regrade(ctx, out, tasks):
|
|
1156
|
+
"""Rebuild every record of the plan from its saved stream and world with the current
|
|
1157
|
+
assessment and graders; no model runs. The batch's tasks must be byte-identical, and the
|
|
1158
|
+
canary token, snapshot and plugin path come from the record, not from the world."""
|
|
1159
|
+
by_id = {t["id"]: t for t in tasks}
|
|
1160
|
+
for entry in ctx["plan"]["tasks"]:
|
|
1161
|
+
if entry["id"] not in by_id or by_id[entry["id"]]["_sha256"] != entry["sha256"]:
|
|
1162
|
+
raise TaskError(f"task {entry['id']} differs from the one the batch ran; regrading would apply other checks")
|
|
1163
|
+
records = load_records(out, ctx["plan"])
|
|
1164
|
+
for old in records:
|
|
1165
|
+
sample_dir = out / "runs" / old["task"] / old["arm"] / str(old["sample"])
|
|
1166
|
+
if (not (isinstance(old.get("canary"), str) and isinstance(old.get("snapshot"), dict) and "plugin_dir" in old)
|
|
1167
|
+
or old.get("legacy_inputs")):
|
|
1168
|
+
raise TaskError(f"{sample_dir.relative_to(out)}: the record holds no runner-recorded canary token, snapshot "
|
|
1169
|
+
"and plugin path, and the world's copies could have been rewritten by the run")
|
|
1170
|
+
if (old.get("run") or {}).get("evidence_version") != RUN_EVIDENCE_VERSION:
|
|
1171
|
+
raise TaskError(f"{sample_dir.relative_to(out)}: the record's run evidence was written under other rules "
|
|
1172
|
+
"(such as an earlier transcript check), so regrading would reuse evidence it cannot trust")
|
|
1173
|
+
token, snapshot = old["canary"], old["snapshot"]
|
|
1174
|
+
plugin_dir = Path(old["plugin_dir"]) if old["plugin_dir"] else None
|
|
1175
|
+
run = dict(old.get("run") or {"timed_out": "timeout" in old["invalid_reasons"],
|
|
1176
|
+
"cleanup_confirmed": "process_cleanup_unconfirmed" not in old["invalid_reasons"]},
|
|
1177
|
+
seconds=old["seconds"], exit_code=old["exit_code"])
|
|
1178
|
+
make_record(ctx, by_id[old["task"]], old["arm"], old["sample"], sample_dir, snapshot, token, run, plugin_dir)
|
|
1179
|
+
return len(records)
|
|
1180
|
+
|
|
1181
|
+
|
|
1182
|
+
def calibrate_canary(ctx):
|
|
1183
|
+
"""Prove the canary can fire: allow project settings and instruction files once, on a tiny
|
|
1184
|
+
prompt, and expect the token in a finished, cleaned-up reply."""
|
|
1185
|
+
probe = {"id": "canary-calibration", "setup": ["git init -q -b main app"], "cwd": "app", "checks": []}
|
|
1186
|
+
sample_dir = ctx["out"] / "calibration"
|
|
1187
|
+
world, gitconfig, _, token = build_world(probe, sample_dir)
|
|
1188
|
+
env = clean_env(gitconfig, sample_dir, agent=True)
|
|
1189
|
+
env.pop("CLAUDE_CODE_DISABLE_CLAUDE_MDS")
|
|
1190
|
+
cmd = agent_command(ctx["claude"], ctx["plan"]["model"], None, 1, None, setting_sources="project")
|
|
1191
|
+
run = run_agent(cmd, "Reply with the single word OK.", world / "app", env, 300,
|
|
1192
|
+
sample_dir / "stream.jsonl", sample_dir / "stderr.txt", ctx)
|
|
1193
|
+
if run is None:
|
|
1194
|
+
return None
|
|
1195
|
+
collect_transcripts(sample_dir / "stream.jsonl", sample_dir)
|
|
1196
|
+
if ctx["interrupted"].is_set():
|
|
1197
|
+
return None
|
|
1198
|
+
parsed = parse_stream(sample_dir / "stream.jsonl")
|
|
1199
|
+
last = parsed["results"][-1] if parsed["results"] else {}
|
|
1200
|
+
finished = (not run["timed_out"] and run["cleanup_confirmed"] and last.get("subtype") == "success"
|
|
1201
|
+
and not last.get("is_error") and not parsed["trailing_activity"])
|
|
1202
|
+
texts = parsed["texts"] + [str(r.get("result", "")) for r in parsed["results"]]
|
|
1203
|
+
result = {"fired": finished and any(token in t for t in texts), "finished": finished,
|
|
1204
|
+
"timed_out": run["timed_out"], "max_utilization": parsed["max_utilization"]}
|
|
1205
|
+
write_atomic(ctx["out"] / "canary-calibration.json", json.dumps(result, indent=1))
|
|
1206
|
+
return result
|
|
1207
|
+
|
|
1208
|
+
|
|
1209
|
+
def read_json(path):
|
|
1210
|
+
return json.loads(path.read_text(encoding="utf-8")) if path.is_file() else None
|
|
1211
|
+
|
|
1212
|
+
|
|
1213
|
+
def load_records(out, plan):
|
|
1214
|
+
"""The plan's records only: each must sit at the path its task, arm and sample name, inside
|
|
1215
|
+
the frozen inventory and under the frozen plan hash, so a stray or copied file cannot count."""
|
|
1216
|
+
inventory = {(t["id"], arm, i) for t in plan["tasks"] for arm in plan["arms"] for i in range(1, t["samples"] + 1)}
|
|
1217
|
+
records = []
|
|
1218
|
+
for path in sorted((out / "runs").glob("*/*/*/record.json")):
|
|
1219
|
+
rel = path.relative_to(out)
|
|
1220
|
+
try:
|
|
1221
|
+
record = json.loads(path.read_text(encoding="utf-8"))
|
|
1222
|
+
except json.JSONDecodeError:
|
|
1223
|
+
raise TaskError(f"{rel}: not valid JSON") from None
|
|
1224
|
+
key = (record.get("task"), record.get("arm"), record.get("sample"))
|
|
1225
|
+
if (record.get("plan_hash") != plan["plan_hash"] or key not in inventory
|
|
1226
|
+
or rel.parts[1:4] != tuple(str(part) for part in key)):
|
|
1227
|
+
raise TaskError(f"{rel}: record does not belong to this plan at this path")
|
|
1228
|
+
records.append(record)
|
|
1229
|
+
return records
|
|
1230
|
+
|
|
1231
|
+
|
|
1232
|
+
def write_report(out, plan, records):
|
|
1233
|
+
calibration = read_json(out / "canary-calibration.json")
|
|
1234
|
+
integrity = read_json(out / "integrity.json")
|
|
1235
|
+
write_atomic(out / "results.json", json.dumps(records, indent=1, ensure_ascii=False))
|
|
1236
|
+
write_atomic(out / "report.md", build_report(plan, records, calibration, integrity))
|
|
1237
|
+
|
|
1238
|
+
|
|
1239
|
+
def lock_output(out):
|
|
1240
|
+
"""Hold an exclusive lock on the output root for this process's lifetime, so two invocations
|
|
1241
|
+
never build worlds over each other."""
|
|
1242
|
+
handle = open(out / ".lock", "a+")
|
|
1243
|
+
try:
|
|
1244
|
+
fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
1245
|
+
except BlockingIOError:
|
|
1246
|
+
handle.close()
|
|
1247
|
+
raise TaskError("another invocation is using --out; wait for it or choose another --out") from None
|
|
1248
|
+
return handle
|
|
1249
|
+
|
|
1250
|
+
|
|
1251
|
+
def record_integrity(out, arm_dirs, manifests):
|
|
1252
|
+
"""Name every path that differs from each frozen export. It runs whenever a report is written,
|
|
1253
|
+
so no report carries an older integrity record. An export or frozen manifest that cannot be
|
|
1254
|
+
read is recorded as unknown with the reason, so the batch's own outcome or failure is never
|
|
1255
|
+
replaced by this check."""
|
|
1256
|
+
if not arm_dirs:
|
|
1257
|
+
return
|
|
1258
|
+
try:
|
|
1259
|
+
changed = {}
|
|
1260
|
+
for arm, root in arm_dirs.items():
|
|
1261
|
+
if manifests.get(arm) is None:
|
|
1262
|
+
raise FileNotFoundError(f"no manifest matching the plan for the {arm} export")
|
|
1263
|
+
if paths := manifest_changes(manifests[arm], tree_manifest(root)):
|
|
1264
|
+
changed[arm] = paths
|
|
1265
|
+
payload = {"unchanged": not changed, "changed": changed}
|
|
1266
|
+
except OSError as exc:
|
|
1267
|
+
payload = {"unchanged": None, "changed": {}, "error": f"{type(exc).__name__}: {exc}"[:300]}
|
|
1268
|
+
write_atomic(out / "integrity.json", json.dumps(payload, indent=1))
|
|
1269
|
+
|
|
1270
|
+
|
|
1271
|
+
def frozen_manifest(out, plan, arm):
|
|
1272
|
+
"""The manifest saved when the arm's export was built, or None unless it reads back and hashes
|
|
1273
|
+
to the export digest the plan froze."""
|
|
1274
|
+
try:
|
|
1275
|
+
manifest = read_json(out / "arms" / f"{arm}.manifest.json")
|
|
1276
|
+
if manifest and tree_digest(None, manifest) == plan["arms"][arm]["export_digest"]:
|
|
1277
|
+
return manifest
|
|
1278
|
+
except (OSError, ValueError, TypeError): # unreadable, not JSON, or not a manifest's shape
|
|
1279
|
+
pass
|
|
1280
|
+
return None
|
|
1281
|
+
|
|
1282
|
+
|
|
1283
|
+
def wait_latched(futures, latched, poll=0.2):
|
|
1284
|
+
"""Wait for every future and return their results. Signal handlers only latch during a batch,
|
|
1285
|
+
so this is where an interrupt takes effect: between waits, never inside other code."""
|
|
1286
|
+
pending = set(futures)
|
|
1287
|
+
while pending:
|
|
1288
|
+
done, pending = cf.wait(pending, timeout=poll, return_when=cf.FIRST_COMPLETED)
|
|
1289
|
+
if latched: # checked before any result, so a failure the signal caused reads as the interrupt
|
|
1290
|
+
raise KeyboardInterrupt
|
|
1291
|
+
for future in done:
|
|
1292
|
+
future.result()
|
|
1293
|
+
return [future.result() for future in futures]
|
|
1294
|
+
|
|
1295
|
+
|
|
1296
|
+
def tool_sha256():
|
|
1297
|
+
return hashlib.sha256(Path(__file__).read_bytes()).hexdigest()
|
|
1298
|
+
|
|
1299
|
+
|
|
1300
|
+
def make_context(out, plan, arm_dirs, roots, claude=None, manifests=None):
|
|
1301
|
+
home = os.path.realpath(os.path.expanduser("~"))
|
|
1302
|
+
return {"out": out, "claude": claude, "plan": plan, "arm_dirs": arm_dirs, "plugin_name": plan["plugin_name"],
|
|
1303
|
+
"arm_dirs_real": [os.path.realpath(d) for d in arm_dirs.values()], "home": home,
|
|
1304
|
+
"watched": [home, *roots, os.path.realpath(out)], "manifests": manifests or {},
|
|
1305
|
+
"tool_sha256": tool_sha256(), "live": set(), "interrupted": threading.Event(),
|
|
1306
|
+
"stop": threading.Event(), "spawn_lock": threading.Lock(), "latched": []}
|
|
1307
|
+
|
|
1308
|
+
|
|
1309
|
+
def run_batch(args, out, arms, only):
|
|
1310
|
+
repo = args.repo.resolve()
|
|
1311
|
+
roots = checkout_roots(repo)
|
|
1312
|
+
if any(_under(str(out), root) for root in roots):
|
|
1313
|
+
raise TaskError("--out must be outside every checkout of this repository")
|
|
1314
|
+
tasks = load_tasks(args.tasks_dir, only)
|
|
1315
|
+
commits = {arm: resolve_commit(repo, ref) for arm, ref in (("base", args.base), ("candidate", args.candidate))
|
|
1316
|
+
if arm in arms}
|
|
1317
|
+
planned = schedule(tasks, arms, args.samples)
|
|
1318
|
+
if args.dry_run:
|
|
1319
|
+
for arm in arms:
|
|
1320
|
+
print(f"arm {arm}: " + (commits[arm][:12] if arm in commits else "no plugin"))
|
|
1321
|
+
for task in tasks:
|
|
1322
|
+
print(f"task {task['id']}: {args.samples or task['samples']} samples per arm")
|
|
1323
|
+
print(f"runs {len(planned)} (+1 canary calibration), max-runs {args.max_runs}")
|
|
1324
|
+
return 0
|
|
1325
|
+
version = claude_version(args.claude)
|
|
1326
|
+
if out.exists() and not out.is_dir():
|
|
1327
|
+
raise TaskError("--out exists and is not a directory")
|
|
1328
|
+
contents = [entry for entry in out.iterdir() if entry.name != ".lock"] if out.is_dir() else []
|
|
1329
|
+
if contents and not (out / "plan.json").is_file():
|
|
1330
|
+
# The runner never deletes what it did not create in this batch: a root holding anything
|
|
1331
|
+
# but a frozen plan, including exports an interrupted run left before freezing it, is refused.
|
|
1332
|
+
raise TaskError("--out is not empty and holds no plan from this tool; use a new or empty directory")
|
|
1333
|
+
out.mkdir(mode=0o700, parents=True, exist_ok=True)
|
|
1334
|
+
lock = lock_output(out)
|
|
1335
|
+
try:
|
|
1336
|
+
return _locked_batch(args, out, arms, tasks, commits, planned, repo, roots, version)
|
|
1337
|
+
finally:
|
|
1338
|
+
lock.close()
|
|
1339
|
+
|
|
1340
|
+
|
|
1341
|
+
def _locked_batch(args, out, arms, tasks, commits, planned, repo, roots, version):
|
|
1342
|
+
try:
|
|
1343
|
+
existing = read_json(out / "plan.json")
|
|
1344
|
+
except json.JSONDecodeError:
|
|
1345
|
+
raise TaskError("plan.json under --out is not valid JSON; use a new --out") from None
|
|
1346
|
+
arm_dirs, arm_info, names, manifests = {}, {}, set(), {}
|
|
1347
|
+
for arm in arms:
|
|
1348
|
+
if arm == "off":
|
|
1349
|
+
arm_info[arm] = {"ref": None, "commit": None, "export_digest": None}
|
|
1350
|
+
continue
|
|
1351
|
+
dest = out / "arms" / arm
|
|
1352
|
+
names.add(plugin_name_of(dest) if dest.exists() else export_arm(repo, commits[arm], dest))
|
|
1353
|
+
arm_dirs[arm] = dest
|
|
1354
|
+
manifests[arm] = tree_manifest(dest)
|
|
1355
|
+
manifest_path = out / "arms" / f"{arm}.manifest.json"
|
|
1356
|
+
if not manifest_path.exists():
|
|
1357
|
+
write_atomic(manifest_path, json.dumps(manifests[arm]))
|
|
1358
|
+
arm_info[arm] = {"ref": args.base if arm == "base" else args.candidate, "commit": commits[arm],
|
|
1359
|
+
"export_digest": tree_digest(dest, manifests[arm])}
|
|
1360
|
+
if len(names) > 1:
|
|
1361
|
+
raise TaskError("base and candidate exports name different plugins")
|
|
1362
|
+
plugin_name = names.pop() if names else plugin_name_of(repo)
|
|
1363
|
+
plan = {
|
|
1364
|
+
"tool_version": TOOL_VERSION, "tool_sha256": tool_sha256(),
|
|
1365
|
+
"model": args.model, "effort": args.effort or "cli-default", "effort_flag": args.effort,
|
|
1366
|
+
"max_budget_usd": args.max_budget_usd, "disallowed_tools": DISALLOWED_TOOLS, "claude_version": version,
|
|
1367
|
+
"arms": arm_info, "plugin_name": plugin_name, "samples_override": args.samples,
|
|
1368
|
+
"tasks": [{"id": t["id"], "sha256": t["_sha256"], "samples": args.samples or t["samples"],
|
|
1369
|
+
"measures": t["measures"], "checks": [{"id": c["id"], "role": c["role"]} for c in t["checks"]]}
|
|
1370
|
+
for t in tasks],
|
|
1371
|
+
}
|
|
1372
|
+
plan["plan_hash"] = hashlib.sha256(canonical(plan).encode()).hexdigest()
|
|
1373
|
+
if existing is not None and existing.get("plan_hash") != plan["plan_hash"]:
|
|
1374
|
+
raise TaskError("plan differs from the one frozen under --out (tasks, refs, flags, tool or claude version "
|
|
1375
|
+
"changed); use a new --out")
|
|
1376
|
+
if existing is None:
|
|
1377
|
+
write_atomic(out / "plan.json", json.dumps(plan, indent=1, ensure_ascii=False))
|
|
1378
|
+
done = {(r["task"], r["arm"], r["sample"]) for r in load_records(out, plan)}
|
|
1379
|
+
pending = [(t, a, i) for t, a, i in planned if (t["id"], a, i) not in done]
|
|
1380
|
+
need_calibration = bool(pending) and not (read_json(out / "canary-calibration.json") or {}).get("fired")
|
|
1381
|
+
if len(pending) + need_calibration > args.max_runs:
|
|
1382
|
+
raise TaskError(f"{len(pending) + need_calibration} runs exceed --max-runs {args.max_runs}")
|
|
1383
|
+
ctx = make_context(out, plan, arm_dirs, roots, args.claude, manifests)
|
|
1384
|
+
stop = ctx["stop"]
|
|
1385
|
+
|
|
1386
|
+
def trip_stop():
|
|
1387
|
+
with ctx["spawn_lock"]: # a worker past its own check cannot spawn after this
|
|
1388
|
+
stop.set()
|
|
1389
|
+
|
|
1390
|
+
def worker(item):
|
|
1391
|
+
task, arm, index = item
|
|
1392
|
+
if stop.is_set():
|
|
1393
|
+
return None
|
|
1394
|
+
record = run_sample(ctx, task, arm, index)
|
|
1395
|
+
if record is None:
|
|
1396
|
+
return None
|
|
1397
|
+
if (record.get("max_utilization") or 0) >= args.stop_util:
|
|
1398
|
+
trip_stop()
|
|
1399
|
+
print(f"{task['id']} {arm} #{index}: valid={record['valid']} "
|
|
1400
|
+
+ " ".join(f"{k}={v['result']}" for k, v in record["checks"].items()), file=sys.stderr, flush=True)
|
|
1401
|
+
return record
|
|
1402
|
+
|
|
1403
|
+
# Before the first launch, SIGINT, SIGTERM and SIGHUP stop raising and only latch; a signal the
|
|
1404
|
+
# caller ignores stays ignored. Nothing is then raised in this thread asynchronously: a latched
|
|
1405
|
+
# signal takes effect in wait_latched, and the evidence below is written in full before exit.
|
|
1406
|
+
# The spawn check reads the same latch, so no run starts once a signal is latched.
|
|
1407
|
+
latched = ctx["latched"]
|
|
1408
|
+
handlers = {sig: signal.signal(sig, lambda signum, frame: latched.append(signum))
|
|
1409
|
+
for sig in (signal.SIGINT, signal.SIGTERM, signal.SIGHUP) if signal.getsignal(sig) != signal.SIG_IGN}
|
|
1410
|
+
pool = cf.ThreadPoolExecutor(max_workers=args.jobs)
|
|
1411
|
+
failure = None
|
|
1412
|
+
try:
|
|
1413
|
+
if need_calibration:
|
|
1414
|
+
calibration = wait_latched([pool.submit(calibrate_canary, ctx)], latched)[0]
|
|
1415
|
+
if calibration is None:
|
|
1416
|
+
raise KeyboardInterrupt
|
|
1417
|
+
if not calibration["fired"]:
|
|
1418
|
+
raise TaskError("the canary calibration did not fire, so a silent canary would prove nothing; "
|
|
1419
|
+
"no samples were run (see calibration/stream.jsonl under --out)")
|
|
1420
|
+
if (calibration.get("max_utilization") or 0) >= args.stop_util:
|
|
1421
|
+
trip_stop()
|
|
1422
|
+
wait_latched([pool.submit(worker, item) for item in pending], latched)
|
|
1423
|
+
except BaseException as exc:
|
|
1424
|
+
failure = exc
|
|
1425
|
+
try:
|
|
1426
|
+
records = _finish_batch(ctx, out, plan, pool, arm_dirs, manifests, failure)
|
|
1427
|
+
finally:
|
|
1428
|
+
for sig, handler in handlers.items():
|
|
1429
|
+
signal.signal(sig, handler)
|
|
1430
|
+
if failure is not None:
|
|
1431
|
+
if isinstance(failure, KeyboardInterrupt):
|
|
1432
|
+
print("interrupted: rerun the same command to resume", file=sys.stderr)
|
|
1433
|
+
return 130
|
|
1434
|
+
raise failure
|
|
1435
|
+
if latched:
|
|
1436
|
+
print("interrupted after the last run finished; the evidence was written in full; rerun the same command "
|
|
1437
|
+
"to resume", file=sys.stderr)
|
|
1438
|
+
return 130
|
|
1439
|
+
print(out / "report.md")
|
|
1440
|
+
recorded = {(r["task"], r["arm"], r["sample"]) for r in records}
|
|
1441
|
+
if any((t["id"], a, i) not in recorded for t, a, i in planned):
|
|
1442
|
+
print("stopped early: rate-limit utilization reached --stop-util; rerun the same command to resume",
|
|
1443
|
+
file=sys.stderr)
|
|
1444
|
+
return 3
|
|
1445
|
+
return 0
|
|
1446
|
+
|
|
1447
|
+
|
|
1448
|
+
def _finish_batch(ctx, out, plan, pool, arm_dirs, manifests, failure):
|
|
1449
|
+
"""Stop what is still running after a failure, then write the batch's integrity record and
|
|
1450
|
+
report. After a failure, evidence is best effort and never replaces that failure."""
|
|
1451
|
+
if failure is not None:
|
|
1452
|
+
with ctx["spawn_lock"]: # nothing starts after this point
|
|
1453
|
+
ctx["stop"].set()
|
|
1454
|
+
ctx["interrupted"].set()
|
|
1455
|
+
live = list(ctx["live"])
|
|
1456
|
+
for proc in live:
|
|
1457
|
+
kill_group(proc)
|
|
1458
|
+
pool.shutdown(wait=True, cancel_futures=failure is not None)
|
|
1459
|
+
try:
|
|
1460
|
+
record_integrity(out, arm_dirs, manifests)
|
|
1461
|
+
records = load_records(out, plan)
|
|
1462
|
+
write_report(out, plan, records)
|
|
1463
|
+
return records
|
|
1464
|
+
except Exception as report_error:
|
|
1465
|
+
if failure is None:
|
|
1466
|
+
raise
|
|
1467
|
+
print(f"report not written after the failure: {type(report_error).__name__}: {report_error}", file=sys.stderr)
|
|
1468
|
+
return None
|
|
1469
|
+
|
|
1470
|
+
|
|
1471
|
+
def main(argv=None):
|
|
1472
|
+
parser = argparse.ArgumentParser(description="Paired behavior eval (harness-patterns-and-eval.md §3.1).")
|
|
1473
|
+
parser.add_argument("--out", type=Path, help="private output root, outside every checkout of this repository")
|
|
1474
|
+
parser.add_argument("--base", help="git ref exported as the base arm")
|
|
1475
|
+
parser.add_argument("--candidate", help="git ref exported as the candidate arm")
|
|
1476
|
+
parser.add_argument("--tasks-dir", type=Path, default=DEFAULT_TASKS)
|
|
1477
|
+
parser.add_argument("--tasks", help="comma-separated task ids (default: all)")
|
|
1478
|
+
parser.add_argument("--arms", default=",".join(ARMS))
|
|
1479
|
+
parser.add_argument("--samples", type=int, help="override every task's sample count")
|
|
1480
|
+
parser.add_argument("--model", default="claude-opus-5-5")
|
|
1481
|
+
parser.add_argument("--effort", choices=("low", "medium", "high", "xhigh", "max"),
|
|
1482
|
+
help="pin the tested agent's effort (default: the CLI default)")
|
|
1483
|
+
parser.add_argument("--max-budget-usd", type=float, default=3.0, help="spend cap per run")
|
|
1484
|
+
parser.add_argument("--max-runs", type=int, default=60, help="refuse a batch that would start more runs")
|
|
1485
|
+
parser.add_argument("--jobs", type=int, default=3)
|
|
1486
|
+
parser.add_argument("--stop-util", type=float, default=0.9,
|
|
1487
|
+
help="stop starting runs once any rate-limit window reaches this utilization")
|
|
1488
|
+
parser.add_argument("--claude", default="claude", help="claude executable")
|
|
1489
|
+
parser.add_argument("--repo", type=Path, default=REPO_ROOT, help="repository the refs are exported from")
|
|
1490
|
+
parser.add_argument("--dry-run", action="store_true", help="print the plan; no runs, nothing written")
|
|
1491
|
+
parser.add_argument("--report-only", action="store_true", help="rebuild report.md from saved records")
|
|
1492
|
+
parser.add_argument("--regrade", action="store_true",
|
|
1493
|
+
help="rebuild every record from its saved stream and world with this tool; no model runs")
|
|
1494
|
+
parser.add_argument("--check-oracles", action="store_true",
|
|
1495
|
+
help="prove every task's checks pass on its good trajectory and fail on its bad ones")
|
|
1496
|
+
args = parser.parse_args(argv)
|
|
1497
|
+
try:
|
|
1498
|
+
only = [x for x in args.tasks.split(",") if x] if args.tasks else None
|
|
1499
|
+
if args.check_oracles:
|
|
1500
|
+
problems = check_oracles(load_tasks(args.tasks_dir, only))
|
|
1501
|
+
for problem in problems:
|
|
1502
|
+
print(f"oracle: {problem}", file=sys.stderr)
|
|
1503
|
+
print("paired_eval_oracles_ok" if not problems else f"paired_eval_oracles_failed {len(problems)}")
|
|
1504
|
+
return 1 if problems else 0
|
|
1505
|
+
if args.out is None:
|
|
1506
|
+
raise TaskError("--out is required")
|
|
1507
|
+
out = args.out.resolve()
|
|
1508
|
+
if args.report_only or args.regrade:
|
|
1509
|
+
plan = read_json(out / "plan.json")
|
|
1510
|
+
if plan is None:
|
|
1511
|
+
raise TaskError("no plan.json under --out")
|
|
1512
|
+
lock = lock_output(out)
|
|
1513
|
+
try:
|
|
1514
|
+
arm_dirs = {a: out / "arms" / a for a, info in plan["arms"].items() if info.get("commit")}
|
|
1515
|
+
if args.regrade:
|
|
1516
|
+
ctx = make_context(out, plan, arm_dirs, checkout_roots(args.repo.resolve()))
|
|
1517
|
+
print(f"regraded {regrade(ctx, out, load_tasks(args.tasks_dir))} records", file=sys.stderr)
|
|
1518
|
+
record_integrity(out, arm_dirs, {arm: frozen_manifest(out, plan, arm) for arm in arm_dirs})
|
|
1519
|
+
write_report(out, plan, load_records(out, plan))
|
|
1520
|
+
finally:
|
|
1521
|
+
lock.close()
|
|
1522
|
+
print(out / "report.md")
|
|
1523
|
+
return 0
|
|
1524
|
+
arms = [a for a in args.arms.split(",") if a]
|
|
1525
|
+
if not arms or len(set(arms)) != len(arms) or any(a not in ARMS for a in arms):
|
|
1526
|
+
raise TaskError(f"--arms must name distinct arms from {ARMS}")
|
|
1527
|
+
for arm, ref in (("base", args.base), ("candidate", args.candidate)):
|
|
1528
|
+
if arm in arms and not ref:
|
|
1529
|
+
raise TaskError(f"--{arm} is required for the {arm} arm")
|
|
1530
|
+
if args.jobs < 1 or args.max_runs < 1 or not 0 < args.stop_util <= 1 or args.max_budget_usd <= 0:
|
|
1531
|
+
raise TaskError("--jobs, --max-runs, --stop-util and --max-budget-usd must be positive")
|
|
1532
|
+
if args.samples is not None and not 1 <= args.samples <= 20:
|
|
1533
|
+
raise TaskError("--samples must be 1..20")
|
|
1534
|
+
return run_batch(args, out, arms, only)
|
|
1535
|
+
except TaskError as exc:
|
|
1536
|
+
print(f"skill-paired-eval: {exc}", file=sys.stderr)
|
|
1537
|
+
return 2
|
|
1538
|
+
except KeyboardInterrupt:
|
|
1539
|
+
print("interrupted", file=sys.stderr)
|
|
1540
|
+
return 130
|
|
1541
|
+
except BrokenPipeError:
|
|
1542
|
+
return 0
|
|
1543
|
+
|
|
1544
|
+
|
|
1545
|
+
if __name__ == "__main__":
|
|
1546
|
+
sys.exit(main())
|