@ccoalm/ccl-skills 0.14.0 → 0.15.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +80 -4
  2. package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +16 -4
  3. package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +13 -2
  4. package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +53 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +17 -3
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +65 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +5 -5
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +8 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +14 -14
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +1 -1
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +40 -83
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +2 -2
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +44 -37
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +1 -1
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +9 -6
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +6 -0
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +4 -4
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +52 -0
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +31 -7
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +188 -13
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +36 -13
  44. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +103 -21
  45. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +261 -14
  46. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +2 -0
  47. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
  48. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_runtime.py +428 -0
  49. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
  50. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +63 -5
  51. package/dist/assets/release.json +76 -56
  52. package/dist/claude-adapter.js +14 -7
  53. package/dist/codex-host.d.ts +1 -3
  54. package/dist/codex-host.js +6 -9
  55. package/dist/host-probe.d.ts +11 -0
  56. package/dist/host-probe.js +27 -0
  57. package/dist/opencode-adapter.js +24 -19
  58. package/dist/unified.js +11 -9
  59. package/package.json +1 -1
@@ -0,0 +1,428 @@
1
+ #!/usr/bin/env python3
2
+ """Offline regression cases for evaluator deadlines and terminal evidence."""
3
+ import importlib.util
4
+ import io
5
+ import json
6
+ import os
7
+ from pathlib import Path
8
+ import runpy
9
+ import signal
10
+ import subprocess
11
+ import sys
12
+ import tempfile
13
+ import time
14
+ import unittest
15
+ from unittest import mock
16
+
17
+
18
+ HERE = Path(__file__).resolve().parent
19
+ ROOT = HERE.parents[2]
20
+ spec = importlib.util.spec_from_file_location("behavior_eval", HERE / "skill-behavior-eval.py")
21
+ behavior = importlib.util.module_from_spec(spec)
22
+ spec.loader.exec_module(behavior)
23
+
24
+
25
+ def success(**overrides):
26
+ return {"type": "result", "subtype": "success", "is_error": False,
27
+ "result": "complete", **overrides}
28
+
29
+
30
+ class BehaviorRuntimeTests(unittest.TestCase):
31
+ def fake_process(self, communicate):
32
+ process = mock.Mock(pid=12345, returncode=0)
33
+ process.stdin = io.StringIO()
34
+ process.stdout = io.StringIO()
35
+ process.communicate.side_effect = communicate
36
+ return process
37
+
38
+ def invoke(self, source, prompt="x", timeout=0.15):
39
+ started = time.monotonic()
40
+ result = behavior._headless_claude([sys.executable, "-c", source], prompt, timeout)
41
+ return result, time.monotonic() - started
42
+
43
+ def test_complete_stream_and_teardown_exit(self):
44
+ for code in (0, 7):
45
+ with self.subTest(code=code):
46
+ result, _ = self.invoke(f"import sys; print({json.dumps(success())!r}); sys.exit({code})", timeout=2)
47
+ self.assertEqual(result[:2], ("complete", None))
48
+
49
+ def test_deadline_covers_all_io_and_process_wait(self):
50
+ cases = {
51
+ "stdin_write": ("import time; time.sleep(1.5)", "x" * 1_000_000),
52
+ "stdout_closed_process_live": ("import os,time; os.close(1); time.sleep(1.5)", "x"),
53
+ "stdout_open": ("import time; time.sleep(1.5)", "x"),
54
+ "terminal_before_process_exit": (f"import os,time; print({json.dumps(success())!r}, flush=True); os.close(1); time.sleep(1.5)", "x"),
55
+ }
56
+ for label, (source, prompt) in cases.items():
57
+ with self.subTest(boundary=label):
58
+ result, elapsed = self.invoke(source, prompt)
59
+ self.assertEqual(result[1], "timeout_0.15s")
60
+ self.assertLess(elapsed, 1.2)
61
+
62
+ @unittest.skipUnless(os.name == "posix", "process groups require POSIX")
63
+ def test_timeout_stops_descendant_with_inherited_stdout(self):
64
+ with tempfile.TemporaryDirectory() as temp:
65
+ marker = Path(temp) / "descendant-finished"
66
+ ready = Path(temp) / "descendant-ready"
67
+ descendant = f"import time; from pathlib import Path; Path({str(ready)!r}).touch(); time.sleep(1.2); Path({str(marker)!r}).touch()"
68
+ source = f"import subprocess,sys; subprocess.Popen([sys.executable,'-c',{descendant!r}])"
69
+ result, elapsed = self.invoke(source, timeout=0.5)
70
+ self.assertEqual(result[1], "timeout_0.5s")
71
+ self.assertLess(elapsed, 1.5)
72
+ self.assertTrue(ready.exists(), "fixture never reached the descendant path")
73
+ time.sleep(1.3)
74
+ self.assertFalse(marker.exists(), "descendant continued after timeout")
75
+
76
+ @unittest.skipUnless(os.name == "posix", "terminal process groups require POSIX")
77
+ def test_cli_signals_stop_detached_child(self):
78
+ for cancel in (signal.SIGINT, signal.SIGTERM):
79
+ with self.subTest(signal=cancel), tempfile.TemporaryDirectory() as temp:
80
+ folder = Path(temp)
81
+ ready, marker, calls = folder / "ready", folder / "late-work", folder / "calls"
82
+ stub = folder / "claude"
83
+ stub.write_text(f"#!{sys.executable}\nimport os,sys,time; from pathlib import Path\n"
84
+ "sys.stdin.read()\n"
85
+ f"with Path({str(calls)!r}).open('a') as log: log.write('started\\n')\n"
86
+ f"Path({str(ready)!r}).write_text(str(os.getpid()))\n"
87
+ f"time.sleep(1.4); Path({str(marker)!r}).touch()\n")
88
+ stub.chmod(0o700)
89
+ fixtures, output = folder / "fixtures.jsonl", folder / "output"
90
+ fixtures.write_text(json.dumps({"id": "synthetic", "prompt": "x"}) + "\n")
91
+ process = subprocess.Popen(
92
+ [sys.executable, str(HERE / "skill-behavior-eval.py"), "--fixtures", str(fixtures),
93
+ "--out", str(output), "--samples", "2", "--timeout", "10", "--no-judge"],
94
+ env={**os.environ, "HOME": temp, "PATH": temp + os.pathsep + os.environ["PATH"]},
95
+ stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, start_new_session=True)
96
+ child_pid = None
97
+ try:
98
+ deadline = time.monotonic() + 3
99
+ while (not ready.exists() or not ready.read_text()) and time.monotonic() < deadline:
100
+ time.sleep(0.01)
101
+ self.assertTrue(ready.exists(), "fixture never reached the child")
102
+ child_pid = int(ready.read_text())
103
+ os.killpg(process.pid, cancel)
104
+ _, stderr = process.communicate(timeout=3)
105
+ self.assertNotEqual(process.returncode, 0)
106
+ self.assertFalse(marker.exists(), "child performed work after cancellation")
107
+ with self.assertRaises(ProcessLookupError, msg="child survived evaluator cancellation"):
108
+ os.kill(child_pid, 0)
109
+ self.assertEqual(calls.read_text().splitlines(), ["started"])
110
+ self.assertEqual(list(output.glob("*.s*.txt")), [])
111
+ if cancel == signal.SIGINT:
112
+ self.assertIn("KeyboardInterrupt", stderr)
113
+ else:
114
+ self.assertEqual(process.returncode, 128 + signal.SIGTERM)
115
+ finally:
116
+ if process.poll() is None:
117
+ os.killpg(process.pid, signal.SIGKILL)
118
+ process.wait(timeout=2)
119
+ if child_pid is not None:
120
+ try:
121
+ os.kill(child_pid, 0)
122
+ os.killpg(child_pid, signal.SIGKILL)
123
+ except ProcessLookupError:
124
+ pass
125
+ process.stdout.close()
126
+ process.stderr.close()
127
+
128
+ @unittest.skipUnless(os.name == "posix", "process group refusal requires POSIX")
129
+ def test_cancellation_reports_cleanup_refusal_and_reraises(self):
130
+ for interrupted in (KeyboardInterrupt(), SystemExit(128 + signal.SIGTERM),
131
+ RuntimeError("custom signal handler"), OSError("pipe failure"),
132
+ BaseException("custom cancellation")):
133
+ with self.subTest(exception=type(interrupted).__name__):
134
+ process = self.fake_process([interrupted, ("", None)])
135
+ with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
136
+ mock.patch.object(behavior.os, "killpg", side_effect=PermissionError()) as killpg, \
137
+ mock.patch.object(sys, "stderr", io.StringIO()) as stderr:
138
+ with self.assertRaises(type(interrupted)) as raised:
139
+ behavior._headless_claude(["synthetic"], "x", 1)
140
+ self.assertIs(raised.exception, interrupted)
141
+ self.assertIn("cleanup_unconfirmed:group_kill_permission_denied", stderr.getvalue())
142
+ killpg.assert_called_once_with(process.pid, signal.SIGKILL)
143
+ self.assertEqual(process.communicate.call_args_list,
144
+ [mock.call(input="x", timeout=1), mock.call(timeout=1)])
145
+
146
+ @unittest.skipUnless(os.name == "posix", "process group cleanup requires POSIX")
147
+ def test_cleanup_io_errors_preserve_original_failure_and_remain_bounded(self):
148
+ for stopped in (subprocess.TimeoutExpired(["synthetic"], 1), OSError("original pipe failure")):
149
+ with self.subTest(exception=type(stopped).__name__):
150
+ process = self.fake_process([stopped, OSError("drain failure")])
151
+ process.wait.side_effect = OSError("reap failure")
152
+ with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
153
+ mock.patch.object(behavior.os, "killpg") as killpg, \
154
+ mock.patch.object(sys, "stderr", io.StringIO()) as stderr:
155
+ if isinstance(stopped, subprocess.TimeoutExpired):
156
+ result = behavior._headless_claude(["synthetic"], "x", 1)
157
+ self.assertEqual(result, (None, "timeout_1s;cleanup_unconfirmed:stdio_error,wait_error", None, []))
158
+ else:
159
+ with self.assertRaises(OSError) as raised:
160
+ behavior._headless_claude(["synthetic"], "x", 1)
161
+ self.assertIs(raised.exception, stopped)
162
+ self.assertIn("cleanup_unconfirmed:stdio_error,wait_error", stderr.getvalue())
163
+ killpg.assert_called_once_with(process.pid, signal.SIGKILL)
164
+ process.kill.assert_called_once_with()
165
+ process.wait.assert_called_once_with(timeout=1)
166
+ self.assertEqual(process.communicate.call_args_list,
167
+ [mock.call(input="x", timeout=1), mock.call(timeout=1)])
168
+ self.assertTrue(process.stdin.closed)
169
+ self.assertTrue(process.stdout.closed)
170
+
171
+ def test_cli_sigterm_handler_is_scoped_and_preserves_existing_handlers(self):
172
+ script = str(HERE / "skill-behavior-eval.py")
173
+ with mock.patch.object(signal, "signal") as install:
174
+ runpy.run_path(script, run_name="imported_fixture")
175
+ install.assert_not_called()
176
+ for previous in (signal.SIG_DFL, signal.SIG_IGN, lambda *_: None):
177
+ with self.subTest(previous=previous), \
178
+ mock.patch.object(signal, "getsignal", return_value=previous), \
179
+ mock.patch.object(signal, "signal") as install, \
180
+ mock.patch.object(sys, "argv", [script, "--help"]), \
181
+ mock.patch.object(sys, "stdout", io.StringIO()):
182
+ with self.assertRaises(SystemExit) as ended:
183
+ runpy.run_path(script, run_name="__main__")
184
+ self.assertEqual(ended.exception.code, 0)
185
+ if previous == signal.SIG_DFL:
186
+ self.assertEqual(install.call_count, 2)
187
+ self.assertEqual(install.call_args, mock.call(signal.SIGTERM, previous))
188
+ with self.assertRaises(SystemExit) as ended:
189
+ install.call_args_list[0].args[1](signal.SIGTERM, None)
190
+ self.assertEqual(ended.exception.code, 128 + signal.SIGTERM)
191
+ else:
192
+ install.assert_not_called()
193
+
194
+ def test_invalid_terminal_never_becomes_a_sample(self):
195
+ streams = {
196
+ "missing": [{"type": "assistant", "message": {"content": [{"type": "text", "text": "partial"}]}}],
197
+ "failed": [success(subtype="error_max_turns")],
198
+ "error_flag": [success(is_error=True)],
199
+ "duplicate": [success(), success()],
200
+ "permission_denial": [success(permission_denials=[{"tool": "Read"}])],
201
+ "api_error": [success(api_error_status=429)],
202
+ "malformed_result": [success(result={"invalid": "text"})],
203
+ "empty_result": [success(result="")],
204
+ }
205
+ for label, events in streams.items():
206
+ with self.subTest(stream=label):
207
+ stream = "\n".join(json.dumps(event) for event in events)
208
+ result, _ = self.invoke(f"print({stream!r})", timeout=2)
209
+ self.assertIsNone(result[0])
210
+ self.assertIsNotNone(result[1])
211
+
212
+ def test_timeout_cleanup_failures_stay_explicit_and_bounded(self):
213
+ timeout = lambda: subprocess.TimeoutExpired(["synthetic"], 0.15)
214
+ complete = (json.dumps(success()), None)
215
+ cases = (
216
+ ("clean", "posix", None, None, None, False, ""),
217
+ ("already_gone", "posix", ProcessLookupError(), None, None, False, ""),
218
+ ("group_denied", "posix", PermissionError(), None, None, False,
219
+ "group_kill_permission_denied"),
220
+ ("direct_only", "nt", None, None, None, False,
221
+ "descendant_cleanup_unsupported"),
222
+ ("direct_gone", "nt", None, ProcessLookupError(), None, False,
223
+ "descendant_cleanup_unsupported"),
224
+ ("direct_denied", "nt", None, PermissionError(), None, False,
225
+ "descendant_cleanup_unsupported,process_kill_permission_denied"),
226
+ ("pipe_still_open", "posix", None, None, None, True, "stdio_timeout"),
227
+ ("fallback_denied", "posix", None, PermissionError(), None, True,
228
+ "stdio_timeout,process_kill_permission_denied"),
229
+ ("wait_timeout", "posix", None, None, timeout(), True,
230
+ "stdio_timeout,wait_timeout"),
231
+ )
232
+ for label, platform, group_error, kill_error, wait_error, drain_timeout, detail in cases:
233
+ with self.subTest(boundary=label):
234
+ process = self.fake_process([timeout(), timeout() if drain_timeout else complete])
235
+ process.kill.side_effect = kill_error
236
+ process.wait.side_effect = wait_error
237
+ with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
238
+ mock.patch.object(behavior.os, "name", platform), \
239
+ mock.patch.object(behavior.os, "killpg", side_effect=group_error, create=True):
240
+ result = behavior._headless_claude(["synthetic"], "x", 0.15)
241
+ error = "timeout_0.15s" + (";cleanup_unconfirmed:" + detail if detail else "")
242
+ self.assertEqual(result, (None, error, None, []))
243
+ self.assertEqual(process.communicate.call_args_list,
244
+ [mock.call(input="x", timeout=0.15), mock.call(timeout=1)])
245
+ if drain_timeout:
246
+ process.wait.assert_called_once_with(timeout=1)
247
+ else:
248
+ process.wait.assert_not_called()
249
+ self.assertTrue(process.stdin.closed)
250
+ self.assertTrue(process.stdout.closed)
251
+
252
+ @unittest.skipUnless(os.name == "posix", "group-kill integration requires POSIX")
253
+ def test_cleanup_uncertainty_stops_new_samples_but_preserves_results(self):
254
+ for unconfirmed in (False, True):
255
+ with self.subTest(cleanup_unconfirmed=unconfirmed), tempfile.TemporaryDirectory() as temp:
256
+ timeout = subprocess.TimeoutExpired(["synthetic"], 1)
257
+ stream = (json.dumps(success()), None)
258
+ failed = self.fake_process([timeout, timeout if unconfirmed else stream])
259
+ failed.wait.side_effect = timeout if unconfirmed else None
260
+ processes = [self.fake_process([stream]), failed, self.fake_process([stream])]
261
+ argv = ["eval", "--samples", "3", "--timeout", "1", "--out", temp]
262
+ with mock.patch.object(sys, "argv", argv), \
263
+ mock.patch.object(behavior, "load_fixtures", return_value=[{"id": "synthetic", "prompt": "x"}]), \
264
+ mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
265
+ mock.patch.object(behavior.os, "killpg", side_effect=PermissionError() if unconfirmed else None), \
266
+ mock.patch.object(sys, "stdout", io.StringIO()) as output:
267
+ if unconfirmed:
268
+ with self.assertRaises(SystemExit) as ended:
269
+ behavior.main()
270
+ self.assertEqual(ended.exception.code, 1)
271
+ else:
272
+ behavior.main()
273
+ rows = [json.loads(line) for line in (Path(temp) / "run-log.jsonl").read_text().splitlines()]
274
+ self.assertEqual(len(rows), 2 if unconfirmed else 3)
275
+ self.assertEqual(spawn.call_count, len(rows))
276
+ expected = "timeout_1s" + (";cleanup_unconfirmed:group_kill_permission_denied,stdio_timeout,wait_timeout"
277
+ if unconfirmed else "")
278
+ self.assertEqual(rows[1]["error"], expected)
279
+ self.assertNotIn("error", rows[0])
280
+ self.assertIn("complete", (Path(temp) / "synthetic.current.s1.txt").read_text())
281
+ self.assertFalse((Path(temp) / "synthetic.current.s2.txt").exists())
282
+ self.assertEqual((Path(temp) / "synthetic.current.s3.txt").exists(), not unconfirmed)
283
+ self.assertEqual("ABORT" in output.getvalue(), unconfirmed)
284
+
285
+ def test_replacement_invalidates_only_the_sample_actually_started(self):
286
+ cases = ("fresh_failure", "stale_failure", "overwrite", "quota_stop", "cache_resume", "report_only")
287
+ for scenario in cases:
288
+ with self.subTest(scenario=scenario), tempfile.TemporaryDirectory() as temp:
289
+ folder = Path(temp)
290
+ contract = folder / "contract.md"
291
+ contract.write_text("synthetic contract")
292
+ fx = {"id": "synthetic", "prompt": "x"}
293
+ current = folder / "synthetic.current.s1.txt"
294
+ candidate = folder / "synthetic.candidate.s1.txt"
295
+ previous = folder / "previous.current.s1.txt"
296
+ previous.write_text("earlier completed sample")
297
+ sig = "stale" if scenario == "stale_failure" else behavior.sample_sig(fx, "current", "")
298
+ current.write_text(f"# sig: {sig}\nOLD_CURRENT\n")
299
+ candidate.write_text(f"# sig: {behavior.sample_sig(fx, 'candidate', contract.read_text())}\nOLD_CANDIDATE\n")
300
+ old_current, old_candidate = current.read_bytes(), candidate.read_bytes()
301
+ failed = scenario.endswith("failure")
302
+ stream = json.dumps(success(result="NEW_CURRENT"))
303
+ if scenario == "quota_stop":
304
+ stream += "\n" + json.dumps({"type": "rate_limit_event", "rate_limit_info": {"utilization": 0.99}})
305
+ first = ([subprocess.TimeoutExpired(["synthetic"], 1), ("", None)]
306
+ if failed else [(stream, None)])
307
+ processes = [self.fake_process(first), self.fake_process([(json.dumps(success(result="NEW_CANDIDATE")), None)])]
308
+ argv = ["eval", "--both-arms", "--samples", "1", "--timeout", "1", "--out", temp,
309
+ "--contract", str(contract), "--no-judge"]
310
+ if scenario not in ("stale_failure", "cache_resume"):
311
+ argv.append("--fresh")
312
+ if scenario == "report_only":
313
+ argv.append("--report-only")
314
+ with mock.patch.object(sys, "argv", argv), \
315
+ mock.patch.object(behavior, "load_fixtures", return_value=[fx]), \
316
+ mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
317
+ mock.patch.object(behavior.os, "killpg", create=True), \
318
+ mock.patch.object(sys, "stdout", io.StringIO()):
319
+ behavior.main()
320
+ self.assertEqual(previous.read_text(), "earlier completed sample")
321
+ if failed:
322
+ self.assertFalse(current.exists(), "failed replacement left the old sample available")
323
+ logs = [json.loads(line) for line in (folder / "run-log.jsonl").read_text().splitlines()]
324
+ self.assertEqual(logs[0]["error"], "timeout_1s")
325
+ elif scenario in ("cache_resume", "report_only"):
326
+ self.assertEqual(current.read_bytes(), old_current)
327
+ self.assertEqual(candidate.read_bytes(), old_candidate)
328
+ self.assertEqual(spawn.call_count, 0)
329
+ else:
330
+ self.assertIn("NEW_CURRENT", current.read_text())
331
+ self.assertNotIn("OLD_CURRENT", current.read_text())
332
+ if scenario == "quota_stop":
333
+ self.assertEqual(candidate.read_bytes(), old_candidate)
334
+ self.assertEqual(spawn.call_count, 1)
335
+ else:
336
+ rows = [json.loads(line) for line in (folder / "judge-verdicts.jsonl").read_text().splitlines()]
337
+ self.assertEqual(rows[0]["status"], "missing-arm" if failed else "scaffold")
338
+ if scenario == "overwrite":
339
+ self.assertIn("NEW_CANDIDATE", candidate.read_text())
340
+
341
+ @unittest.skipUnless(os.name == "posix", "group-kill integration requires POSIX")
342
+ def test_cleanup_uncertainty_stops_judges_and_writes_incomplete_report(self):
343
+ verdict = {"delta": "tie", "reduced_capabilities": [], "added_capabilities": [],
344
+ "behavior_change": "same", "confidence": "high", "needs_human": False}
345
+ stream = (json.dumps(success(result=json.dumps(verdict))), None)
346
+ for unconfirmed in (False, True):
347
+ with self.subTest(cleanup_unconfirmed=unconfirmed), tempfile.TemporaryDirectory() as temp:
348
+ timeout = subprocess.TimeoutExpired(["synthetic"], 1)
349
+ failed = self.fake_process([timeout, timeout if unconfirmed else stream])
350
+ failed.wait.side_effect = timeout if unconfirmed else None
351
+ processes = [self.fake_process([stream]), failed, self.fake_process([stream])]
352
+ fixtures = [{"id": name} for name in ("first", "failed", "last")]
353
+ argv = ["eval", "--report-only", "--timeout", "1", "--out", temp]
354
+ with mock.patch.object(sys, "argv", argv), \
355
+ mock.patch.object(behavior, "load_fixtures", return_value=fixtures), \
356
+ mock.patch.object(behavior, "read_saved_response", return_value="complete"), \
357
+ mock.patch.object(behavior, "sample_sig", return_value="sig"), \
358
+ mock.patch.object(behavior, "_saved_sig", return_value="sig"), \
359
+ mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
360
+ mock.patch.object(behavior.os, "killpg", side_effect=PermissionError() if unconfirmed else None), \
361
+ mock.patch.object(sys, "stdout", io.StringIO()):
362
+ if unconfirmed:
363
+ with self.assertRaises(SystemExit) as ended:
364
+ behavior.main()
365
+ self.assertEqual(ended.exception.code, 1)
366
+ else:
367
+ behavior.main()
368
+ rows = [json.loads(line) for line in (Path(temp) / "judge-verdicts.jsonl").read_text().splitlines()]
369
+ self.assertEqual(spawn.call_count, 2 if unconfirmed else 3)
370
+ self.assertEqual(rows[0]["status"], "judged")
371
+ self.assertTrue(rows[1]["status"].startswith("judge-error:timeout_1s"))
372
+ self.assertEqual(rows[2]["status"], "judge-skipped" if unconfirmed else "judged")
373
+ if unconfirmed:
374
+ self.assertIn("cleanup_unconfirmed", rows[2]["note"])
375
+ self.assertEqual(len(rows), 3)
376
+ self.assertIn("INCOMPLETE", (Path(temp) / "capability-delta-report.md").read_text())
377
+
378
+
379
+ class GoldenRuntimeTests(unittest.TestCase):
380
+ def run_golden(self, events, forbidden=False):
381
+ with tempfile.TemporaryDirectory() as temp:
382
+ folder = Path(temp)
383
+ stub = folder / "claude"
384
+ stub.write_text(f"#!{sys.executable}\nimport sys\nsys.stdin.read()\nprint({events!r})\n")
385
+ stub.chmod(0o700)
386
+ traces = folder / "traces"
387
+ traces.mkdir()
388
+ (traces / "case.json").write_text(json.dumps({
389
+ "id": "synthetic", "hub_skill": "testing-strategy", "frozen_at_sha": "root",
390
+ "trigger_prompt": "synthetic", "assert": {
391
+ "must_invoke_skill": ["testing-strategy"],
392
+ "must_not_invoke_skill": ["testing-strategy"] if forbidden else [],
393
+ },
394
+ }))
395
+ report = folder / "report.json"
396
+ proc = subprocess.run(["ruby", str(HERE / "eval-golden-trace.rb"), str(ROOT),
397
+ "--traces", str(traces), "--json", str(report)],
398
+ env={**os.environ, "PATH": str(folder) + os.pathsep + os.environ["PATH"]},
399
+ capture_output=True, text=True, timeout=5)
400
+ self.assertIn(proc.returncode, (0, 3), proc.stderr)
401
+ return json.loads(report.read_text())["results"][0]
402
+
403
+ def test_only_complete_valid_stream_can_pass(self):
404
+ skill = {"type": "assistant", "message": {"content": [
405
+ {"type": "tool_use", "name": "Skill", "input": {"skill": "testing-strategy"}},
406
+ ]}}
407
+ cases = {
408
+ "complete": ([skill, success()], "PASS"),
409
+ "missing": ([skill], "INCONCLUSIVE"),
410
+ "failed": ([skill, success(subtype="error_max_turns")], "INCONCLUSIVE"),
411
+ "error_flag": ([skill, success(is_error=True)], "INCONCLUSIVE"),
412
+ "duplicate": ([skill, success(), success()], "INCONCLUSIVE"),
413
+ "malformed_event": ([skill, [], success()], "INCONCLUSIVE"),
414
+ "malformed_content": ([skill, {"type": "assistant", "message": {"content": "bad"}}, success()], "INCONCLUSIVE"),
415
+ "permission_denial": ([skill, success(permission_denials=[{"tool": "Read"}])], "INCONCLUSIVE"),
416
+ }
417
+ for label, (events, expected) in cases.items():
418
+ with self.subTest(stream=label):
419
+ row = self.run_golden("\n".join(json.dumps(event) for event in events))
420
+ self.assertEqual(row["status"], expected)
421
+ if expected == "INCONCLUSIVE":
422
+ self.assertTrue(row["error"])
423
+ row = self.run_golden("\n".join(json.dumps(event) for event in (skill, success())), forbidden=True)
424
+ self.assertEqual(row["status"], "FAIL")
425
+
426
+
427
+ if __name__ == "__main__":
428
+ unittest.main()
@@ -139,28 +139,52 @@ git -C "$REPO_ROOT" show "$BASELINE_GATE_COMMIT:$GATE_PATH" > "$BASELINE_GATE" 2
139
139
  exit 1
140
140
  }
141
141
 
142
- # EXPECTED DIVERGENCES. Four landings back-filled a ledger row by corrective
143
- # rewrite — the repair for a round that merged with this gate red — so the owner
144
- # change sits below their base and the row cites an owner the range does not
145
- # touch. The candidate refuses that shape by design and deliberately offers no
146
- # author-declared escape, so these four historical ranges diverge. The gate is
147
- # diff-scoped and never re-judges landed history, so nothing operational depends
148
- # on them; this differential is the only thing that replays them.
142
+ # EXPECTED DIVERGENCES, two named classes, both "newly refused".
149
143
  #
150
- # Each exemption is constrained to ONE direction and ONE diagnostic. A blanket
151
- # "any mismatch at this SHA is fine" would also swallow the opposite direction —
152
- # a loosening — which is the failure this whole suite exists to catch. Entries are
153
- # named individually, never matched by pattern, and an entry that stops diverging
154
- # is reported as stale rather than tolerated.
155
- EXPECTED_DIVERGENCE_SHAS="f03b1140f 93d09c563 9f233728a 046612652"
156
- EXPECTED_DIVERGENCE_DIRECTION="newly refused"
157
- EXPECTED_DIVERGENCE_TOKEN="impact_chain_row_vouches_for_unchanged_owner"
158
- expected_divergence() { # <full sha> <direction> <candidate output>
159
- local short="${1:0:9}" direction="$2" out="$3" known
160
- [ "$direction" = "$EXPECTED_DIVERGENCE_DIRECTION" ] || return 1
161
- case "$out" in *"$EXPECTED_DIVERGENCE_TOKEN"*) : ;; *) return 1 ;; esac
162
- for known in $EXPECTED_DIVERGENCE_SHAS; do
163
- [ "$known" = "$short" ] && return 0
144
+ # `impact_chain_row_vouches_for_unchanged_owner` (four points): landings that
145
+ # back-filled a ledger row by corrective rewrite — the repair for a round that
146
+ # merged with this gate red — so the owner change sits below their base and the
147
+ # row cites an owner the range does not touch. The candidate refuses that shape by
148
+ # design and deliberately offers no author-declared escape.
149
+ #
150
+ # `impact_chain_gate_missing` (six points): merges from before CI judged the
151
+ # branch head (register row on the checkout ref binding, 2026-08-25). The gate now
152
+ # expands a merge git rebuilds from its parents into the branch's own rounds, so a
153
+ # merge is judged exactly as its branch was; these six branches carry owner work
154
+ # outside the round that declares it (a row appended before the work, or work
155
+ # after the last append) and the baseline accepted them only through the
156
+ # collapsed merged view, which no longer exists as a distinct verdict. Three of
157
+ # the six are refused on their own branch head by the baseline gate too; the other
158
+ # three are promotions or syncs whose second parent is the integration branch,
159
+ # where the same shapes sit one merge deeper.
160
+ #
161
+ # The gate is diff-scoped and never re-judges landed history, so nothing
162
+ # operational depends on these points; this differential is the only thing that
163
+ # replays them. Each entry is constrained to ONE direction and ONE diagnostic. A
164
+ # blanket "any mismatch at this SHA is fine" would also swallow the opposite
165
+ # direction — a loosening — which is the failure this whole suite exists to
166
+ # catch. Entries are named individually, never matched by pattern, and an entry
167
+ # that stops diverging is reported as stale rather than tolerated.
168
+ EXPECTED_DIVERGENCES="
169
+ f03b1140f:refused:impact_chain_row_vouches_for_unchanged_owner
170
+ 93d09c563:refused:impact_chain_row_vouches_for_unchanged_owner
171
+ 9f233728a:refused:impact_chain_row_vouches_for_unchanged_owner
172
+ 046612652:refused:impact_chain_row_vouches_for_unchanged_owner
173
+ b9de13869:refused:impact_chain_gate_missing
174
+ 8cea35e6d:refused:impact_chain_gate_missing
175
+ 95f06b2e6:refused:impact_chain_gate_missing
176
+ c0561c74e:refused:impact_chain_gate_missing
177
+ fad480296:refused:impact_chain_gate_missing
178
+ 90ec533e1:refused:impact_chain_gate_missing
179
+ "
180
+ expected_divergence() { # <full sha> <direction> <candidate output>; prints the matched token
181
+ local short="${1:0:9}" direction="$2" out="$3" entry sha dir token
182
+ case "$direction" in "newly refused") direction=refused ;; "newly accepted") direction=accepted ;; esac
183
+ for entry in $EXPECTED_DIVERGENCES; do
184
+ sha="${entry%%:*}"; token="${entry##*:}"; dir="${entry#*:}"; dir="${dir%%:*}"
185
+ [ "$sha" = "$short" ] || continue
186
+ [ "$dir" = "$direction" ] || return 1
187
+ case "$out" in *"$token"*) printf '%s' "$token"; return 0 ;; *) return 1 ;; esac
164
188
  done
165
189
  return 1
166
190
  }
@@ -170,7 +194,7 @@ expected_divergence() { # <full sha> <direction> <candidate output>
170
194
  # them, so the exemptions are stale and the run would pass while silently failing
171
195
  # the staleness check it never reaches.
172
196
  if cmp -s "$BASELINE_GATE" "$CANDIDATE_GATE"; then
173
- if [ -n "$(printf '%s' "$EXPECTED_DIVERGENCE_SHAS" | tr -d '[:space:]')" ]; then
197
+ if [ -n "$(printf '%s' "$EXPECTED_DIVERGENCES" | tr -d '[:space:]')" ]; then
174
198
  echo "FAIL: baseline and candidate are byte-identical, yet expected divergences are configured" >&2
175
199
  echo " identical gates cannot diverge — the exemptions are stale and must be removed" >&2
176
200
  exit 1
@@ -261,8 +285,8 @@ for point in $INTEGRATION_POINTS; do
261
285
  direction="newly accepted"
262
286
  fi
263
287
  if [ -n "$direction" ]; then
264
- if expected_divergence "$point" "$direction" "$candidate_out"; then
265
- flag=" (expected divergence: corrective-rewrite back-fill, $direction)"
288
+ if matched_token="$(expected_divergence "$point" "$direction" "$candidate_out")"; then
289
+ flag=" (expected divergence: $matched_token, $direction)"
266
290
  expected_seen=$((expected_seen + 1))
267
291
  else
268
292
  flag=" <== VERDICT MISMATCH: $direction"
@@ -412,7 +436,7 @@ if [ "$total_failures" -gt 0 ]; then
412
436
  fi
413
437
  # An expected divergence that stops diverging means the exemption is stale and
414
438
  # should be removed, so it is reported rather than silently tolerated.
415
- expected_total="$(for e in $EXPECTED_DIVERGENCE_SHAS; do echo "$e"; done | wc -l | tr -d ' ')"
439
+ expected_total="$(for e in $EXPECTED_DIVERGENCES; do echo "$e"; done | wc -l | tr -d ' ')"
416
440
  if [ "$expected_seen" != "$expected_total" ]; then
417
441
  echo "FAIL: $expected_seen of $expected_total expected divergences actually diverged" >&2
418
442
  echo " an exemption that no longer fires is stale — remove it" >&2
@@ -616,6 +616,64 @@ out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
616
616
  check "a round with no accepted ledger at its own base refuses the whole chain, naming that round" \
617
617
  '[ "$rc" = 1 ] && case "$out" in *"round $UNBOUND_MERGE"*"does not bind at its own base"*) true;; *) false;; esac'
618
618
 
619
+ # Evidence is read from the LANDING tree, not from the round's own checkout. A
620
+ # round that merged without its ledger is still a set of bytes some later review
621
+ # can freeze and inspect; a validator-accepted closeout for exactly that candidate,
622
+ # committed on the integration branch afterwards, binds it. A closeout for any
623
+ # other digest does not, wherever it sits — the candidate hash, not the file's
624
+ # location, is what makes a ledger evidence. The retro ledger lands as a round of
625
+ # its own (a merge whose only change is receipt-shaped evidence), because a
626
+ # direct commit on the integration branch is refused for its own reason.
627
+ D_BASE="$(git -C "$CHAIN" rev-parse HEAD^1)"
628
+ git -C "$CHAIN" checkout -q round-d
629
+ D_DIGEST="$(run_chain --base "$D_BASE" --print-candidate)"
630
+ git -C "$CHAIN" checkout -q dev
631
+ edit_retro_wrong() {
632
+ mkdir -p "$CHAIN/specs/d-retro-wrong/evidence"
633
+ write_closeout "$CHAIN/specs/d-retro-wrong/evidence/closeout.json" "$(printf '%064d' 7)"
634
+ }
635
+ land_round d-retro-wrong no-ledger edit_retro_wrong
636
+ out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
637
+ check "a later closeout for a different digest does not bind the unbound round" \
638
+ '[ "$rc" = 1 ] && case "$out" in *"round $UNBOUND_MERGE"*"does not bind at its own base"*) true;; *) false;; esac'
639
+ # The right digest in a closeout the validator rejects is still not evidence.
640
+ edit_retro_rejected() {
641
+ mkdir -p "$CHAIN/specs/d-retro-rejected/evidence"
642
+ python3 - "$CHAIN/specs/d-retro-rejected/evidence/closeout.json" "$D_DIGEST" <<'PY'
643
+ import json, sys
644
+ from pathlib import Path
645
+ Path(sys.argv[1]).write_text(json.dumps({"schema_version": 3, "closeout_state": "forged", "controller_receipts": [], "candidate_sha256": sys.argv[2]}))
646
+ PY
647
+ }
648
+ land_round d-retro-rejected no-ledger edit_retro_rejected
649
+ out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
650
+ check "a later closeout for the right digest that the validator rejects does not bind the unbound round" \
651
+ '[ "$rc" = 1 ] && case "$out" in *"round $UNBOUND_MERGE"*"does not bind at its own base"*) true;; *) false;; esac'
652
+ # Evidence in the landing tree must be COMMITTED there, exactly as before: an
653
+ # untracked closeout for the right digest is refused outright, not read from disk.
654
+ mkdir -p "$CHAIN/specs/d-retro/evidence"
655
+ write_closeout "$CHAIN/specs/d-retro/evidence/closeout.json" "$D_DIGEST"
656
+ out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
657
+ check "an uncommitted closeout in the landing tree is refused before any round is bound" \
658
+ '[ "$rc" != 0 ] && case "$out" in *"uncommitted changes"*) true;; *) false;; esac'
659
+ rm -r "$CHAIN/specs/d-retro"
660
+ edit_retro_right() {
661
+ mkdir -p "$CHAIN/specs/d-retro/evidence"
662
+ write_closeout "$CHAIN/specs/d-retro/evidence/closeout.json" "$D_DIGEST"
663
+ }
664
+ land_round d-retro no-ledger edit_retro_right
665
+ out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
666
+ check "a validator-accepted closeout for the round's own candidate, committed on the integration branch after the merge, binds it through the chain" \
667
+ '[ "$rc" = 0 ] && case "$out" in *"first-parent chain"*"specs/d-retro/evidence/closeout.json"*) true;; *) false;; esac'
668
+ # Later evidence changes where a ledger may be found, never what the round is:
669
+ # the unbound round's candidate at its own base hashes exactly as it did before
670
+ # any retro ledger existed, so its packet, base and excludes are untouched.
671
+ git -C "$CHAIN" checkout -q round-d
672
+ D_DIGEST_AFTER="$(run_chain --base "$D_BASE" --print-candidate)"
673
+ git -C "$CHAIN" checkout -q dev
674
+ check "the round's candidate hash is unchanged by evidence committed later in the landing tree" \
675
+ '[ "$D_DIGEST_AFTER" = "$D_DIGEST" ]'
676
+
619
677
  probe_chain
620
678
  printf 'pushed straight to the integration branch\n' >"$CHAIN/lane/direct.txt"
621
679
  git -C "$CHAIN" add -A && git -C "$CHAIN" commit -qm "direct commit on dev"
@@ -827,10 +885,10 @@ check "the passing chain state still passes after the probes" \
827
885
  # the real checkout it ships in, or a break in that path passes every test here.
828
886
  REAL_ROOT="$(cd "$DIR/../../.." && pwd -P)"
829
887
  real_out="$(env -u CCL_SKILL_BASE_REF python3 "$GATE" --repo-root "$REAL_ROOT" --base HEAD~1 --print-candidate 2>&1)"; real_rc=$?
830
- # Both outputs are legitimate and which one appears depends on what the parent
831
- # commit happened to touch: a candidate hash when a reviewed path moved, the
832
- # no-change signal when it did not. Accepting only the first makes this assertion
833
- # a function of repository state rather than of the gate.
888
+ # Three outputs are legitimate and which one appears depends on what the parent
889
+ # commit happened to touch: a candidate hash when a bounded reviewed path moved,
890
+ # the no-change signal when it did not, or the packet-ceiling refusal when the
891
+ # parent diff is too large for one reviewer packet.
834
892
  # Which answer is legitimate depends on the checkout, so decide that FIRST and then
835
893
  # assert the one matching branch. Accepting every output would make this assertion a
836
894
  # function of repository state; branching on the state keeps it an assertion about
@@ -842,7 +900,7 @@ if [ -n "$real_dirty" ]; then
842
900
  '[ "$real_rc" != 0 ] && case "$real_out" in *"uncommitted changes"*) true;; *) false;; esac'
843
901
  else
844
902
  check "the gate answers for the clean real checkout it ships in, whichever legitimate answer applies" \
845
- '[ "$real_rc" = 0 ] && { case "$real_out" in [0-9a-f]*) [ ${#real_out} = 64 ];; *) false;; esac || case "$real_out" in *review_ledger_binding_no_change*) true;; *) false;; esac; }'
903
+ '{ [ "$real_rc" = 0 ] && { case "$real_out" in [0-9a-f]*) [ ${#real_out} = 64 ];; *) false;; esac || case "$real_out" in *review_ledger_binding_no_change*) true;; *) false;; esac; }; } || { [ "$real_rc" = 1 ] && case "$real_out" in *"cannot freeze the candidate packet"*"review packet exceeds 200000 bytes"*) true;; *) false;; esac; }'
846
904
  fi
847
905
 
848
906
  if [ "$fails" -gt 0 ]; then