@ccoalm/ccl-skills 0.14.0 → 0.15.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +80 -4
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +16 -4
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +13 -2
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +17 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +14 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +40 -83
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +44 -37
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +9 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +52 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +31 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +188 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +36 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +103 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +261 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_runtime.py +428 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +63 -5
- package/dist/assets/release.json +76 -56
- package/dist/claude-adapter.js +14 -7
- package/dist/codex-host.d.ts +1 -3
- package/dist/codex-host.js +6 -9
- package/dist/host-probe.d.ts +11 -0
- package/dist/host-probe.js +27 -0
- package/dist/opencode-adapter.js +24 -19
- package/dist/unified.js +11 -9
- package/package.json +1 -1
|
@@ -0,0 +1,428 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Offline regression cases for evaluator deadlines and terminal evidence."""
|
|
3
|
+
import importlib.util
|
|
4
|
+
import io
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
import runpy
|
|
9
|
+
import signal
|
|
10
|
+
import subprocess
|
|
11
|
+
import sys
|
|
12
|
+
import tempfile
|
|
13
|
+
import time
|
|
14
|
+
import unittest
|
|
15
|
+
from unittest import mock
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
HERE = Path(__file__).resolve().parent
|
|
19
|
+
ROOT = HERE.parents[2]
|
|
20
|
+
spec = importlib.util.spec_from_file_location("behavior_eval", HERE / "skill-behavior-eval.py")
|
|
21
|
+
behavior = importlib.util.module_from_spec(spec)
|
|
22
|
+
spec.loader.exec_module(behavior)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def success(**overrides):
|
|
26
|
+
return {"type": "result", "subtype": "success", "is_error": False,
|
|
27
|
+
"result": "complete", **overrides}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class BehaviorRuntimeTests(unittest.TestCase):
|
|
31
|
+
def fake_process(self, communicate):
|
|
32
|
+
process = mock.Mock(pid=12345, returncode=0)
|
|
33
|
+
process.stdin = io.StringIO()
|
|
34
|
+
process.stdout = io.StringIO()
|
|
35
|
+
process.communicate.side_effect = communicate
|
|
36
|
+
return process
|
|
37
|
+
|
|
38
|
+
def invoke(self, source, prompt="x", timeout=0.15):
|
|
39
|
+
started = time.monotonic()
|
|
40
|
+
result = behavior._headless_claude([sys.executable, "-c", source], prompt, timeout)
|
|
41
|
+
return result, time.monotonic() - started
|
|
42
|
+
|
|
43
|
+
def test_complete_stream_and_teardown_exit(self):
|
|
44
|
+
for code in (0, 7):
|
|
45
|
+
with self.subTest(code=code):
|
|
46
|
+
result, _ = self.invoke(f"import sys; print({json.dumps(success())!r}); sys.exit({code})", timeout=2)
|
|
47
|
+
self.assertEqual(result[:2], ("complete", None))
|
|
48
|
+
|
|
49
|
+
def test_deadline_covers_all_io_and_process_wait(self):
|
|
50
|
+
cases = {
|
|
51
|
+
"stdin_write": ("import time; time.sleep(1.5)", "x" * 1_000_000),
|
|
52
|
+
"stdout_closed_process_live": ("import os,time; os.close(1); time.sleep(1.5)", "x"),
|
|
53
|
+
"stdout_open": ("import time; time.sleep(1.5)", "x"),
|
|
54
|
+
"terminal_before_process_exit": (f"import os,time; print({json.dumps(success())!r}, flush=True); os.close(1); time.sleep(1.5)", "x"),
|
|
55
|
+
}
|
|
56
|
+
for label, (source, prompt) in cases.items():
|
|
57
|
+
with self.subTest(boundary=label):
|
|
58
|
+
result, elapsed = self.invoke(source, prompt)
|
|
59
|
+
self.assertEqual(result[1], "timeout_0.15s")
|
|
60
|
+
self.assertLess(elapsed, 1.2)
|
|
61
|
+
|
|
62
|
+
@unittest.skipUnless(os.name == "posix", "process groups require POSIX")
|
|
63
|
+
def test_timeout_stops_descendant_with_inherited_stdout(self):
|
|
64
|
+
with tempfile.TemporaryDirectory() as temp:
|
|
65
|
+
marker = Path(temp) / "descendant-finished"
|
|
66
|
+
ready = Path(temp) / "descendant-ready"
|
|
67
|
+
descendant = f"import time; from pathlib import Path; Path({str(ready)!r}).touch(); time.sleep(1.2); Path({str(marker)!r}).touch()"
|
|
68
|
+
source = f"import subprocess,sys; subprocess.Popen([sys.executable,'-c',{descendant!r}])"
|
|
69
|
+
result, elapsed = self.invoke(source, timeout=0.5)
|
|
70
|
+
self.assertEqual(result[1], "timeout_0.5s")
|
|
71
|
+
self.assertLess(elapsed, 1.5)
|
|
72
|
+
self.assertTrue(ready.exists(), "fixture never reached the descendant path")
|
|
73
|
+
time.sleep(1.3)
|
|
74
|
+
self.assertFalse(marker.exists(), "descendant continued after timeout")
|
|
75
|
+
|
|
76
|
+
@unittest.skipUnless(os.name == "posix", "terminal process groups require POSIX")
|
|
77
|
+
def test_cli_signals_stop_detached_child(self):
|
|
78
|
+
for cancel in (signal.SIGINT, signal.SIGTERM):
|
|
79
|
+
with self.subTest(signal=cancel), tempfile.TemporaryDirectory() as temp:
|
|
80
|
+
folder = Path(temp)
|
|
81
|
+
ready, marker, calls = folder / "ready", folder / "late-work", folder / "calls"
|
|
82
|
+
stub = folder / "claude"
|
|
83
|
+
stub.write_text(f"#!{sys.executable}\nimport os,sys,time; from pathlib import Path\n"
|
|
84
|
+
"sys.stdin.read()\n"
|
|
85
|
+
f"with Path({str(calls)!r}).open('a') as log: log.write('started\\n')\n"
|
|
86
|
+
f"Path({str(ready)!r}).write_text(str(os.getpid()))\n"
|
|
87
|
+
f"time.sleep(1.4); Path({str(marker)!r}).touch()\n")
|
|
88
|
+
stub.chmod(0o700)
|
|
89
|
+
fixtures, output = folder / "fixtures.jsonl", folder / "output"
|
|
90
|
+
fixtures.write_text(json.dumps({"id": "synthetic", "prompt": "x"}) + "\n")
|
|
91
|
+
process = subprocess.Popen(
|
|
92
|
+
[sys.executable, str(HERE / "skill-behavior-eval.py"), "--fixtures", str(fixtures),
|
|
93
|
+
"--out", str(output), "--samples", "2", "--timeout", "10", "--no-judge"],
|
|
94
|
+
env={**os.environ, "HOME": temp, "PATH": temp + os.pathsep + os.environ["PATH"]},
|
|
95
|
+
stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, start_new_session=True)
|
|
96
|
+
child_pid = None
|
|
97
|
+
try:
|
|
98
|
+
deadline = time.monotonic() + 3
|
|
99
|
+
while (not ready.exists() or not ready.read_text()) and time.monotonic() < deadline:
|
|
100
|
+
time.sleep(0.01)
|
|
101
|
+
self.assertTrue(ready.exists(), "fixture never reached the child")
|
|
102
|
+
child_pid = int(ready.read_text())
|
|
103
|
+
os.killpg(process.pid, cancel)
|
|
104
|
+
_, stderr = process.communicate(timeout=3)
|
|
105
|
+
self.assertNotEqual(process.returncode, 0)
|
|
106
|
+
self.assertFalse(marker.exists(), "child performed work after cancellation")
|
|
107
|
+
with self.assertRaises(ProcessLookupError, msg="child survived evaluator cancellation"):
|
|
108
|
+
os.kill(child_pid, 0)
|
|
109
|
+
self.assertEqual(calls.read_text().splitlines(), ["started"])
|
|
110
|
+
self.assertEqual(list(output.glob("*.s*.txt")), [])
|
|
111
|
+
if cancel == signal.SIGINT:
|
|
112
|
+
self.assertIn("KeyboardInterrupt", stderr)
|
|
113
|
+
else:
|
|
114
|
+
self.assertEqual(process.returncode, 128 + signal.SIGTERM)
|
|
115
|
+
finally:
|
|
116
|
+
if process.poll() is None:
|
|
117
|
+
os.killpg(process.pid, signal.SIGKILL)
|
|
118
|
+
process.wait(timeout=2)
|
|
119
|
+
if child_pid is not None:
|
|
120
|
+
try:
|
|
121
|
+
os.kill(child_pid, 0)
|
|
122
|
+
os.killpg(child_pid, signal.SIGKILL)
|
|
123
|
+
except ProcessLookupError:
|
|
124
|
+
pass
|
|
125
|
+
process.stdout.close()
|
|
126
|
+
process.stderr.close()
|
|
127
|
+
|
|
128
|
+
@unittest.skipUnless(os.name == "posix", "process group refusal requires POSIX")
|
|
129
|
+
def test_cancellation_reports_cleanup_refusal_and_reraises(self):
|
|
130
|
+
for interrupted in (KeyboardInterrupt(), SystemExit(128 + signal.SIGTERM),
|
|
131
|
+
RuntimeError("custom signal handler"), OSError("pipe failure"),
|
|
132
|
+
BaseException("custom cancellation")):
|
|
133
|
+
with self.subTest(exception=type(interrupted).__name__):
|
|
134
|
+
process = self.fake_process([interrupted, ("", None)])
|
|
135
|
+
with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
|
|
136
|
+
mock.patch.object(behavior.os, "killpg", side_effect=PermissionError()) as killpg, \
|
|
137
|
+
mock.patch.object(sys, "stderr", io.StringIO()) as stderr:
|
|
138
|
+
with self.assertRaises(type(interrupted)) as raised:
|
|
139
|
+
behavior._headless_claude(["synthetic"], "x", 1)
|
|
140
|
+
self.assertIs(raised.exception, interrupted)
|
|
141
|
+
self.assertIn("cleanup_unconfirmed:group_kill_permission_denied", stderr.getvalue())
|
|
142
|
+
killpg.assert_called_once_with(process.pid, signal.SIGKILL)
|
|
143
|
+
self.assertEqual(process.communicate.call_args_list,
|
|
144
|
+
[mock.call(input="x", timeout=1), mock.call(timeout=1)])
|
|
145
|
+
|
|
146
|
+
@unittest.skipUnless(os.name == "posix", "process group cleanup requires POSIX")
|
|
147
|
+
def test_cleanup_io_errors_preserve_original_failure_and_remain_bounded(self):
|
|
148
|
+
for stopped in (subprocess.TimeoutExpired(["synthetic"], 1), OSError("original pipe failure")):
|
|
149
|
+
with self.subTest(exception=type(stopped).__name__):
|
|
150
|
+
process = self.fake_process([stopped, OSError("drain failure")])
|
|
151
|
+
process.wait.side_effect = OSError("reap failure")
|
|
152
|
+
with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
|
|
153
|
+
mock.patch.object(behavior.os, "killpg") as killpg, \
|
|
154
|
+
mock.patch.object(sys, "stderr", io.StringIO()) as stderr:
|
|
155
|
+
if isinstance(stopped, subprocess.TimeoutExpired):
|
|
156
|
+
result = behavior._headless_claude(["synthetic"], "x", 1)
|
|
157
|
+
self.assertEqual(result, (None, "timeout_1s;cleanup_unconfirmed:stdio_error,wait_error", None, []))
|
|
158
|
+
else:
|
|
159
|
+
with self.assertRaises(OSError) as raised:
|
|
160
|
+
behavior._headless_claude(["synthetic"], "x", 1)
|
|
161
|
+
self.assertIs(raised.exception, stopped)
|
|
162
|
+
self.assertIn("cleanup_unconfirmed:stdio_error,wait_error", stderr.getvalue())
|
|
163
|
+
killpg.assert_called_once_with(process.pid, signal.SIGKILL)
|
|
164
|
+
process.kill.assert_called_once_with()
|
|
165
|
+
process.wait.assert_called_once_with(timeout=1)
|
|
166
|
+
self.assertEqual(process.communicate.call_args_list,
|
|
167
|
+
[mock.call(input="x", timeout=1), mock.call(timeout=1)])
|
|
168
|
+
self.assertTrue(process.stdin.closed)
|
|
169
|
+
self.assertTrue(process.stdout.closed)
|
|
170
|
+
|
|
171
|
+
def test_cli_sigterm_handler_is_scoped_and_preserves_existing_handlers(self):
|
|
172
|
+
script = str(HERE / "skill-behavior-eval.py")
|
|
173
|
+
with mock.patch.object(signal, "signal") as install:
|
|
174
|
+
runpy.run_path(script, run_name="imported_fixture")
|
|
175
|
+
install.assert_not_called()
|
|
176
|
+
for previous in (signal.SIG_DFL, signal.SIG_IGN, lambda *_: None):
|
|
177
|
+
with self.subTest(previous=previous), \
|
|
178
|
+
mock.patch.object(signal, "getsignal", return_value=previous), \
|
|
179
|
+
mock.patch.object(signal, "signal") as install, \
|
|
180
|
+
mock.patch.object(sys, "argv", [script, "--help"]), \
|
|
181
|
+
mock.patch.object(sys, "stdout", io.StringIO()):
|
|
182
|
+
with self.assertRaises(SystemExit) as ended:
|
|
183
|
+
runpy.run_path(script, run_name="__main__")
|
|
184
|
+
self.assertEqual(ended.exception.code, 0)
|
|
185
|
+
if previous == signal.SIG_DFL:
|
|
186
|
+
self.assertEqual(install.call_count, 2)
|
|
187
|
+
self.assertEqual(install.call_args, mock.call(signal.SIGTERM, previous))
|
|
188
|
+
with self.assertRaises(SystemExit) as ended:
|
|
189
|
+
install.call_args_list[0].args[1](signal.SIGTERM, None)
|
|
190
|
+
self.assertEqual(ended.exception.code, 128 + signal.SIGTERM)
|
|
191
|
+
else:
|
|
192
|
+
install.assert_not_called()
|
|
193
|
+
|
|
194
|
+
def test_invalid_terminal_never_becomes_a_sample(self):
|
|
195
|
+
streams = {
|
|
196
|
+
"missing": [{"type": "assistant", "message": {"content": [{"type": "text", "text": "partial"}]}}],
|
|
197
|
+
"failed": [success(subtype="error_max_turns")],
|
|
198
|
+
"error_flag": [success(is_error=True)],
|
|
199
|
+
"duplicate": [success(), success()],
|
|
200
|
+
"permission_denial": [success(permission_denials=[{"tool": "Read"}])],
|
|
201
|
+
"api_error": [success(api_error_status=429)],
|
|
202
|
+
"malformed_result": [success(result={"invalid": "text"})],
|
|
203
|
+
"empty_result": [success(result="")],
|
|
204
|
+
}
|
|
205
|
+
for label, events in streams.items():
|
|
206
|
+
with self.subTest(stream=label):
|
|
207
|
+
stream = "\n".join(json.dumps(event) for event in events)
|
|
208
|
+
result, _ = self.invoke(f"print({stream!r})", timeout=2)
|
|
209
|
+
self.assertIsNone(result[0])
|
|
210
|
+
self.assertIsNotNone(result[1])
|
|
211
|
+
|
|
212
|
+
def test_timeout_cleanup_failures_stay_explicit_and_bounded(self):
|
|
213
|
+
timeout = lambda: subprocess.TimeoutExpired(["synthetic"], 0.15)
|
|
214
|
+
complete = (json.dumps(success()), None)
|
|
215
|
+
cases = (
|
|
216
|
+
("clean", "posix", None, None, None, False, ""),
|
|
217
|
+
("already_gone", "posix", ProcessLookupError(), None, None, False, ""),
|
|
218
|
+
("group_denied", "posix", PermissionError(), None, None, False,
|
|
219
|
+
"group_kill_permission_denied"),
|
|
220
|
+
("direct_only", "nt", None, None, None, False,
|
|
221
|
+
"descendant_cleanup_unsupported"),
|
|
222
|
+
("direct_gone", "nt", None, ProcessLookupError(), None, False,
|
|
223
|
+
"descendant_cleanup_unsupported"),
|
|
224
|
+
("direct_denied", "nt", None, PermissionError(), None, False,
|
|
225
|
+
"descendant_cleanup_unsupported,process_kill_permission_denied"),
|
|
226
|
+
("pipe_still_open", "posix", None, None, None, True, "stdio_timeout"),
|
|
227
|
+
("fallback_denied", "posix", None, PermissionError(), None, True,
|
|
228
|
+
"stdio_timeout,process_kill_permission_denied"),
|
|
229
|
+
("wait_timeout", "posix", None, None, timeout(), True,
|
|
230
|
+
"stdio_timeout,wait_timeout"),
|
|
231
|
+
)
|
|
232
|
+
for label, platform, group_error, kill_error, wait_error, drain_timeout, detail in cases:
|
|
233
|
+
with self.subTest(boundary=label):
|
|
234
|
+
process = self.fake_process([timeout(), timeout() if drain_timeout else complete])
|
|
235
|
+
process.kill.side_effect = kill_error
|
|
236
|
+
process.wait.side_effect = wait_error
|
|
237
|
+
with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
|
|
238
|
+
mock.patch.object(behavior.os, "name", platform), \
|
|
239
|
+
mock.patch.object(behavior.os, "killpg", side_effect=group_error, create=True):
|
|
240
|
+
result = behavior._headless_claude(["synthetic"], "x", 0.15)
|
|
241
|
+
error = "timeout_0.15s" + (";cleanup_unconfirmed:" + detail if detail else "")
|
|
242
|
+
self.assertEqual(result, (None, error, None, []))
|
|
243
|
+
self.assertEqual(process.communicate.call_args_list,
|
|
244
|
+
[mock.call(input="x", timeout=0.15), mock.call(timeout=1)])
|
|
245
|
+
if drain_timeout:
|
|
246
|
+
process.wait.assert_called_once_with(timeout=1)
|
|
247
|
+
else:
|
|
248
|
+
process.wait.assert_not_called()
|
|
249
|
+
self.assertTrue(process.stdin.closed)
|
|
250
|
+
self.assertTrue(process.stdout.closed)
|
|
251
|
+
|
|
252
|
+
@unittest.skipUnless(os.name == "posix", "group-kill integration requires POSIX")
|
|
253
|
+
def test_cleanup_uncertainty_stops_new_samples_but_preserves_results(self):
|
|
254
|
+
for unconfirmed in (False, True):
|
|
255
|
+
with self.subTest(cleanup_unconfirmed=unconfirmed), tempfile.TemporaryDirectory() as temp:
|
|
256
|
+
timeout = subprocess.TimeoutExpired(["synthetic"], 1)
|
|
257
|
+
stream = (json.dumps(success()), None)
|
|
258
|
+
failed = self.fake_process([timeout, timeout if unconfirmed else stream])
|
|
259
|
+
failed.wait.side_effect = timeout if unconfirmed else None
|
|
260
|
+
processes = [self.fake_process([stream]), failed, self.fake_process([stream])]
|
|
261
|
+
argv = ["eval", "--samples", "3", "--timeout", "1", "--out", temp]
|
|
262
|
+
with mock.patch.object(sys, "argv", argv), \
|
|
263
|
+
mock.patch.object(behavior, "load_fixtures", return_value=[{"id": "synthetic", "prompt": "x"}]), \
|
|
264
|
+
mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
|
|
265
|
+
mock.patch.object(behavior.os, "killpg", side_effect=PermissionError() if unconfirmed else None), \
|
|
266
|
+
mock.patch.object(sys, "stdout", io.StringIO()) as output:
|
|
267
|
+
if unconfirmed:
|
|
268
|
+
with self.assertRaises(SystemExit) as ended:
|
|
269
|
+
behavior.main()
|
|
270
|
+
self.assertEqual(ended.exception.code, 1)
|
|
271
|
+
else:
|
|
272
|
+
behavior.main()
|
|
273
|
+
rows = [json.loads(line) for line in (Path(temp) / "run-log.jsonl").read_text().splitlines()]
|
|
274
|
+
self.assertEqual(len(rows), 2 if unconfirmed else 3)
|
|
275
|
+
self.assertEqual(spawn.call_count, len(rows))
|
|
276
|
+
expected = "timeout_1s" + (";cleanup_unconfirmed:group_kill_permission_denied,stdio_timeout,wait_timeout"
|
|
277
|
+
if unconfirmed else "")
|
|
278
|
+
self.assertEqual(rows[1]["error"], expected)
|
|
279
|
+
self.assertNotIn("error", rows[0])
|
|
280
|
+
self.assertIn("complete", (Path(temp) / "synthetic.current.s1.txt").read_text())
|
|
281
|
+
self.assertFalse((Path(temp) / "synthetic.current.s2.txt").exists())
|
|
282
|
+
self.assertEqual((Path(temp) / "synthetic.current.s3.txt").exists(), not unconfirmed)
|
|
283
|
+
self.assertEqual("ABORT" in output.getvalue(), unconfirmed)
|
|
284
|
+
|
|
285
|
+
def test_replacement_invalidates_only_the_sample_actually_started(self):
|
|
286
|
+
cases = ("fresh_failure", "stale_failure", "overwrite", "quota_stop", "cache_resume", "report_only")
|
|
287
|
+
for scenario in cases:
|
|
288
|
+
with self.subTest(scenario=scenario), tempfile.TemporaryDirectory() as temp:
|
|
289
|
+
folder = Path(temp)
|
|
290
|
+
contract = folder / "contract.md"
|
|
291
|
+
contract.write_text("synthetic contract")
|
|
292
|
+
fx = {"id": "synthetic", "prompt": "x"}
|
|
293
|
+
current = folder / "synthetic.current.s1.txt"
|
|
294
|
+
candidate = folder / "synthetic.candidate.s1.txt"
|
|
295
|
+
previous = folder / "previous.current.s1.txt"
|
|
296
|
+
previous.write_text("earlier completed sample")
|
|
297
|
+
sig = "stale" if scenario == "stale_failure" else behavior.sample_sig(fx, "current", "")
|
|
298
|
+
current.write_text(f"# sig: {sig}\nOLD_CURRENT\n")
|
|
299
|
+
candidate.write_text(f"# sig: {behavior.sample_sig(fx, 'candidate', contract.read_text())}\nOLD_CANDIDATE\n")
|
|
300
|
+
old_current, old_candidate = current.read_bytes(), candidate.read_bytes()
|
|
301
|
+
failed = scenario.endswith("failure")
|
|
302
|
+
stream = json.dumps(success(result="NEW_CURRENT"))
|
|
303
|
+
if scenario == "quota_stop":
|
|
304
|
+
stream += "\n" + json.dumps({"type": "rate_limit_event", "rate_limit_info": {"utilization": 0.99}})
|
|
305
|
+
first = ([subprocess.TimeoutExpired(["synthetic"], 1), ("", None)]
|
|
306
|
+
if failed else [(stream, None)])
|
|
307
|
+
processes = [self.fake_process(first), self.fake_process([(json.dumps(success(result="NEW_CANDIDATE")), None)])]
|
|
308
|
+
argv = ["eval", "--both-arms", "--samples", "1", "--timeout", "1", "--out", temp,
|
|
309
|
+
"--contract", str(contract), "--no-judge"]
|
|
310
|
+
if scenario not in ("stale_failure", "cache_resume"):
|
|
311
|
+
argv.append("--fresh")
|
|
312
|
+
if scenario == "report_only":
|
|
313
|
+
argv.append("--report-only")
|
|
314
|
+
with mock.patch.object(sys, "argv", argv), \
|
|
315
|
+
mock.patch.object(behavior, "load_fixtures", return_value=[fx]), \
|
|
316
|
+
mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
|
|
317
|
+
mock.patch.object(behavior.os, "killpg", create=True), \
|
|
318
|
+
mock.patch.object(sys, "stdout", io.StringIO()):
|
|
319
|
+
behavior.main()
|
|
320
|
+
self.assertEqual(previous.read_text(), "earlier completed sample")
|
|
321
|
+
if failed:
|
|
322
|
+
self.assertFalse(current.exists(), "failed replacement left the old sample available")
|
|
323
|
+
logs = [json.loads(line) for line in (folder / "run-log.jsonl").read_text().splitlines()]
|
|
324
|
+
self.assertEqual(logs[0]["error"], "timeout_1s")
|
|
325
|
+
elif scenario in ("cache_resume", "report_only"):
|
|
326
|
+
self.assertEqual(current.read_bytes(), old_current)
|
|
327
|
+
self.assertEqual(candidate.read_bytes(), old_candidate)
|
|
328
|
+
self.assertEqual(spawn.call_count, 0)
|
|
329
|
+
else:
|
|
330
|
+
self.assertIn("NEW_CURRENT", current.read_text())
|
|
331
|
+
self.assertNotIn("OLD_CURRENT", current.read_text())
|
|
332
|
+
if scenario == "quota_stop":
|
|
333
|
+
self.assertEqual(candidate.read_bytes(), old_candidate)
|
|
334
|
+
self.assertEqual(spawn.call_count, 1)
|
|
335
|
+
else:
|
|
336
|
+
rows = [json.loads(line) for line in (folder / "judge-verdicts.jsonl").read_text().splitlines()]
|
|
337
|
+
self.assertEqual(rows[0]["status"], "missing-arm" if failed else "scaffold")
|
|
338
|
+
if scenario == "overwrite":
|
|
339
|
+
self.assertIn("NEW_CANDIDATE", candidate.read_text())
|
|
340
|
+
|
|
341
|
+
@unittest.skipUnless(os.name == "posix", "group-kill integration requires POSIX")
|
|
342
|
+
def test_cleanup_uncertainty_stops_judges_and_writes_incomplete_report(self):
|
|
343
|
+
verdict = {"delta": "tie", "reduced_capabilities": [], "added_capabilities": [],
|
|
344
|
+
"behavior_change": "same", "confidence": "high", "needs_human": False}
|
|
345
|
+
stream = (json.dumps(success(result=json.dumps(verdict))), None)
|
|
346
|
+
for unconfirmed in (False, True):
|
|
347
|
+
with self.subTest(cleanup_unconfirmed=unconfirmed), tempfile.TemporaryDirectory() as temp:
|
|
348
|
+
timeout = subprocess.TimeoutExpired(["synthetic"], 1)
|
|
349
|
+
failed = self.fake_process([timeout, timeout if unconfirmed else stream])
|
|
350
|
+
failed.wait.side_effect = timeout if unconfirmed else None
|
|
351
|
+
processes = [self.fake_process([stream]), failed, self.fake_process([stream])]
|
|
352
|
+
fixtures = [{"id": name} for name in ("first", "failed", "last")]
|
|
353
|
+
argv = ["eval", "--report-only", "--timeout", "1", "--out", temp]
|
|
354
|
+
with mock.patch.object(sys, "argv", argv), \
|
|
355
|
+
mock.patch.object(behavior, "load_fixtures", return_value=fixtures), \
|
|
356
|
+
mock.patch.object(behavior, "read_saved_response", return_value="complete"), \
|
|
357
|
+
mock.patch.object(behavior, "sample_sig", return_value="sig"), \
|
|
358
|
+
mock.patch.object(behavior, "_saved_sig", return_value="sig"), \
|
|
359
|
+
mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
|
|
360
|
+
mock.patch.object(behavior.os, "killpg", side_effect=PermissionError() if unconfirmed else None), \
|
|
361
|
+
mock.patch.object(sys, "stdout", io.StringIO()):
|
|
362
|
+
if unconfirmed:
|
|
363
|
+
with self.assertRaises(SystemExit) as ended:
|
|
364
|
+
behavior.main()
|
|
365
|
+
self.assertEqual(ended.exception.code, 1)
|
|
366
|
+
else:
|
|
367
|
+
behavior.main()
|
|
368
|
+
rows = [json.loads(line) for line in (Path(temp) / "judge-verdicts.jsonl").read_text().splitlines()]
|
|
369
|
+
self.assertEqual(spawn.call_count, 2 if unconfirmed else 3)
|
|
370
|
+
self.assertEqual(rows[0]["status"], "judged")
|
|
371
|
+
self.assertTrue(rows[1]["status"].startswith("judge-error:timeout_1s"))
|
|
372
|
+
self.assertEqual(rows[2]["status"], "judge-skipped" if unconfirmed else "judged")
|
|
373
|
+
if unconfirmed:
|
|
374
|
+
self.assertIn("cleanup_unconfirmed", rows[2]["note"])
|
|
375
|
+
self.assertEqual(len(rows), 3)
|
|
376
|
+
self.assertIn("INCOMPLETE", (Path(temp) / "capability-delta-report.md").read_text())
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
class GoldenRuntimeTests(unittest.TestCase):
|
|
380
|
+
def run_golden(self, events, forbidden=False):
|
|
381
|
+
with tempfile.TemporaryDirectory() as temp:
|
|
382
|
+
folder = Path(temp)
|
|
383
|
+
stub = folder / "claude"
|
|
384
|
+
stub.write_text(f"#!{sys.executable}\nimport sys\nsys.stdin.read()\nprint({events!r})\n")
|
|
385
|
+
stub.chmod(0o700)
|
|
386
|
+
traces = folder / "traces"
|
|
387
|
+
traces.mkdir()
|
|
388
|
+
(traces / "case.json").write_text(json.dumps({
|
|
389
|
+
"id": "synthetic", "hub_skill": "testing-strategy", "frozen_at_sha": "root",
|
|
390
|
+
"trigger_prompt": "synthetic", "assert": {
|
|
391
|
+
"must_invoke_skill": ["testing-strategy"],
|
|
392
|
+
"must_not_invoke_skill": ["testing-strategy"] if forbidden else [],
|
|
393
|
+
},
|
|
394
|
+
}))
|
|
395
|
+
report = folder / "report.json"
|
|
396
|
+
proc = subprocess.run(["ruby", str(HERE / "eval-golden-trace.rb"), str(ROOT),
|
|
397
|
+
"--traces", str(traces), "--json", str(report)],
|
|
398
|
+
env={**os.environ, "PATH": str(folder) + os.pathsep + os.environ["PATH"]},
|
|
399
|
+
capture_output=True, text=True, timeout=5)
|
|
400
|
+
self.assertIn(proc.returncode, (0, 3), proc.stderr)
|
|
401
|
+
return json.loads(report.read_text())["results"][0]
|
|
402
|
+
|
|
403
|
+
def test_only_complete_valid_stream_can_pass(self):
|
|
404
|
+
skill = {"type": "assistant", "message": {"content": [
|
|
405
|
+
{"type": "tool_use", "name": "Skill", "input": {"skill": "testing-strategy"}},
|
|
406
|
+
]}}
|
|
407
|
+
cases = {
|
|
408
|
+
"complete": ([skill, success()], "PASS"),
|
|
409
|
+
"missing": ([skill], "INCONCLUSIVE"),
|
|
410
|
+
"failed": ([skill, success(subtype="error_max_turns")], "INCONCLUSIVE"),
|
|
411
|
+
"error_flag": ([skill, success(is_error=True)], "INCONCLUSIVE"),
|
|
412
|
+
"duplicate": ([skill, success(), success()], "INCONCLUSIVE"),
|
|
413
|
+
"malformed_event": ([skill, [], success()], "INCONCLUSIVE"),
|
|
414
|
+
"malformed_content": ([skill, {"type": "assistant", "message": {"content": "bad"}}, success()], "INCONCLUSIVE"),
|
|
415
|
+
"permission_denial": ([skill, success(permission_denials=[{"tool": "Read"}])], "INCONCLUSIVE"),
|
|
416
|
+
}
|
|
417
|
+
for label, (events, expected) in cases.items():
|
|
418
|
+
with self.subTest(stream=label):
|
|
419
|
+
row = self.run_golden("\n".join(json.dumps(event) for event in events))
|
|
420
|
+
self.assertEqual(row["status"], expected)
|
|
421
|
+
if expected == "INCONCLUSIVE":
|
|
422
|
+
self.assertTrue(row["error"])
|
|
423
|
+
row = self.run_golden("\n".join(json.dumps(event) for event in (skill, success())), forbidden=True)
|
|
424
|
+
self.assertEqual(row["status"], "FAIL")
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
if __name__ == "__main__":
|
|
428
|
+
unittest.main()
|
|
@@ -139,28 +139,52 @@ git -C "$REPO_ROOT" show "$BASELINE_GATE_COMMIT:$GATE_PATH" > "$BASELINE_GATE" 2
|
|
|
139
139
|
exit 1
|
|
140
140
|
}
|
|
141
141
|
|
|
142
|
-
# EXPECTED DIVERGENCES
|
|
143
|
-
# rewrite — the repair for a round that merged with this gate red — so the owner
|
|
144
|
-
# change sits below their base and the row cites an owner the range does not
|
|
145
|
-
# touch. The candidate refuses that shape by design and deliberately offers no
|
|
146
|
-
# author-declared escape, so these four historical ranges diverge. The gate is
|
|
147
|
-
# diff-scoped and never re-judges landed history, so nothing operational depends
|
|
148
|
-
# on them; this differential is the only thing that replays them.
|
|
142
|
+
# EXPECTED DIVERGENCES, two named classes, both "newly refused".
|
|
149
143
|
#
|
|
150
|
-
#
|
|
151
|
-
#
|
|
152
|
-
#
|
|
153
|
-
#
|
|
154
|
-
#
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
144
|
+
# `impact_chain_row_vouches_for_unchanged_owner` (four points): landings that
|
|
145
|
+
# back-filled a ledger row by corrective rewrite — the repair for a round that
|
|
146
|
+
# merged with this gate red — so the owner change sits below their base and the
|
|
147
|
+
# row cites an owner the range does not touch. The candidate refuses that shape by
|
|
148
|
+
# design and deliberately offers no author-declared escape.
|
|
149
|
+
#
|
|
150
|
+
# `impact_chain_gate_missing` (six points): merges from before CI judged the
|
|
151
|
+
# branch head (register row on the checkout ref binding, 2026-08-25). The gate now
|
|
152
|
+
# expands a merge git rebuilds from its parents into the branch's own rounds, so a
|
|
153
|
+
# merge is judged exactly as its branch was; these six branches carry owner work
|
|
154
|
+
# outside the round that declares it (a row appended before the work, or work
|
|
155
|
+
# after the last append) and the baseline accepted them only through the
|
|
156
|
+
# collapsed merged view, which no longer exists as a distinct verdict. Three of
|
|
157
|
+
# the six are refused on their own branch head by the baseline gate too; the other
|
|
158
|
+
# three are promotions or syncs whose second parent is the integration branch,
|
|
159
|
+
# where the same shapes sit one merge deeper.
|
|
160
|
+
#
|
|
161
|
+
# The gate is diff-scoped and never re-judges landed history, so nothing
|
|
162
|
+
# operational depends on these points; this differential is the only thing that
|
|
163
|
+
# replays them. Each entry is constrained to ONE direction and ONE diagnostic. A
|
|
164
|
+
# blanket "any mismatch at this SHA is fine" would also swallow the opposite
|
|
165
|
+
# direction — a loosening — which is the failure this whole suite exists to
|
|
166
|
+
# catch. Entries are named individually, never matched by pattern, and an entry
|
|
167
|
+
# that stops diverging is reported as stale rather than tolerated.
|
|
168
|
+
EXPECTED_DIVERGENCES="
|
|
169
|
+
f03b1140f:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
170
|
+
93d09c563:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
171
|
+
9f233728a:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
172
|
+
046612652:refused:impact_chain_row_vouches_for_unchanged_owner
|
|
173
|
+
b9de13869:refused:impact_chain_gate_missing
|
|
174
|
+
8cea35e6d:refused:impact_chain_gate_missing
|
|
175
|
+
95f06b2e6:refused:impact_chain_gate_missing
|
|
176
|
+
c0561c74e:refused:impact_chain_gate_missing
|
|
177
|
+
fad480296:refused:impact_chain_gate_missing
|
|
178
|
+
90ec533e1:refused:impact_chain_gate_missing
|
|
179
|
+
"
|
|
180
|
+
expected_divergence() { # <full sha> <direction> <candidate output>; prints the matched token
|
|
181
|
+
local short="${1:0:9}" direction="$2" out="$3" entry sha dir token
|
|
182
|
+
case "$direction" in "newly refused") direction=refused ;; "newly accepted") direction=accepted ;; esac
|
|
183
|
+
for entry in $EXPECTED_DIVERGENCES; do
|
|
184
|
+
sha="${entry%%:*}"; token="${entry##*:}"; dir="${entry#*:}"; dir="${dir%%:*}"
|
|
185
|
+
[ "$sha" = "$short" ] || continue
|
|
186
|
+
[ "$dir" = "$direction" ] || return 1
|
|
187
|
+
case "$out" in *"$token"*) printf '%s' "$token"; return 0 ;; *) return 1 ;; esac
|
|
164
188
|
done
|
|
165
189
|
return 1
|
|
166
190
|
}
|
|
@@ -170,7 +194,7 @@ expected_divergence() { # <full sha> <direction> <candidate output>
|
|
|
170
194
|
# them, so the exemptions are stale and the run would pass while silently failing
|
|
171
195
|
# the staleness check it never reaches.
|
|
172
196
|
if cmp -s "$BASELINE_GATE" "$CANDIDATE_GATE"; then
|
|
173
|
-
if [ -n "$(printf '%s' "$
|
|
197
|
+
if [ -n "$(printf '%s' "$EXPECTED_DIVERGENCES" | tr -d '[:space:]')" ]; then
|
|
174
198
|
echo "FAIL: baseline and candidate are byte-identical, yet expected divergences are configured" >&2
|
|
175
199
|
echo " identical gates cannot diverge — the exemptions are stale and must be removed" >&2
|
|
176
200
|
exit 1
|
|
@@ -261,8 +285,8 @@ for point in $INTEGRATION_POINTS; do
|
|
|
261
285
|
direction="newly accepted"
|
|
262
286
|
fi
|
|
263
287
|
if [ -n "$direction" ]; then
|
|
264
|
-
if expected_divergence "$point" "$direction" "$candidate_out"; then
|
|
265
|
-
flag=" (expected divergence:
|
|
288
|
+
if matched_token="$(expected_divergence "$point" "$direction" "$candidate_out")"; then
|
|
289
|
+
flag=" (expected divergence: $matched_token, $direction)"
|
|
266
290
|
expected_seen=$((expected_seen + 1))
|
|
267
291
|
else
|
|
268
292
|
flag=" <== VERDICT MISMATCH: $direction"
|
|
@@ -412,7 +436,7 @@ if [ "$total_failures" -gt 0 ]; then
|
|
|
412
436
|
fi
|
|
413
437
|
# An expected divergence that stops diverging means the exemption is stale and
|
|
414
438
|
# should be removed, so it is reported rather than silently tolerated.
|
|
415
|
-
expected_total="$(for e in $
|
|
439
|
+
expected_total="$(for e in $EXPECTED_DIVERGENCES; do echo "$e"; done | wc -l | tr -d ' ')"
|
|
416
440
|
if [ "$expected_seen" != "$expected_total" ]; then
|
|
417
441
|
echo "FAIL: $expected_seen of $expected_total expected divergences actually diverged" >&2
|
|
418
442
|
echo " an exemption that no longer fires is stale — remove it" >&2
|
|
@@ -616,6 +616,64 @@ out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
|
|
|
616
616
|
check "a round with no accepted ledger at its own base refuses the whole chain, naming that round" \
|
|
617
617
|
'[ "$rc" = 1 ] && case "$out" in *"round $UNBOUND_MERGE"*"does not bind at its own base"*) true;; *) false;; esac'
|
|
618
618
|
|
|
619
|
+
# Evidence is read from the LANDING tree, not from the round's own checkout. A
|
|
620
|
+
# round that merged without its ledger is still a set of bytes some later review
|
|
621
|
+
# can freeze and inspect; a validator-accepted closeout for exactly that candidate,
|
|
622
|
+
# committed on the integration branch afterwards, binds it. A closeout for any
|
|
623
|
+
# other digest does not, wherever it sits — the candidate hash, not the file's
|
|
624
|
+
# location, is what makes a ledger evidence. The retro ledger lands as a round of
|
|
625
|
+
# its own (a merge whose only change is receipt-shaped evidence), because a
|
|
626
|
+
# direct commit on the integration branch is refused for its own reason.
|
|
627
|
+
D_BASE="$(git -C "$CHAIN" rev-parse HEAD^1)"
|
|
628
|
+
git -C "$CHAIN" checkout -q round-d
|
|
629
|
+
D_DIGEST="$(run_chain --base "$D_BASE" --print-candidate)"
|
|
630
|
+
git -C "$CHAIN" checkout -q dev
|
|
631
|
+
edit_retro_wrong() {
|
|
632
|
+
mkdir -p "$CHAIN/specs/d-retro-wrong/evidence"
|
|
633
|
+
write_closeout "$CHAIN/specs/d-retro-wrong/evidence/closeout.json" "$(printf '%064d' 7)"
|
|
634
|
+
}
|
|
635
|
+
land_round d-retro-wrong no-ledger edit_retro_wrong
|
|
636
|
+
out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
|
|
637
|
+
check "a later closeout for a different digest does not bind the unbound round" \
|
|
638
|
+
'[ "$rc" = 1 ] && case "$out" in *"round $UNBOUND_MERGE"*"does not bind at its own base"*) true;; *) false;; esac'
|
|
639
|
+
# The right digest in a closeout the validator rejects is still not evidence.
|
|
640
|
+
edit_retro_rejected() {
|
|
641
|
+
mkdir -p "$CHAIN/specs/d-retro-rejected/evidence"
|
|
642
|
+
python3 - "$CHAIN/specs/d-retro-rejected/evidence/closeout.json" "$D_DIGEST" <<'PY'
|
|
643
|
+
import json, sys
|
|
644
|
+
from pathlib import Path
|
|
645
|
+
Path(sys.argv[1]).write_text(json.dumps({"schema_version": 3, "closeout_state": "forged", "controller_receipts": [], "candidate_sha256": sys.argv[2]}))
|
|
646
|
+
PY
|
|
647
|
+
}
|
|
648
|
+
land_round d-retro-rejected no-ledger edit_retro_rejected
|
|
649
|
+
out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
|
|
650
|
+
check "a later closeout for the right digest that the validator rejects does not bind the unbound round" \
|
|
651
|
+
'[ "$rc" = 1 ] && case "$out" in *"round $UNBOUND_MERGE"*"does not bind at its own base"*) true;; *) false;; esac'
|
|
652
|
+
# Evidence in the landing tree must be COMMITTED there, exactly as before: an
|
|
653
|
+
# untracked closeout for the right digest is refused outright, not read from disk.
|
|
654
|
+
mkdir -p "$CHAIN/specs/d-retro/evidence"
|
|
655
|
+
write_closeout "$CHAIN/specs/d-retro/evidence/closeout.json" "$D_DIGEST"
|
|
656
|
+
out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
|
|
657
|
+
check "an uncommitted closeout in the landing tree is refused before any round is bound" \
|
|
658
|
+
'[ "$rc" != 0 ] && case "$out" in *"uncommitted changes"*) true;; *) false;; esac'
|
|
659
|
+
rm -r "$CHAIN/specs/d-retro"
|
|
660
|
+
edit_retro_right() {
|
|
661
|
+
mkdir -p "$CHAIN/specs/d-retro/evidence"
|
|
662
|
+
write_closeout "$CHAIN/specs/d-retro/evidence/closeout.json" "$D_DIGEST"
|
|
663
|
+
}
|
|
664
|
+
land_round d-retro no-ledger edit_retro_right
|
|
665
|
+
out="$(run_chain --base "$CHAIN_MAIN")"; rc=$?
|
|
666
|
+
check "a validator-accepted closeout for the round's own candidate, committed on the integration branch after the merge, binds it through the chain" \
|
|
667
|
+
'[ "$rc" = 0 ] && case "$out" in *"first-parent chain"*"specs/d-retro/evidence/closeout.json"*) true;; *) false;; esac'
|
|
668
|
+
# Later evidence changes where a ledger may be found, never what the round is:
|
|
669
|
+
# the unbound round's candidate at its own base hashes exactly as it did before
|
|
670
|
+
# any retro ledger existed, so its packet, base and excludes are untouched.
|
|
671
|
+
git -C "$CHAIN" checkout -q round-d
|
|
672
|
+
D_DIGEST_AFTER="$(run_chain --base "$D_BASE" --print-candidate)"
|
|
673
|
+
git -C "$CHAIN" checkout -q dev
|
|
674
|
+
check "the round's candidate hash is unchanged by evidence committed later in the landing tree" \
|
|
675
|
+
'[ "$D_DIGEST_AFTER" = "$D_DIGEST" ]'
|
|
676
|
+
|
|
619
677
|
probe_chain
|
|
620
678
|
printf 'pushed straight to the integration branch\n' >"$CHAIN/lane/direct.txt"
|
|
621
679
|
git -C "$CHAIN" add -A && git -C "$CHAIN" commit -qm "direct commit on dev"
|
|
@@ -827,10 +885,10 @@ check "the passing chain state still passes after the probes" \
|
|
|
827
885
|
# the real checkout it ships in, or a break in that path passes every test here.
|
|
828
886
|
REAL_ROOT="$(cd "$DIR/../../.." && pwd -P)"
|
|
829
887
|
real_out="$(env -u CCL_SKILL_BASE_REF python3 "$GATE" --repo-root "$REAL_ROOT" --base HEAD~1 --print-candidate 2>&1)"; real_rc=$?
|
|
830
|
-
#
|
|
831
|
-
# commit happened to touch: a candidate hash when a reviewed path moved,
|
|
832
|
-
# no-change signal when it did not
|
|
833
|
-
#
|
|
888
|
+
# Three outputs are legitimate and which one appears depends on what the parent
|
|
889
|
+
# commit happened to touch: a candidate hash when a bounded reviewed path moved,
|
|
890
|
+
# the no-change signal when it did not, or the packet-ceiling refusal when the
|
|
891
|
+
# parent diff is too large for one reviewer packet.
|
|
834
892
|
# Which answer is legitimate depends on the checkout, so decide that FIRST and then
|
|
835
893
|
# assert the one matching branch. Accepting every output would make this assertion a
|
|
836
894
|
# function of repository state; branching on the state keeps it an assertion about
|
|
@@ -842,7 +900,7 @@ if [ -n "$real_dirty" ]; then
|
|
|
842
900
|
'[ "$real_rc" != 0 ] && case "$real_out" in *"uncommitted changes"*) true;; *) false;; esac'
|
|
843
901
|
else
|
|
844
902
|
check "the gate answers for the clean real checkout it ships in, whichever legitimate answer applies" \
|
|
845
|
-
'[ "$real_rc" = 0 ] && { case "$real_out" in [0-9a-f]*) [ ${#real_out} = 64 ];; *) false;; esac || case "$real_out" in *review_ledger_binding_no_change*) true;; *) false;; esac; }'
|
|
903
|
+
'{ [ "$real_rc" = 0 ] && { case "$real_out" in [0-9a-f]*) [ ${#real_out} = 64 ];; *) false;; esac || case "$real_out" in *review_ledger_binding_no_change*) true;; *) false;; esac; }; } || { [ "$real_rc" = 1 ] && case "$real_out" in *"cannot freeze the candidate packet"*"review packet exceeds 200000 bytes"*) true;; *) false;; esac; }'
|
|
846
904
|
fi
|
|
847
905
|
|
|
848
906
|
if [ "$fails" -gt 0 ]; then
|