@ccoalm/ccl-skills 0.15.0 → 0.15.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +80 -4
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +16 -4
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +13 -2
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +37 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +5 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +77 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_packet_mcp.py +98 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_cli_review.py +48 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +237 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +165 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_kimi_packet_mcp.py +143 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +572 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +7 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +40 -83
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +44 -37
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +20 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/refactoring-discipline.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +8 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +17 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/resume-paused-delivery.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +31 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +74 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +103 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +83 -48
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +80 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +74 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_runtime.py +428 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +190 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +106 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -1
- package/dist/assets/release.json +87 -67
- package/dist/claude-adapter.js +14 -7
- package/dist/codex-host.d.ts +2 -4
- package/dist/codex-host.js +40 -19
- package/dist/host-probe.d.ts +27 -0
- package/dist/host-probe.js +51 -0
- package/dist/opencode-adapter.js +24 -19
- package/dist/operations.js +34 -8
- package/dist/unified.d.ts +1 -1
- package/dist/unified.js +18 -9
- package/package.json +1 -1
|
@@ -0,0 +1,428 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Offline regression cases for evaluator deadlines and terminal evidence."""
|
|
3
|
+
import importlib.util
|
|
4
|
+
import io
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
import runpy
|
|
9
|
+
import signal
|
|
10
|
+
import subprocess
|
|
11
|
+
import sys
|
|
12
|
+
import tempfile
|
|
13
|
+
import time
|
|
14
|
+
import unittest
|
|
15
|
+
from unittest import mock
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
HERE = Path(__file__).resolve().parent
|
|
19
|
+
ROOT = HERE.parents[2]
|
|
20
|
+
spec = importlib.util.spec_from_file_location("behavior_eval", HERE / "skill-behavior-eval.py")
|
|
21
|
+
behavior = importlib.util.module_from_spec(spec)
|
|
22
|
+
spec.loader.exec_module(behavior)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def success(**overrides):
|
|
26
|
+
return {"type": "result", "subtype": "success", "is_error": False,
|
|
27
|
+
"result": "complete", **overrides}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class BehaviorRuntimeTests(unittest.TestCase):
|
|
31
|
+
def fake_process(self, communicate):
|
|
32
|
+
process = mock.Mock(pid=12345, returncode=0)
|
|
33
|
+
process.stdin = io.StringIO()
|
|
34
|
+
process.stdout = io.StringIO()
|
|
35
|
+
process.communicate.side_effect = communicate
|
|
36
|
+
return process
|
|
37
|
+
|
|
38
|
+
def invoke(self, source, prompt="x", timeout=0.15):
|
|
39
|
+
started = time.monotonic()
|
|
40
|
+
result = behavior._headless_claude([sys.executable, "-c", source], prompt, timeout)
|
|
41
|
+
return result, time.monotonic() - started
|
|
42
|
+
|
|
43
|
+
def test_complete_stream_and_teardown_exit(self):
|
|
44
|
+
for code in (0, 7):
|
|
45
|
+
with self.subTest(code=code):
|
|
46
|
+
result, _ = self.invoke(f"import sys; print({json.dumps(success())!r}); sys.exit({code})", timeout=2)
|
|
47
|
+
self.assertEqual(result[:2], ("complete", None))
|
|
48
|
+
|
|
49
|
+
def test_deadline_covers_all_io_and_process_wait(self):
|
|
50
|
+
cases = {
|
|
51
|
+
"stdin_write": ("import time; time.sleep(1.5)", "x" * 1_000_000),
|
|
52
|
+
"stdout_closed_process_live": ("import os,time; os.close(1); time.sleep(1.5)", "x"),
|
|
53
|
+
"stdout_open": ("import time; time.sleep(1.5)", "x"),
|
|
54
|
+
"terminal_before_process_exit": (f"import os,time; print({json.dumps(success())!r}, flush=True); os.close(1); time.sleep(1.5)", "x"),
|
|
55
|
+
}
|
|
56
|
+
for label, (source, prompt) in cases.items():
|
|
57
|
+
with self.subTest(boundary=label):
|
|
58
|
+
result, elapsed = self.invoke(source, prompt)
|
|
59
|
+
self.assertEqual(result[1], "timeout_0.15s")
|
|
60
|
+
self.assertLess(elapsed, 1.2)
|
|
61
|
+
|
|
62
|
+
@unittest.skipUnless(os.name == "posix", "process groups require POSIX")
|
|
63
|
+
def test_timeout_stops_descendant_with_inherited_stdout(self):
|
|
64
|
+
with tempfile.TemporaryDirectory() as temp:
|
|
65
|
+
marker = Path(temp) / "descendant-finished"
|
|
66
|
+
ready = Path(temp) / "descendant-ready"
|
|
67
|
+
descendant = f"import time; from pathlib import Path; Path({str(ready)!r}).touch(); time.sleep(1.2); Path({str(marker)!r}).touch()"
|
|
68
|
+
source = f"import subprocess,sys; subprocess.Popen([sys.executable,'-c',{descendant!r}])"
|
|
69
|
+
result, elapsed = self.invoke(source, timeout=0.5)
|
|
70
|
+
self.assertEqual(result[1], "timeout_0.5s")
|
|
71
|
+
self.assertLess(elapsed, 1.5)
|
|
72
|
+
self.assertTrue(ready.exists(), "fixture never reached the descendant path")
|
|
73
|
+
time.sleep(1.3)
|
|
74
|
+
self.assertFalse(marker.exists(), "descendant continued after timeout")
|
|
75
|
+
|
|
76
|
+
@unittest.skipUnless(os.name == "posix", "terminal process groups require POSIX")
|
|
77
|
+
def test_cli_signals_stop_detached_child(self):
|
|
78
|
+
for cancel in (signal.SIGINT, signal.SIGTERM):
|
|
79
|
+
with self.subTest(signal=cancel), tempfile.TemporaryDirectory() as temp:
|
|
80
|
+
folder = Path(temp)
|
|
81
|
+
ready, marker, calls = folder / "ready", folder / "late-work", folder / "calls"
|
|
82
|
+
stub = folder / "claude"
|
|
83
|
+
stub.write_text(f"#!{sys.executable}\nimport os,sys,time; from pathlib import Path\n"
|
|
84
|
+
"sys.stdin.read()\n"
|
|
85
|
+
f"with Path({str(calls)!r}).open('a') as log: log.write('started\\n')\n"
|
|
86
|
+
f"Path({str(ready)!r}).write_text(str(os.getpid()))\n"
|
|
87
|
+
f"time.sleep(1.4); Path({str(marker)!r}).touch()\n")
|
|
88
|
+
stub.chmod(0o700)
|
|
89
|
+
fixtures, output = folder / "fixtures.jsonl", folder / "output"
|
|
90
|
+
fixtures.write_text(json.dumps({"id": "synthetic", "prompt": "x"}) + "\n")
|
|
91
|
+
process = subprocess.Popen(
|
|
92
|
+
[sys.executable, str(HERE / "skill-behavior-eval.py"), "--fixtures", str(fixtures),
|
|
93
|
+
"--out", str(output), "--samples", "2", "--timeout", "10", "--no-judge"],
|
|
94
|
+
env={**os.environ, "HOME": temp, "PATH": temp + os.pathsep + os.environ["PATH"]},
|
|
95
|
+
stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, start_new_session=True)
|
|
96
|
+
child_pid = None
|
|
97
|
+
try:
|
|
98
|
+
deadline = time.monotonic() + 3
|
|
99
|
+
while (not ready.exists() or not ready.read_text()) and time.monotonic() < deadline:
|
|
100
|
+
time.sleep(0.01)
|
|
101
|
+
self.assertTrue(ready.exists(), "fixture never reached the child")
|
|
102
|
+
child_pid = int(ready.read_text())
|
|
103
|
+
os.killpg(process.pid, cancel)
|
|
104
|
+
_, stderr = process.communicate(timeout=3)
|
|
105
|
+
self.assertNotEqual(process.returncode, 0)
|
|
106
|
+
self.assertFalse(marker.exists(), "child performed work after cancellation")
|
|
107
|
+
with self.assertRaises(ProcessLookupError, msg="child survived evaluator cancellation"):
|
|
108
|
+
os.kill(child_pid, 0)
|
|
109
|
+
self.assertEqual(calls.read_text().splitlines(), ["started"])
|
|
110
|
+
self.assertEqual(list(output.glob("*.s*.txt")), [])
|
|
111
|
+
if cancel == signal.SIGINT:
|
|
112
|
+
self.assertIn("KeyboardInterrupt", stderr)
|
|
113
|
+
else:
|
|
114
|
+
self.assertEqual(process.returncode, 128 + signal.SIGTERM)
|
|
115
|
+
finally:
|
|
116
|
+
if process.poll() is None:
|
|
117
|
+
os.killpg(process.pid, signal.SIGKILL)
|
|
118
|
+
process.wait(timeout=2)
|
|
119
|
+
if child_pid is not None:
|
|
120
|
+
try:
|
|
121
|
+
os.kill(child_pid, 0)
|
|
122
|
+
os.killpg(child_pid, signal.SIGKILL)
|
|
123
|
+
except ProcessLookupError:
|
|
124
|
+
pass
|
|
125
|
+
process.stdout.close()
|
|
126
|
+
process.stderr.close()
|
|
127
|
+
|
|
128
|
+
@unittest.skipUnless(os.name == "posix", "process group refusal requires POSIX")
|
|
129
|
+
def test_cancellation_reports_cleanup_refusal_and_reraises(self):
|
|
130
|
+
for interrupted in (KeyboardInterrupt(), SystemExit(128 + signal.SIGTERM),
|
|
131
|
+
RuntimeError("custom signal handler"), OSError("pipe failure"),
|
|
132
|
+
BaseException("custom cancellation")):
|
|
133
|
+
with self.subTest(exception=type(interrupted).__name__):
|
|
134
|
+
process = self.fake_process([interrupted, ("", None)])
|
|
135
|
+
with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
|
|
136
|
+
mock.patch.object(behavior.os, "killpg", side_effect=PermissionError()) as killpg, \
|
|
137
|
+
mock.patch.object(sys, "stderr", io.StringIO()) as stderr:
|
|
138
|
+
with self.assertRaises(type(interrupted)) as raised:
|
|
139
|
+
behavior._headless_claude(["synthetic"], "x", 1)
|
|
140
|
+
self.assertIs(raised.exception, interrupted)
|
|
141
|
+
self.assertIn("cleanup_unconfirmed:group_kill_permission_denied", stderr.getvalue())
|
|
142
|
+
killpg.assert_called_once_with(process.pid, signal.SIGKILL)
|
|
143
|
+
self.assertEqual(process.communicate.call_args_list,
|
|
144
|
+
[mock.call(input="x", timeout=1), mock.call(timeout=1)])
|
|
145
|
+
|
|
146
|
+
@unittest.skipUnless(os.name == "posix", "process group cleanup requires POSIX")
|
|
147
|
+
def test_cleanup_io_errors_preserve_original_failure_and_remain_bounded(self):
|
|
148
|
+
for stopped in (subprocess.TimeoutExpired(["synthetic"], 1), OSError("original pipe failure")):
|
|
149
|
+
with self.subTest(exception=type(stopped).__name__):
|
|
150
|
+
process = self.fake_process([stopped, OSError("drain failure")])
|
|
151
|
+
process.wait.side_effect = OSError("reap failure")
|
|
152
|
+
with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
|
|
153
|
+
mock.patch.object(behavior.os, "killpg") as killpg, \
|
|
154
|
+
mock.patch.object(sys, "stderr", io.StringIO()) as stderr:
|
|
155
|
+
if isinstance(stopped, subprocess.TimeoutExpired):
|
|
156
|
+
result = behavior._headless_claude(["synthetic"], "x", 1)
|
|
157
|
+
self.assertEqual(result, (None, "timeout_1s;cleanup_unconfirmed:stdio_error,wait_error", None, []))
|
|
158
|
+
else:
|
|
159
|
+
with self.assertRaises(OSError) as raised:
|
|
160
|
+
behavior._headless_claude(["synthetic"], "x", 1)
|
|
161
|
+
self.assertIs(raised.exception, stopped)
|
|
162
|
+
self.assertIn("cleanup_unconfirmed:stdio_error,wait_error", stderr.getvalue())
|
|
163
|
+
killpg.assert_called_once_with(process.pid, signal.SIGKILL)
|
|
164
|
+
process.kill.assert_called_once_with()
|
|
165
|
+
process.wait.assert_called_once_with(timeout=1)
|
|
166
|
+
self.assertEqual(process.communicate.call_args_list,
|
|
167
|
+
[mock.call(input="x", timeout=1), mock.call(timeout=1)])
|
|
168
|
+
self.assertTrue(process.stdin.closed)
|
|
169
|
+
self.assertTrue(process.stdout.closed)
|
|
170
|
+
|
|
171
|
+
def test_cli_sigterm_handler_is_scoped_and_preserves_existing_handlers(self):
|
|
172
|
+
script = str(HERE / "skill-behavior-eval.py")
|
|
173
|
+
with mock.patch.object(signal, "signal") as install:
|
|
174
|
+
runpy.run_path(script, run_name="imported_fixture")
|
|
175
|
+
install.assert_not_called()
|
|
176
|
+
for previous in (signal.SIG_DFL, signal.SIG_IGN, lambda *_: None):
|
|
177
|
+
with self.subTest(previous=previous), \
|
|
178
|
+
mock.patch.object(signal, "getsignal", return_value=previous), \
|
|
179
|
+
mock.patch.object(signal, "signal") as install, \
|
|
180
|
+
mock.patch.object(sys, "argv", [script, "--help"]), \
|
|
181
|
+
mock.patch.object(sys, "stdout", io.StringIO()):
|
|
182
|
+
with self.assertRaises(SystemExit) as ended:
|
|
183
|
+
runpy.run_path(script, run_name="__main__")
|
|
184
|
+
self.assertEqual(ended.exception.code, 0)
|
|
185
|
+
if previous == signal.SIG_DFL:
|
|
186
|
+
self.assertEqual(install.call_count, 2)
|
|
187
|
+
self.assertEqual(install.call_args, mock.call(signal.SIGTERM, previous))
|
|
188
|
+
with self.assertRaises(SystemExit) as ended:
|
|
189
|
+
install.call_args_list[0].args[1](signal.SIGTERM, None)
|
|
190
|
+
self.assertEqual(ended.exception.code, 128 + signal.SIGTERM)
|
|
191
|
+
else:
|
|
192
|
+
install.assert_not_called()
|
|
193
|
+
|
|
194
|
+
def test_invalid_terminal_never_becomes_a_sample(self):
|
|
195
|
+
streams = {
|
|
196
|
+
"missing": [{"type": "assistant", "message": {"content": [{"type": "text", "text": "partial"}]}}],
|
|
197
|
+
"failed": [success(subtype="error_max_turns")],
|
|
198
|
+
"error_flag": [success(is_error=True)],
|
|
199
|
+
"duplicate": [success(), success()],
|
|
200
|
+
"permission_denial": [success(permission_denials=[{"tool": "Read"}])],
|
|
201
|
+
"api_error": [success(api_error_status=429)],
|
|
202
|
+
"malformed_result": [success(result={"invalid": "text"})],
|
|
203
|
+
"empty_result": [success(result="")],
|
|
204
|
+
}
|
|
205
|
+
for label, events in streams.items():
|
|
206
|
+
with self.subTest(stream=label):
|
|
207
|
+
stream = "\n".join(json.dumps(event) for event in events)
|
|
208
|
+
result, _ = self.invoke(f"print({stream!r})", timeout=2)
|
|
209
|
+
self.assertIsNone(result[0])
|
|
210
|
+
self.assertIsNotNone(result[1])
|
|
211
|
+
|
|
212
|
+
def test_timeout_cleanup_failures_stay_explicit_and_bounded(self):
|
|
213
|
+
timeout = lambda: subprocess.TimeoutExpired(["synthetic"], 0.15)
|
|
214
|
+
complete = (json.dumps(success()), None)
|
|
215
|
+
cases = (
|
|
216
|
+
("clean", "posix", None, None, None, False, ""),
|
|
217
|
+
("already_gone", "posix", ProcessLookupError(), None, None, False, ""),
|
|
218
|
+
("group_denied", "posix", PermissionError(), None, None, False,
|
|
219
|
+
"group_kill_permission_denied"),
|
|
220
|
+
("direct_only", "nt", None, None, None, False,
|
|
221
|
+
"descendant_cleanup_unsupported"),
|
|
222
|
+
("direct_gone", "nt", None, ProcessLookupError(), None, False,
|
|
223
|
+
"descendant_cleanup_unsupported"),
|
|
224
|
+
("direct_denied", "nt", None, PermissionError(), None, False,
|
|
225
|
+
"descendant_cleanup_unsupported,process_kill_permission_denied"),
|
|
226
|
+
("pipe_still_open", "posix", None, None, None, True, "stdio_timeout"),
|
|
227
|
+
("fallback_denied", "posix", None, PermissionError(), None, True,
|
|
228
|
+
"stdio_timeout,process_kill_permission_denied"),
|
|
229
|
+
("wait_timeout", "posix", None, None, timeout(), True,
|
|
230
|
+
"stdio_timeout,wait_timeout"),
|
|
231
|
+
)
|
|
232
|
+
for label, platform, group_error, kill_error, wait_error, drain_timeout, detail in cases:
|
|
233
|
+
with self.subTest(boundary=label):
|
|
234
|
+
process = self.fake_process([timeout(), timeout() if drain_timeout else complete])
|
|
235
|
+
process.kill.side_effect = kill_error
|
|
236
|
+
process.wait.side_effect = wait_error
|
|
237
|
+
with mock.patch.object(behavior.subprocess, "Popen", return_value=process), \
|
|
238
|
+
mock.patch.object(behavior.os, "name", platform), \
|
|
239
|
+
mock.patch.object(behavior.os, "killpg", side_effect=group_error, create=True):
|
|
240
|
+
result = behavior._headless_claude(["synthetic"], "x", 0.15)
|
|
241
|
+
error = "timeout_0.15s" + (";cleanup_unconfirmed:" + detail if detail else "")
|
|
242
|
+
self.assertEqual(result, (None, error, None, []))
|
|
243
|
+
self.assertEqual(process.communicate.call_args_list,
|
|
244
|
+
[mock.call(input="x", timeout=0.15), mock.call(timeout=1)])
|
|
245
|
+
if drain_timeout:
|
|
246
|
+
process.wait.assert_called_once_with(timeout=1)
|
|
247
|
+
else:
|
|
248
|
+
process.wait.assert_not_called()
|
|
249
|
+
self.assertTrue(process.stdin.closed)
|
|
250
|
+
self.assertTrue(process.stdout.closed)
|
|
251
|
+
|
|
252
|
+
@unittest.skipUnless(os.name == "posix", "group-kill integration requires POSIX")
|
|
253
|
+
def test_cleanup_uncertainty_stops_new_samples_but_preserves_results(self):
|
|
254
|
+
for unconfirmed in (False, True):
|
|
255
|
+
with self.subTest(cleanup_unconfirmed=unconfirmed), tempfile.TemporaryDirectory() as temp:
|
|
256
|
+
timeout = subprocess.TimeoutExpired(["synthetic"], 1)
|
|
257
|
+
stream = (json.dumps(success()), None)
|
|
258
|
+
failed = self.fake_process([timeout, timeout if unconfirmed else stream])
|
|
259
|
+
failed.wait.side_effect = timeout if unconfirmed else None
|
|
260
|
+
processes = [self.fake_process([stream]), failed, self.fake_process([stream])]
|
|
261
|
+
argv = ["eval", "--samples", "3", "--timeout", "1", "--out", temp]
|
|
262
|
+
with mock.patch.object(sys, "argv", argv), \
|
|
263
|
+
mock.patch.object(behavior, "load_fixtures", return_value=[{"id": "synthetic", "prompt": "x"}]), \
|
|
264
|
+
mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
|
|
265
|
+
mock.patch.object(behavior.os, "killpg", side_effect=PermissionError() if unconfirmed else None), \
|
|
266
|
+
mock.patch.object(sys, "stdout", io.StringIO()) as output:
|
|
267
|
+
if unconfirmed:
|
|
268
|
+
with self.assertRaises(SystemExit) as ended:
|
|
269
|
+
behavior.main()
|
|
270
|
+
self.assertEqual(ended.exception.code, 1)
|
|
271
|
+
else:
|
|
272
|
+
behavior.main()
|
|
273
|
+
rows = [json.loads(line) for line in (Path(temp) / "run-log.jsonl").read_text().splitlines()]
|
|
274
|
+
self.assertEqual(len(rows), 2 if unconfirmed else 3)
|
|
275
|
+
self.assertEqual(spawn.call_count, len(rows))
|
|
276
|
+
expected = "timeout_1s" + (";cleanup_unconfirmed:group_kill_permission_denied,stdio_timeout,wait_timeout"
|
|
277
|
+
if unconfirmed else "")
|
|
278
|
+
self.assertEqual(rows[1]["error"], expected)
|
|
279
|
+
self.assertNotIn("error", rows[0])
|
|
280
|
+
self.assertIn("complete", (Path(temp) / "synthetic.current.s1.txt").read_text())
|
|
281
|
+
self.assertFalse((Path(temp) / "synthetic.current.s2.txt").exists())
|
|
282
|
+
self.assertEqual((Path(temp) / "synthetic.current.s3.txt").exists(), not unconfirmed)
|
|
283
|
+
self.assertEqual("ABORT" in output.getvalue(), unconfirmed)
|
|
284
|
+
|
|
285
|
+
def test_replacement_invalidates_only_the_sample_actually_started(self):
|
|
286
|
+
cases = ("fresh_failure", "stale_failure", "overwrite", "quota_stop", "cache_resume", "report_only")
|
|
287
|
+
for scenario in cases:
|
|
288
|
+
with self.subTest(scenario=scenario), tempfile.TemporaryDirectory() as temp:
|
|
289
|
+
folder = Path(temp)
|
|
290
|
+
contract = folder / "contract.md"
|
|
291
|
+
contract.write_text("synthetic contract")
|
|
292
|
+
fx = {"id": "synthetic", "prompt": "x"}
|
|
293
|
+
current = folder / "synthetic.current.s1.txt"
|
|
294
|
+
candidate = folder / "synthetic.candidate.s1.txt"
|
|
295
|
+
previous = folder / "previous.current.s1.txt"
|
|
296
|
+
previous.write_text("earlier completed sample")
|
|
297
|
+
sig = "stale" if scenario == "stale_failure" else behavior.sample_sig(fx, "current", "")
|
|
298
|
+
current.write_text(f"# sig: {sig}\nOLD_CURRENT\n")
|
|
299
|
+
candidate.write_text(f"# sig: {behavior.sample_sig(fx, 'candidate', contract.read_text())}\nOLD_CANDIDATE\n")
|
|
300
|
+
old_current, old_candidate = current.read_bytes(), candidate.read_bytes()
|
|
301
|
+
failed = scenario.endswith("failure")
|
|
302
|
+
stream = json.dumps(success(result="NEW_CURRENT"))
|
|
303
|
+
if scenario == "quota_stop":
|
|
304
|
+
stream += "\n" + json.dumps({"type": "rate_limit_event", "rate_limit_info": {"utilization": 0.99}})
|
|
305
|
+
first = ([subprocess.TimeoutExpired(["synthetic"], 1), ("", None)]
|
|
306
|
+
if failed else [(stream, None)])
|
|
307
|
+
processes = [self.fake_process(first), self.fake_process([(json.dumps(success(result="NEW_CANDIDATE")), None)])]
|
|
308
|
+
argv = ["eval", "--both-arms", "--samples", "1", "--timeout", "1", "--out", temp,
|
|
309
|
+
"--contract", str(contract), "--no-judge"]
|
|
310
|
+
if scenario not in ("stale_failure", "cache_resume"):
|
|
311
|
+
argv.append("--fresh")
|
|
312
|
+
if scenario == "report_only":
|
|
313
|
+
argv.append("--report-only")
|
|
314
|
+
with mock.patch.object(sys, "argv", argv), \
|
|
315
|
+
mock.patch.object(behavior, "load_fixtures", return_value=[fx]), \
|
|
316
|
+
mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
|
|
317
|
+
mock.patch.object(behavior.os, "killpg", create=True), \
|
|
318
|
+
mock.patch.object(sys, "stdout", io.StringIO()):
|
|
319
|
+
behavior.main()
|
|
320
|
+
self.assertEqual(previous.read_text(), "earlier completed sample")
|
|
321
|
+
if failed:
|
|
322
|
+
self.assertFalse(current.exists(), "failed replacement left the old sample available")
|
|
323
|
+
logs = [json.loads(line) for line in (folder / "run-log.jsonl").read_text().splitlines()]
|
|
324
|
+
self.assertEqual(logs[0]["error"], "timeout_1s")
|
|
325
|
+
elif scenario in ("cache_resume", "report_only"):
|
|
326
|
+
self.assertEqual(current.read_bytes(), old_current)
|
|
327
|
+
self.assertEqual(candidate.read_bytes(), old_candidate)
|
|
328
|
+
self.assertEqual(spawn.call_count, 0)
|
|
329
|
+
else:
|
|
330
|
+
self.assertIn("NEW_CURRENT", current.read_text())
|
|
331
|
+
self.assertNotIn("OLD_CURRENT", current.read_text())
|
|
332
|
+
if scenario == "quota_stop":
|
|
333
|
+
self.assertEqual(candidate.read_bytes(), old_candidate)
|
|
334
|
+
self.assertEqual(spawn.call_count, 1)
|
|
335
|
+
else:
|
|
336
|
+
rows = [json.loads(line) for line in (folder / "judge-verdicts.jsonl").read_text().splitlines()]
|
|
337
|
+
self.assertEqual(rows[0]["status"], "missing-arm" if failed else "scaffold")
|
|
338
|
+
if scenario == "overwrite":
|
|
339
|
+
self.assertIn("NEW_CANDIDATE", candidate.read_text())
|
|
340
|
+
|
|
341
|
+
@unittest.skipUnless(os.name == "posix", "group-kill integration requires POSIX")
|
|
342
|
+
def test_cleanup_uncertainty_stops_judges_and_writes_incomplete_report(self):
|
|
343
|
+
verdict = {"delta": "tie", "reduced_capabilities": [], "added_capabilities": [],
|
|
344
|
+
"behavior_change": "same", "confidence": "high", "needs_human": False}
|
|
345
|
+
stream = (json.dumps(success(result=json.dumps(verdict))), None)
|
|
346
|
+
for unconfirmed in (False, True):
|
|
347
|
+
with self.subTest(cleanup_unconfirmed=unconfirmed), tempfile.TemporaryDirectory() as temp:
|
|
348
|
+
timeout = subprocess.TimeoutExpired(["synthetic"], 1)
|
|
349
|
+
failed = self.fake_process([timeout, timeout if unconfirmed else stream])
|
|
350
|
+
failed.wait.side_effect = timeout if unconfirmed else None
|
|
351
|
+
processes = [self.fake_process([stream]), failed, self.fake_process([stream])]
|
|
352
|
+
fixtures = [{"id": name} for name in ("first", "failed", "last")]
|
|
353
|
+
argv = ["eval", "--report-only", "--timeout", "1", "--out", temp]
|
|
354
|
+
with mock.patch.object(sys, "argv", argv), \
|
|
355
|
+
mock.patch.object(behavior, "load_fixtures", return_value=fixtures), \
|
|
356
|
+
mock.patch.object(behavior, "read_saved_response", return_value="complete"), \
|
|
357
|
+
mock.patch.object(behavior, "sample_sig", return_value="sig"), \
|
|
358
|
+
mock.patch.object(behavior, "_saved_sig", return_value="sig"), \
|
|
359
|
+
mock.patch.object(behavior.subprocess, "Popen", side_effect=processes) as spawn, \
|
|
360
|
+
mock.patch.object(behavior.os, "killpg", side_effect=PermissionError() if unconfirmed else None), \
|
|
361
|
+
mock.patch.object(sys, "stdout", io.StringIO()):
|
|
362
|
+
if unconfirmed:
|
|
363
|
+
with self.assertRaises(SystemExit) as ended:
|
|
364
|
+
behavior.main()
|
|
365
|
+
self.assertEqual(ended.exception.code, 1)
|
|
366
|
+
else:
|
|
367
|
+
behavior.main()
|
|
368
|
+
rows = [json.loads(line) for line in (Path(temp) / "judge-verdicts.jsonl").read_text().splitlines()]
|
|
369
|
+
self.assertEqual(spawn.call_count, 2 if unconfirmed else 3)
|
|
370
|
+
self.assertEqual(rows[0]["status"], "judged")
|
|
371
|
+
self.assertTrue(rows[1]["status"].startswith("judge-error:timeout_1s"))
|
|
372
|
+
self.assertEqual(rows[2]["status"], "judge-skipped" if unconfirmed else "judged")
|
|
373
|
+
if unconfirmed:
|
|
374
|
+
self.assertIn("cleanup_unconfirmed", rows[2]["note"])
|
|
375
|
+
self.assertEqual(len(rows), 3)
|
|
376
|
+
self.assertIn("INCOMPLETE", (Path(temp) / "capability-delta-report.md").read_text())
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
class GoldenRuntimeTests(unittest.TestCase):
|
|
380
|
+
def run_golden(self, events, forbidden=False):
|
|
381
|
+
with tempfile.TemporaryDirectory() as temp:
|
|
382
|
+
folder = Path(temp)
|
|
383
|
+
stub = folder / "claude"
|
|
384
|
+
stub.write_text(f"#!{sys.executable}\nimport sys\nsys.stdin.read()\nprint({events!r})\n")
|
|
385
|
+
stub.chmod(0o700)
|
|
386
|
+
traces = folder / "traces"
|
|
387
|
+
traces.mkdir()
|
|
388
|
+
(traces / "case.json").write_text(json.dumps({
|
|
389
|
+
"id": "synthetic", "hub_skill": "testing-strategy", "frozen_at_sha": "root",
|
|
390
|
+
"trigger_prompt": "synthetic", "assert": {
|
|
391
|
+
"must_invoke_skill": ["testing-strategy"],
|
|
392
|
+
"must_not_invoke_skill": ["testing-strategy"] if forbidden else [],
|
|
393
|
+
},
|
|
394
|
+
}))
|
|
395
|
+
report = folder / "report.json"
|
|
396
|
+
proc = subprocess.run(["ruby", str(HERE / "eval-golden-trace.rb"), str(ROOT),
|
|
397
|
+
"--traces", str(traces), "--json", str(report)],
|
|
398
|
+
env={**os.environ, "PATH": str(folder) + os.pathsep + os.environ["PATH"]},
|
|
399
|
+
capture_output=True, text=True, timeout=5)
|
|
400
|
+
self.assertIn(proc.returncode, (0, 3), proc.stderr)
|
|
401
|
+
return json.loads(report.read_text())["results"][0]
|
|
402
|
+
|
|
403
|
+
def test_only_complete_valid_stream_can_pass(self):
|
|
404
|
+
skill = {"type": "assistant", "message": {"content": [
|
|
405
|
+
{"type": "tool_use", "name": "Skill", "input": {"skill": "testing-strategy"}},
|
|
406
|
+
]}}
|
|
407
|
+
cases = {
|
|
408
|
+
"complete": ([skill, success()], "PASS"),
|
|
409
|
+
"missing": ([skill], "INCONCLUSIVE"),
|
|
410
|
+
"failed": ([skill, success(subtype="error_max_turns")], "INCONCLUSIVE"),
|
|
411
|
+
"error_flag": ([skill, success(is_error=True)], "INCONCLUSIVE"),
|
|
412
|
+
"duplicate": ([skill, success(), success()], "INCONCLUSIVE"),
|
|
413
|
+
"malformed_event": ([skill, [], success()], "INCONCLUSIVE"),
|
|
414
|
+
"malformed_content": ([skill, {"type": "assistant", "message": {"content": "bad"}}, success()], "INCONCLUSIVE"),
|
|
415
|
+
"permission_denial": ([skill, success(permission_denials=[{"tool": "Read"}])], "INCONCLUSIVE"),
|
|
416
|
+
}
|
|
417
|
+
for label, (events, expected) in cases.items():
|
|
418
|
+
with self.subTest(stream=label):
|
|
419
|
+
row = self.run_golden("\n".join(json.dumps(event) for event in events))
|
|
420
|
+
self.assertEqual(row["status"], expected)
|
|
421
|
+
if expected == "INCONCLUSIVE":
|
|
422
|
+
self.assertTrue(row["error"])
|
|
423
|
+
row = self.run_golden("\n".join(json.dumps(event) for event in (skill, success())), forbidden=True)
|
|
424
|
+
self.assertEqual(row["status"], "FAIL")
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
if __name__ == "__main__":
|
|
428
|
+
unittest.main()
|
|
@@ -287,11 +287,11 @@ def make_fixture(
|
|
|
287
287
|
|
|
288
288
|
occurrences = []
|
|
289
289
|
for receipt, receipt_hash in zip(receipts, receipt_hashes):
|
|
290
|
-
for item in receipt["findings"]:
|
|
290
|
+
for finding_hash in dict.fromkeys(canonical_hash(item) for item in receipt["findings"]):
|
|
291
291
|
occurrences.append(
|
|
292
292
|
{
|
|
293
293
|
"receipt_sha256": receipt_hash,
|
|
294
|
-
"finding_sha256":
|
|
294
|
+
"finding_sha256": finding_hash,
|
|
295
295
|
"disposition": "fixed",
|
|
296
296
|
}
|
|
297
297
|
)
|
|
@@ -1285,5 +1285,193 @@ assert [
|
|
|
1285
1285
|
(name, result.returncode, result.stdout) for name, result, _token in new_regressions
|
|
1286
1286
|
]
|
|
1287
1287
|
|
|
1288
|
+
def make_refuted_completion(name, **kwargs):
|
|
1289
|
+
kwargs.setdefault("receipt_findings", [[finding(1)], [finding(2)]])
|
|
1290
|
+
case = make_fixture(name, **kwargs)
|
|
1291
|
+
ledger = case["ledger"]
|
|
1292
|
+
dispositions = []
|
|
1293
|
+
for row in ledger["finding_classes"]:
|
|
1294
|
+
for index, occurrence in enumerate(row["occurrences"]):
|
|
1295
|
+
occurrence["disposition"] = "source_refuted"
|
|
1296
|
+
witness = bind_disposition_evidence(
|
|
1297
|
+
f"{name}-refuted-{index}.json", occurrence,
|
|
1298
|
+
evidence=[f"synthetic source refutes occurrence {index}"],
|
|
1299
|
+
)
|
|
1300
|
+
dispositions.append({
|
|
1301
|
+
**occurrence_ref(occurrence),
|
|
1302
|
+
"disposition": "source_refuted",
|
|
1303
|
+
"evidence": witness["evidence"],
|
|
1304
|
+
})
|
|
1305
|
+
document = {
|
|
1306
|
+
"schema_version": 1,
|
|
1307
|
+
"candidate_sha256": CANDIDATE,
|
|
1308
|
+
"review_result_sha256": list(case["receipt_hashes"]),
|
|
1309
|
+
"dispositions": dispositions,
|
|
1310
|
+
}
|
|
1311
|
+
name = f"{name}-finding-dispositions.json"
|
|
1312
|
+
ledger["finding_dispositions"] = {"file": name, "sha256": write_json(name, document)}
|
|
1313
|
+
completion_ref = ledger["completion_receipt"]
|
|
1314
|
+
complete = json.loads((root / completion_ref["file"]).read_text())
|
|
1315
|
+
complete.update(
|
|
1316
|
+
completion_basis="source_refuted_findings",
|
|
1317
|
+
finding_dispositions_sha256=ledger["finding_dispositions"]["sha256"],
|
|
1318
|
+
resolved_finding_occurrences=[occurrence_ref(item) for item in dispositions],
|
|
1319
|
+
)
|
|
1320
|
+
completion_ref["sha256"] = write_json(completion_ref["file"], complete)
|
|
1321
|
+
return case
|
|
1322
|
+
|
|
1323
|
+
|
|
1324
|
+
def mutate_refuted_completion(case, mutator):
|
|
1325
|
+
ref = case["ledger"]["completion_receipt"]
|
|
1326
|
+
document = json.loads((root / ref["file"]).read_text())
|
|
1327
|
+
mutator(document)
|
|
1328
|
+
ref["sha256"] = write_json(ref["file"], document)
|
|
1329
|
+
|
|
1330
|
+
|
|
1331
|
+
def mutate_refuted_document(case, mutator, *, duplicate_key=None):
|
|
1332
|
+
ref = case["ledger"]["finding_dispositions"]
|
|
1333
|
+
document = json.loads((root / ref["file"]).read_text())
|
|
1334
|
+
mutator(document)
|
|
1335
|
+
ref["sha256"] = (
|
|
1336
|
+
write_duplicate_json(ref["file"], document, duplicate_key, "ignored")
|
|
1337
|
+
if duplicate_key else write_json(ref["file"], document)
|
|
1338
|
+
)
|
|
1339
|
+
mutate_refuted_completion(case, lambda row: row.update(finding_dispositions_sha256=ref["sha256"]))
|
|
1340
|
+
|
|
1341
|
+
|
|
1342
|
+
adjudicated = make_refuted_completion("adjudicated")
|
|
1343
|
+
run("adjudicated", adjudicated["ledger"], 0, "ready_for_human_decision")
|
|
1344
|
+
for ref in adjudicated["ledger"]["controller_receipts"]:
|
|
1345
|
+
raw = (root / ref["file"]).read_bytes()
|
|
1346
|
+
assert hashlib.sha256(raw).hexdigest() == ref["sha256"]
|
|
1347
|
+
assert json.loads(raw)["status"] == "findings"
|
|
1348
|
+
|
|
1349
|
+
historical_refuted = make_refuted_completion("historical-refuted", receipt_findings=[[finding(1)], []])
|
|
1350
|
+
run("historical-refuted", historical_refuted["ledger"], 0, "ready_for_human_decision")
|
|
1351
|
+
|
|
1352
|
+
duplicate_finding_cases = [
|
|
1353
|
+
("duplicate-only", [[finding(1), dict(reversed(list(finding(1).items())))], []], 1),
|
|
1354
|
+
("duplicate-plus-distinct", [[finding(2), finding(1), finding(2)], []], 2),
|
|
1355
|
+
("duplicate-across-receipts", [[finding(1), finding(1)], [finding(1), finding(1)]], 2),
|
|
1356
|
+
]
|
|
1357
|
+
duplicate_cases = {}
|
|
1358
|
+
for name, receipt_findings, identity_count in duplicate_finding_cases:
|
|
1359
|
+
case = make_refuted_completion(name, receipt_findings=receipt_findings)
|
|
1360
|
+
duplicate_cases[name] = case
|
|
1361
|
+
raw_receipts = [(root / ref["file"]).read_bytes() for ref in case["ledger"]["controller_receipts"]]
|
|
1362
|
+
document = json.loads((root / case["ledger"]["finding_dispositions"]["file"]).read_text())
|
|
1363
|
+
assert len(document["dispositions"]) == identity_count
|
|
1364
|
+
run(name, case["ledger"], 0, "ready_for_human_decision")
|
|
1365
|
+
for ref, raw, original in zip(case["ledger"]["controller_receipts"], raw_receipts, receipt_findings):
|
|
1366
|
+
assert (root / ref["file"]).read_bytes() == raw
|
|
1367
|
+
assert hashlib.sha256(raw).hexdigest() == ref["sha256"]
|
|
1368
|
+
assert json.loads(raw)["findings"] == original
|
|
1369
|
+
|
|
1370
|
+
for name in ("duplicate-plus-distinct", "duplicate-across-receipts"):
|
|
1371
|
+
omitted = copy.deepcopy(duplicate_cases[name]["ledger"])
|
|
1372
|
+
del omitted["finding_classes"][0]["occurrences"][-1]
|
|
1373
|
+
run(f"{name}-missing-class", omitted, 1, "omits controller findings")
|
|
1374
|
+
|
|
1375
|
+
missing_distinct_disposition = make_refuted_completion(
|
|
1376
|
+
"duplicate-missing-disposition", receipt_findings=[[finding(2), finding(1), finding(2)], []]
|
|
1377
|
+
)
|
|
1378
|
+
mutate_refuted_document(missing_distinct_disposition, lambda row: row["dispositions"].pop())
|
|
1379
|
+
run("duplicate-missing-disposition", missing_distinct_disposition["ledger"], 1, "ordered controller findings")
|
|
1380
|
+
|
|
1381
|
+
repeated_class = copy.deepcopy(duplicate_cases["duplicate-only"]["ledger"])
|
|
1382
|
+
repeated_occurrences = repeated_class["finding_classes"][0]["occurrences"]
|
|
1383
|
+
repeated_occurrences.append(copy.deepcopy(repeated_occurrences[0]))
|
|
1384
|
+
run("duplicate-finding-repeated-class", repeated_class, 1, "classified more than once")
|
|
1385
|
+
|
|
1386
|
+
for name, kwargs, token in [
|
|
1387
|
+
("one-round", {"receipt_findings": [[finding(1)]]}, "review and challenge"),
|
|
1388
|
+
("empty", {"receipt_findings": [[], []]}, "at least one controller finding"),
|
|
1389
|
+
("succession", {"receipt_findings": [[finding(1)], [finding(2)], [finding(3)]], "succession": True}, "without succession"),
|
|
1390
|
+
]:
|
|
1391
|
+
case = make_refuted_completion(f"adjudicated-{name}", **kwargs)
|
|
1392
|
+
run(f"adjudicated-{name}", case["ledger"], 1, token)
|
|
1393
|
+
|
|
1394
|
+
external_basis = make_fixture("external-basis", completion_mutator=lambda row: row.update(
|
|
1395
|
+
completion_basis="external_pass", finding_dispositions_sha256=None,
|
|
1396
|
+
resolved_finding_occurrences=[],
|
|
1397
|
+
))
|
|
1398
|
+
run("external-basis", external_basis["ledger"], 0, "ready_for_human_decision")
|
|
1399
|
+
|
|
1400
|
+
orphan_dispositions = make_fixture("orphan-dispositions")
|
|
1401
|
+
orphan_dispositions["ledger"]["finding_dispositions"] = dict(adjudicated["ledger"]["finding_dispositions"])
|
|
1402
|
+
run("orphan-dispositions", orphan_dispositions["ledger"], 1, "requires a source_refuted_findings completion")
|
|
1403
|
+
|
|
1404
|
+
missing_ref = make_refuted_completion("adjudicated-missing-ref")
|
|
1405
|
+
del missing_ref["ledger"]["finding_dispositions"]
|
|
1406
|
+
run("adjudicated-missing-ref", missing_ref["ledger"], 1, "finding_dispositions")
|
|
1407
|
+
|
|
1408
|
+
bad_ref_hash = make_refuted_completion("adjudicated-bad-ref-hash")
|
|
1409
|
+
bad_ref_hash["ledger"]["finding_dispositions"]["sha256"] = "f" * 64
|
|
1410
|
+
run("adjudicated-bad-ref-hash", bad_ref_hash["ledger"], 1, "digest does not match")
|
|
1411
|
+
|
|
1412
|
+
for name, mutate, token in [
|
|
1413
|
+
("stale-candidate", lambda row: row.update(candidate_sha256=OTHER_CANDIDATE), "current candidate"),
|
|
1414
|
+
("missing-prefix", lambda row: row["review_result_sha256"].pop(0), "ordered controller receipts"),
|
|
1415
|
+
("reordered-prefix", lambda row: row["review_result_sha256"].reverse(), "ordered controller receipts"),
|
|
1416
|
+
("missing-occurrence", lambda row: row["dispositions"].pop(0), "ordered controller findings"),
|
|
1417
|
+
("duplicate-occurrence", lambda row: row["dispositions"].append(copy.deepcopy(row["dispositions"][0])), "ordered controller findings"),
|
|
1418
|
+
("reordered-occurrences", lambda row: row["dispositions"].reverse(), "ordered controller findings"),
|
|
1419
|
+
("wrong-finding", lambda row: row["dispositions"][0].update(finding_sha256="f" * 64), "ordered controller findings"),
|
|
1420
|
+
("empty-evidence", lambda row: row["dispositions"][0].update(evidence=[]), "non-empty evidence"),
|
|
1421
|
+
("blank-evidence", lambda row: row["dispositions"][0].update(evidence=[" "]), "normalized string"),
|
|
1422
|
+
("long-evidence", lambda row: row["dispositions"][0].update(evidence=["x" * 1001]), "1000 characters"),
|
|
1423
|
+
("different-evidence", lambda row: row["dispositions"][0].update(evidence=["different source claim"]), "class disposition evidence"),
|
|
1424
|
+
]:
|
|
1425
|
+
case = make_refuted_completion(f"adjudicated-{name}")
|
|
1426
|
+
mutate_refuted_document(case, mutate)
|
|
1427
|
+
run(f"adjudicated-{name}", case["ledger"], 1, token)
|
|
1428
|
+
|
|
1429
|
+
for disposition in ("fixed", "accepted_tradeoff", "pre_existing_out_of_scope", "open", "needs_human_decision"):
|
|
1430
|
+
case = make_refuted_completion(f"adjudicated-mixed-{disposition}")
|
|
1431
|
+
mutate_refuted_document(case, lambda row: row["dispositions"][0].update(disposition=disposition))
|
|
1432
|
+
run(f"adjudicated-mixed-{disposition}", case["ledger"], 1, "only source_refuted")
|
|
1433
|
+
class_case = make_refuted_completion(f"adjudicated-class-{disposition}")
|
|
1434
|
+
occurrence = class_case["ledger"]["finding_classes"][0]["occurrences"][0]
|
|
1435
|
+
occurrence["disposition"] = disposition
|
|
1436
|
+
clear_disposition_evidence(occurrence)
|
|
1437
|
+
if disposition in CLOSED_DISPOSITIONS:
|
|
1438
|
+
bind_disposition_evidence(f"adjudicated-class-{disposition}-witness.json", occurrence)
|
|
1439
|
+
run(f"adjudicated-class-{disposition}", class_case["ledger"], 1, "only source_refuted")
|
|
1440
|
+
|
|
1441
|
+
for name, mutate, token in [
|
|
1442
|
+
("wrong-complete-hash", lambda row: row.update(finding_dispositions_sha256="f" * 64), "completion disposition digest"),
|
|
1443
|
+
("missing-complete-pairs", lambda row: row.update(resolved_finding_occurrences=[]), "resolved_finding_occurrences"),
|
|
1444
|
+
("reordered-complete-pairs", lambda row: row["resolved_finding_occurrences"].reverse(), "resolved_finding_occurrences"),
|
|
1445
|
+
("unknown-basis", lambda row: row.update(completion_basis="author-approved"), "completion_basis"),
|
|
1446
|
+
("basis-downgrade", lambda row: row.update(completion_basis="external_pass"), "final external receipt with findings"),
|
|
1447
|
+
]:
|
|
1448
|
+
case = make_refuted_completion(f"adjudicated-{name}")
|
|
1449
|
+
mutate_refuted_completion(case, mutate)
|
|
1450
|
+
run(f"adjudicated-{name}", case["ledger"], 1, token)
|
|
1451
|
+
|
|
1452
|
+
missing_witness = make_refuted_completion("adjudicated-missing-witness")
|
|
1453
|
+
clear_disposition_evidence(missing_witness["ledger"]["finding_classes"][0]["occurrences"][0])
|
|
1454
|
+
run("adjudicated-missing-witness", missing_witness["ledger"], 1, "disposition_evidence_file")
|
|
1455
|
+
|
|
1456
|
+
duplicate_document = make_refuted_completion("adjudicated-duplicate-key")
|
|
1457
|
+
mutate_refuted_document(duplicate_document, lambda row: None, duplicate_key="candidate_sha256")
|
|
1458
|
+
run("adjudicated-duplicate-key", duplicate_document["ledger"], 1, "duplicate object key")
|
|
1459
|
+
|
|
1460
|
+
linked_document = make_refuted_completion("adjudicated-linked-document")
|
|
1461
|
+
ref = linked_document["ledger"]["finding_dispositions"]
|
|
1462
|
+
linked_path = root / ref["file"]
|
|
1463
|
+
original_path = linked_path.with_suffix(".original")
|
|
1464
|
+
linked_path.rename(original_path)
|
|
1465
|
+
linked_path.symlink_to(original_path)
|
|
1466
|
+
run("adjudicated-linked-document", linked_document["ledger"], 1, "unreadable")
|
|
1467
|
+
|
|
1468
|
+
path_escape = make_refuted_completion("adjudicated-path-escape")
|
|
1469
|
+
path_escape["ledger"]["finding_dispositions"]["file"] = "../outside.json"
|
|
1470
|
+
run("adjudicated-path-escape", path_escape["ledger"], 1, "ledger directory")
|
|
1471
|
+
|
|
1472
|
+
delta_case = make_refuted_completion("adjudicated-delta")
|
|
1473
|
+
delta_case["ledger"]["unreviewed_delta"] = ["post-review implementation edit"]
|
|
1474
|
+
run("adjudicated-delta", delta_case["ledger"], 1, "unreviewed delta")
|
|
1475
|
+
|
|
1288
1476
|
print("test_validate_extraction_review_state: ok")
|
|
1289
1477
|
PY
|