@ccoalm/ccl-skills 0.1.1 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +93 -16
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +10 -0
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +249 -5
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate_abort_leak.sh +394 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +7 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md +50 -1
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +1 -0
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +1 -1
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +59 -1
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +3 -1
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +34 -0
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +16 -4
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +391 -14
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +74 -0
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +74 -33
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh +11 -4
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +421 -0
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_round_attribution.sh +576 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_source_refuted.sh +176 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_regression_runner_lanes.sh +101 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_root_depth.sh +6 -2
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +1 -1
  23. package/dist/assets/release.json +45 -20
  24. package/package.json +1 -1
@@ -121,6 +121,19 @@ PY
121
121
  printf ' sensitive to %s: %s\n' "$name" "${out%%$'\n'*}"
122
122
  }
123
123
 
124
+ # The walk below REGISTERS its mutants instead of running them inline, so they can
125
+ # be dispatched concurrently further down. Registration changes nothing about what
126
+ # is asserted: every registered triple is handed to the same
127
+ # mutate_and_expect_mismatch above, unmodified.
128
+ mutation_names=()
129
+ mutation_finds=()
130
+ mutation_replaces=()
131
+ register_mutation() {
132
+ mutation_names+=("$1")
133
+ mutation_finds+=("$2")
134
+ mutation_replaces+=("$3")
135
+ }
136
+
124
137
  # Prove the guard above can actually fail before trusting any verdict it gives.
125
138
  # The first version of this walk accepted ANY nonzero exit as sensitivity, so a
126
139
  # mutant that merely broke the parser would have banked as proof — an
@@ -147,21 +160,21 @@ if ! grep -q 'produced unparseable verdicts' "$self_check_stderr"; then
147
160
  fi
148
161
  printf ' guard self-check: a broken mutant is rejected, and for the right reason\n'
149
162
 
150
- mutate_and_expect_mismatch tolerate-all-unknown-containers \
163
+ register_mutation tolerate-all-unknown-containers \
151
164
  ' if isinstance(value, (list, dict)) and value:
152
165
  unknown_fields.add(field)' \
153
166
  ' if False:
154
167
  unknown_fields.add(field)'
155
168
 
156
- mutate_and_expect_mismatch drop-authority-name-guard \
169
+ register_mutation drop-authority-name-guard \
157
170
  ' segments = re.split(r"[_.\-]+|(?<=[a-z0-9])(?=[A-Z])", name)' \
158
171
  ' return False'
159
172
 
160
- mutate_and_expect_mismatch drop-authority-presence-requirement \
173
+ register_mutation drop-authority-presence-requirement \
161
174
  'REQUIRED_PRESENT_INIT_FIELDS = ("permissionMode",)' \
162
175
  'REQUIRED_PRESENT_INIT_FIELDS = ()'
163
176
 
164
- mutate_and_expect_mismatch drop-field-name-sanitizer \
177
+ register_mutation drop-field-name-sanitizer \
165
178
  ' text = name if isinstance(name, str) else repr(name)' \
166
179
  ' return name if isinstance(name, str) else repr(name)'
167
180
 
@@ -169,11 +182,11 @@ mutate_and_expect_mismatch drop-field-name-sanitizer \
169
182
  # they fail for opposite reasons: widening it launders a proven customization
170
183
  # into the cascadable class, while removing it restores the total-outage
171
184
  # behaviour this class exists to prevent.
172
- mutate_and_expect_mismatch widen-host-vocabulary-to-any-entry \
185
+ register_mutation widen-host-vocabulary-to-any-entry \
173
186
  ' return bool(BARE_HOST_IDENTIFIER.fullmatch(identifier))' \
174
187
  ' return True'
175
188
 
176
- mutate_and_expect_mismatch terminalize-host-vocabulary \
189
+ register_mutation terminalize-host-vocabulary \
177
190
  ' return bool(BARE_HOST_IDENTIFIER.fullmatch(identifier))' \
178
191
  ' return False'
179
192
 
@@ -181,7 +194,7 @@ mutate_and_expect_mismatch terminalize-host-vocabulary \
181
194
  # the baseline must affect the allow decision, and its CLI version must bind the
182
195
  # formal init. Baseline skills are deliberately not authority because a leaked
183
196
  # user skill would otherwise become callable in the formal run.
184
- mutate_and_expect_mismatch drop-host-baseline-vocabulary \
197
+ register_mutation drop-host-baseline-vocabulary \
185
198
  ' if customization_entry_allowed(
186
199
  field,
187
200
  entry,
@@ -197,18 +210,18 @@ mutate_and_expect_mismatch drop-host-baseline-vocabulary \
197
210
  set(),
198
211
  ):'
199
212
 
200
- mutate_and_expect_mismatch authorize-host-baseline-skill \
213
+ register_mutation authorize-host-baseline-skill \
201
214
  ' identifier in KNOWN_SAFE_BUILTIN_SKILLS
202
215
  or identifier in selected_names' \
203
216
  ' identifier in KNOWN_SAFE_BUILTIN_SKILLS
204
217
  or identifier in baseline_skills
205
218
  or identifier in selected_names'
206
219
 
207
- mutate_and_expect_mismatch drop-host-baseline-required-empty-check \
220
+ register_mutation drop-host-baseline-required-empty-check \
208
221
  ' if field not in HOST_VOCABULARY_FIELDS and init_event.get(field) != []:' \
209
222
  ' if field not in HOST_VOCABULARY_FIELDS and False:'
210
223
 
211
- mutate_and_expect_mismatch widen-host-baseline-to-namespaced-entries \
224
+ register_mutation widen-host-baseline-to-namespaced-entries \
212
225
  ' or any(
213
226
  identifier not in known_host_identifiers
214
227
  and not is_bare_host_identifier(identifier)
@@ -216,7 +229,7 @@ mutate_and_expect_mismatch widen-host-baseline-to-namespaced-entries \
216
229
  )' \
217
230
  ' or any(identifier == "<unidentified>" for identifier in identifiers)'
218
231
 
219
- mutate_and_expect_mismatch drop-host-baseline-version-binding \
232
+ register_mutation drop-host-baseline-version-binding \
220
233
  ' if baseline_version is not None and ev.get("claude_code_version") != baseline_version:
221
234
  # The two invocations no longer prove one same-version host
222
235
  # vocabulary snapshot. Refuse this lane, but treat the mismatch as
@@ -230,7 +243,7 @@ mutate_and_expect_mismatch drop-host-baseline-version-binding \
230
243
  # alone, which reaches TOLERATED when that name is an allowed built-in -- the
231
244
  # most severe class in this file, so it needs its own mutant rather than riding
232
245
  # on the bare-identifier one.
233
- mutate_and_expect_mismatch drop-whole-value-gate \
246
+ register_mutation drop-whole-value-gate \
234
247
  ' if field in HOST_VOCABULARY_FIELDS and (
235
248
  not host_entry_is_whole(entry, identifier)
236
249
  ):' \
@@ -239,7 +252,7 @@ mutate_and_expect_mismatch drop-whole-value-gate \
239
252
  # ...and the weaker version of the same gate: checking only the SHAPE (a plain
240
253
  # string) while still judging a truncated token. This is what the gate looked
241
254
  # like before the third finding, so it must be detectable on its own.
242
- mutate_and_expect_mismatch weaken-whole-value-gate-to-shape-only \
255
+ register_mutation weaken-whole-value-gate-to-shape-only \
243
256
  ' normalized = entry.lower()
244
257
  if normalized.startswith("/"):
245
258
  normalized = normalized[1:]
@@ -249,21 +262,85 @@ mutate_and_expect_mismatch weaken-whole-value-gate-to-shape-only \
249
262
  # The regression a round-9 review found in the gate itself: stripping before the
250
263
  # comparison re-introduces the lossiness the gate exists to reject, and wrapping
251
264
  # an ALLOWLISTED name in whitespace then reaches TOLERATED.
252
- mutate_and_expect_mismatch strip-before-the-whole-value-comparison \
265
+ register_mutation strip-before-the-whole-value-comparison \
253
266
  ' normalized = entry.lower()' \
254
267
  ' normalized = entry.strip().lower()'
255
268
 
256
- mutate_and_expect_mismatch drop-host-vocabulary-breach-guard \
269
+ register_mutation drop-host-vocabulary-breach-guard \
257
270
  ' if unclassifiable_vocabulary and not surface_breached:' \
258
271
  ' if unclassifiable_vocabulary:'
259
272
 
260
273
  # The two parse paths implement the class separately, so each needs its own
261
274
  # mutant: dropping it from the main-invocation predicate leaves the probe path
262
275
  # correct, which is exactly the shape of divergence this oracle exists to catch.
263
- mutate_and_expect_mismatch drop-host-vocabulary-from-main-path \
276
+ register_mutation drop-host-vocabulary-from-main-path \
264
277
  'runtime_drift_only = bool(unknown or unverifiable or vocabulary) and not (' \
265
278
  'runtime_drift_only = bool(unknown or unverifiable) and not ('
266
279
 
280
+ # Dispatch the registered walk with bounded concurrency. The mutants are
281
+ # independent by construction: each writes its own `mutant_<name>.py` copy under
282
+ # $tmp_dir and runs the oracle against that copy, sharing nothing writable. Only
283
+ # the SCHEDULE changes -- every assertion in mutate_and_expect_mismatch still runs
284
+ # once per mutant. This suite was CI's slowest single entry (pure subprocess time,
285
+ # zero sleeps), so serial dispatch left the runner's other cores idle.
286
+ # Output is replayed in REGISTRATION order so a parallel run reads like the serial
287
+ # one, and any failure's stderr is replayed before the run is failed.
288
+ mutation_jobs="${INIT_POLICY_MATRIX_JOBS:-${SUITE_JOBS:-4}}"
289
+ case "$mutation_jobs" in ''|*[!0-9]*) mutation_jobs=4 ;; esac
290
+ [ "$mutation_jobs" -ge 1 ] || mutation_jobs=1
291
+
292
+ walk_total=${#mutation_names[@]}
293
+ if [ "$walk_total" -lt 1 ]; then
294
+ printf 'the mutation walk registered nothing; the sensitivity check is disarmed\n' >&2
295
+ exit 1
296
+ fi
297
+
298
+ walk_i=0
299
+ while [ "$walk_i" -lt "$walk_total" ]; do
300
+ # `-p` is load-bearing: plain `jobs -r` prints multi-line command text, so a
301
+ # line count would read one running job as several and degrade to serial.
302
+ while [ "$(jobs -rp 2>/dev/null | wc -l)" -ge "$mutation_jobs" ]; do sleep 0.1; done
303
+ (
304
+ # The status is recorded by an EXIT trap, and the helper is a SIMPLE command.
305
+ # Putting the call in an `if` condition would disable errexit for the whole
306
+ # function body, so an unguarded failure partway through could still be
307
+ # followed by the helper's final successful command and bank as `ok`. As a
308
+ # simple command under `set -e`, any such failure aborts the subshell and the
309
+ # trap records the real status.
310
+ trap 'printf "%s\n" "$?" >"$tmp_dir/walk_status_$walk_i"' EXIT
311
+ mutate_and_expect_mismatch \
312
+ "${mutation_names[$walk_i]}" "${mutation_finds[$walk_i]}" "${mutation_replaces[$walk_i]}" \
313
+ >"$tmp_dir/walk_out_$walk_i" 2>"$tmp_dir/walk_err_$walk_i"
314
+ ) &
315
+ walk_i=$((walk_i + 1))
316
+ done
317
+ wait
318
+
319
+ # Every registered index must have written an EXPLICIT terminal status. A worker
320
+ # killed before it could report (fatal shell error, OOM, cancellation) leaves no
321
+ # file, and treating that absence as success would bank a mutant that never ran
322
+ # as proof of sensitivity — the same absence-means-pass defect this walk's own
323
+ # guard exists to prevent one level down.
324
+ walk_failed=0
325
+ walk_i=0
326
+ while [ "$walk_i" -lt "$walk_total" ]; do
327
+ cat "$tmp_dir/walk_out_$walk_i" 2>/dev/null
328
+ if [ ! -f "$tmp_dir/walk_status_$walk_i" ]; then
329
+ printf 'mutation %s never reported a terminal status; its worker died without completing\n' \
330
+ "${mutation_names[$walk_i]}" >&2
331
+ cat "$tmp_dir/walk_err_$walk_i" 2>/dev/null >&2
332
+ walk_failed=1
333
+ elif [ "$(cat "$tmp_dir/walk_status_$walk_i")" != 0 ]; then
334
+ cat "$tmp_dir/walk_err_$walk_i" >&2
335
+ walk_failed=1
336
+ fi
337
+ walk_i=$((walk_i + 1))
338
+ done
339
+ if [ "$walk_failed" -ne 0 ]; then
340
+ printf 'the mutation walk reported at least one non-sensitive or broken mutant (see above)\n' >&2
341
+ exit 1
342
+ fi
343
+
267
344
  if [ "$(shasum -a 256 "$parser" | awk '{print $1}')" != "$pristine_digest" ]; then
268
345
  printf 'the mutation walk modified the real parser; every mutation must stay in a copy\n' >&2
269
346
  exit 1
@@ -302,6 +302,16 @@ run_ok_expected_native_skills 'testing-strategy' 'testing-strategy' 0 \
302
302
  run_reason_expected_native_skills 'testing-strategy' 'testing-strategy' 0 \
303
303
  $'{"type":"system","subtype":"init","permissionMode":"default","tools":[],"mcp_servers":[],"slash_commands":["ccl-skills:testing-strategy","workflow-launch-exec","ultrareview","unrelated:danger"],"skills":["testing-strategy"],"plugins":["ccl-skills"]}\n{"type":"result","subtype":"success","is_error":false,"result":"ok"}' \
304
304
  'runtime capability surface is not empty'
305
+ # A dict-shaped entry is never whole, however allowed its `name` reads. The
306
+ # entry hides a sibling key the identifier helper discards, so clearing it on
307
+ # the truncated name would accept a customization whose proof was in the part
308
+ # that was thrown away. host_entry_is_whole must reject the shape BEFORE the
309
+ # allowlist reads it; flipping its non-string branch to True makes this case
310
+ # pass, which is exactly the regression this asserts.
311
+ run_reason_expected_native_skills 'testing-strategy' 'testing-strategy' 0 \
312
+ $'{"type":"system","subtype":"init","permissionMode":"default","tools":[],"mcp_servers":[],"slash_commands":["ccl-skills:testing-strategy","workflow-launch-exec",{"name":"ultrareview","path":"hidden-sibling-value"}],"skills":["testing-strategy"],"plugins":["ccl-skills"]}\n{"type":"result","subtype":"success","is_error":false,"result":"ok"}' \
313
+ 'runtime capability surface is not empty'
314
+
305
315
  # A matching name in either executable surface is still terminal. Built-in UI
306
316
  # registration never authorizes a tool declaration or invocation.
307
317
  run_reason_expected_native_skills 'testing-strategy' 'testing-strategy' 0 \
@@ -17,9 +17,175 @@ cleanup_escaped_descendant() {
17
17
  case "$escaped_cleanup_pid" in
18
18
  ''|*[!0-9]*) return 0 ;;
19
19
  esac
20
- kill -KILL "$escaped_cleanup_pid" 2>/dev/null || true
20
+ # Same ownership re-proof as the wrapper reaper: this pid was cached earlier and the
21
+ # case may already have killed it, so by the time an abort runs this trap the number
22
+ # can belong to someone else. The descendant setsid's away from the wrapper's group but
23
+ # keeps this run's state path in its argv, which is what identifies it.
24
+ case "$(ps -o command= -p "$escaped_cleanup_pid" 2>/dev/null || true)" in
25
+ *"$WORK/"*|*"$WORK_REAL/"*) kill -KILL "$escaped_cleanup_pid" 2>/dev/null || true ;;
26
+ esac
27
+ }
28
+ # The controller starts every reviewer wrapper with start_new_session=True, so each
29
+ # sits in its own session where no group- or terminal-directed signal reaches it, and
30
+ # several fixture behaviors deliberately ignore SIGTERM — which leaves the controller
31
+ # as the only party that reaps them on the happy path. An abort that removes the
32
+ # controller first (tree-kill, CI cancellation, host suspend) leaves them with no
33
+ # reaper at all; one such wrapper was observed alive for 39 hours with ppid=1. This is
34
+ # the suite's own reaper for that case.
35
+ #
36
+ # Two records answer "did this run start it", and the union is what the suite audits
37
+ # and reaps:
38
+ #
39
+ # by GROUP — every wrapper writes its process-group id under state/pgids before it
40
+ # does anything else, and the controller starts each wrapper with
41
+ # start_new_session, so that group holds the wrapper AND everything it
42
+ # spawns, including a descendant whose argv names no path at all.
43
+ # by PATH — anything naming this run's private WORK directory. This still matters:
44
+ # the controller itself, and any descendant that outlives its group's
45
+ # record, is caught here.
46
+ #
47
+ # Neither alone is enough. A path-only scan misses a bare `sleep`; a pgid-only scan
48
+ # misses whatever ran before its wrapper recorded a group. Known residual boundary: a
49
+ # descendant that deliberately setsid's out of its wrapper's group AND carries no WORK
50
+ # path leaves both records — the escaped-descendant case does exactly that on purpose
51
+ # and carries its own dedicated cleanup above.
52
+ #
53
+ # The needles go through the environment, NOT through `awk -v`: `ps -e` lists this very
54
+ # awk, and an argv-passed needle makes the scanner match itself — a fresh pid every call
55
+ # that is already gone by the time anything inspects it. Measured as a false leak report
56
+ # on a clean run.
57
+ #
58
+ # Zombies are excluded: an unreaped corpse still appears in ps, and whether its parent
59
+ # has gotten around to reaping it is the OS's business, not this suite's.
60
+ prune_dead_pgid_records() {
61
+ local pgid
62
+ for pgid in $(review_owned_pgids); do
63
+ ps -eo pgid= 2>/dev/null | tr -d ' ' | grep -qx "$pgid" && continue
64
+ rm -f "$WORK/state/pgids/$pgid" 2>/dev/null || true
65
+ done
66
+ }
67
+ # A group id is a recycled number, so the record stores the LEADER's start time and
68
+ # every read re-proves the group is still the one that registered. Three cases, and the
69
+ # middle one is why the record is trusted at all:
70
+ # leader alive, start time matches -> ours
71
+ # no process holds pid == pgid -> ours: the leader died, so anything still
72
+ # carrying that group id descends from it, and
73
+ # nothing can have become that group's leader
74
+ # leader alive, start time differs -> the number was recycled; drop the record and
75
+ # never signal that group
76
+ # REPORTING may be broad; SIGNALLING may only be what identity proves. That split is
77
+ # the structural answer to a class of narrowing objections about recycled group ids,
78
+ # and it is deliberate rather than a compromise: on a shared gate a false leak REPORT
79
+ # costs a red run, while a false SIGNAL kills a stranger's process group.
80
+ #
81
+ # verified — a process still holds pid == pgid and its start time matches the record.
82
+ # The group is provably ours: report AND signal it.
83
+ # leaderless— nothing holds pid == pgid. Anything still carrying the id is most likely
84
+ # our dead leader's descendant, but the id could also have been recycled by
85
+ # a process that led a group and exited while its children ran on. Report
86
+ # it, never signal it.
87
+ # recycled — a leader exists with a different start time. Drop the record entirely.
88
+ review_owned_pgids() { # $1: "verified" to exclude leaderless records
89
+ local pgid_file pgid recorded_start current_start leader_command leader_owned
90
+ for pgid_file in "$WORK"/state/pgids/*; do
91
+ [ -e "$pgid_file" ] || continue
92
+ pgid="${pgid_file##*/}"
93
+ case "$pgid" in ''|*[!0-9]*) continue ;; esac
94
+ current_start="$(ps -o lstart= -p "$pgid" 2>/dev/null || true)"
95
+ if [ -z "$current_start" ]; then
96
+ [ "${1:-}" = "verified" ] && continue
97
+ printf '%s\n' "$pgid"
98
+ continue
99
+ fi
100
+ recorded_start="$(cat "$pgid_file" 2>/dev/null || true)"
101
+ leader_command="$(ps -o command= -p "$pgid" 2>/dev/null || true)"
102
+ # Start time alone is second-granular, so a pid recycled inside the same second would
103
+ # match. The leader of one of our groups is always a wrapper, and a wrapper's command
104
+ # line names this run's private WORK path — require both. Matched with `case` rather
105
+ # than a regex so the path needs no escaping.
106
+ leader_owned=0
107
+ case "$leader_command" in
108
+ *"$WORK/"*|*"$WORK_REAL/"*) leader_owned=1 ;;
109
+ esac
110
+ if [ "$current_start" != "$recorded_start" ] || [ "$leader_owned" != 1 ]; then
111
+ rm -f "$pgid_file" 2>/dev/null || true
112
+ continue
113
+ fi
114
+ printf '%s\n' "$pgid"
115
+ done
116
+ }
117
+ # The escaped-descendant fixture deliberately setsid's out of its wrapper's group and
118
+ # carries no WORK path, so it is invisible to both records above — the one member of the
119
+ # residual boundary this suite actually knows the pid of. Its own case kills it inline
120
+ # and the EXIT trap kills it again, but neither is what PROVES it is gone: read the pid
121
+ # the fixture recorded, so a cleanup that silently stops working is caught here instead
122
+ # of leaving a detached process the exit assertion cannot see.
123
+ review_escaped_descendant_alive() {
124
+ local pid
125
+ local record start
126
+ record="$(cat "$WORK/state/audit/escaped_pid" 2>/dev/null || true)"
127
+ pid="${record%% *}"
128
+ case "$pid" in ''|*[!0-9]*) return 0 ;; esac
129
+ start="$(ps -o lstart= -p "$pid" 2>/dev/null || true)"
130
+ [ -n "$start" ] || return 0
131
+ # The record carries the descendant's start time for the same reason every other
132
+ # ownership check here does: a bare pid outlives its process and a recycled one would
133
+ # be reported as this suite's leak.
134
+ [ "$start" = "${record#* }" ] || return 0
135
+ case "$(ps -o stat= -p "$pid" 2>/dev/null || true)" in
136
+ *Z*) return 0 ;;
137
+ esac
138
+ printf '%s\n' "$pid"
139
+ }
140
+ review_harness_pids_by_path() {
141
+ ps -eo pid=,stat=,command= 2>/dev/null |
142
+ REVIEW_WORK_DIR="$WORK/" REVIEW_WORK_DIR_REAL="$WORK_REAL/" \
143
+ awk 'BEGIN { a = ENVIRON["REVIEW_WORK_DIR"]; b = ENVIRON["REVIEW_WORK_DIR_REAL"] }
144
+ (index($0, a) || index($0, b)) && $2 !~ /Z/ { print $1 }'
21
145
  }
22
- trap 'cleanup_escaped_descendant; rm -rf "$WORK"' EXIT
146
+ review_harness_pids_alive() {
147
+ {
148
+ review_escaped_descendant_alive
149
+ review_harness_pids_by_path
150
+ review_owned_pgids | while read -r owned_pgid; do
151
+ ps -eo pid=,pgid=,stat= 2>/dev/null |
152
+ awk -v want="$owned_pgid" '$2 == want && $3 !~ /Z/ { print $1 }'
153
+ done
154
+ } | sort -un
155
+ }
156
+ # Irreducible residual, stated rather than patched further: every ownership proof here is
157
+ # a `ps` read followed by a `kill`, and nothing in a shell binds the two — a group or pid
158
+ # that dies in that gap and is recycled would receive the signal. Successive review rounds
159
+ # can always name a narrower instance of this window; what the code can do is prove
160
+ # ownership as late as possible (which it does, immediately before each signal) and keep
161
+ # the dangerous half narrow: only groups with a verified live leader are signalled, and a
162
+ # leaderless record is reported but never signalled.
163
+ reap_review_wrappers() {
164
+ local pid pgid
165
+ # Groups first, so a wrapper's descendants go with it rather than being re-parented
166
+ # into a second pass that no longer recognises them.
167
+ for pgid in $(review_owned_pgids verified); do
168
+ kill -KILL -"$pgid" 2>/dev/null || true
169
+ done
170
+ # By pid, and only for pids the PATH scan just matched: a pid known solely through a
171
+ # leaderless group record is reported, not signalled. Re-prove ownership immediately
172
+ # before signalling — the scan above is already history, and a pid that exited in
173
+ # between is a number the OS hands to someone else.
174
+ for pid in $(review_harness_pids_by_path); do
175
+ case "$(ps -o command= -p "$pid" 2>/dev/null || true)" in
176
+ *"$WORK/"*|*"$WORK_REAL/"*) : ;;
177
+ *) continue ;;
178
+ esac
179
+ kill -KILL -"$pid" 2>/dev/null || true
180
+ kill -KILL "$pid" 2>/dev/null || true
181
+ done
182
+ }
183
+ trap 'cleanup_escaped_descendant; reap_review_wrappers; rm -rf "$WORK"' EXIT
184
+ # `exit` runs the EXIT trap above, so the abort paths reuse one cleanup body. Without
185
+ # these, an interrupted run leaves the wrappers behind: that is the defect this
186
+ # guards, and EXIT alone never fires on a signal.
187
+ trap 'exit 130' INT
188
+ trap 'exit 143' TERM HUP
23
189
  fails=0
24
190
 
25
191
  mkdir -p "$WORK/harness/scripts" "$WORK/state" "$WORK/repo" "$WORK/empty-registry"
@@ -394,6 +560,18 @@ cat >"$WORK/harness/scripts/claude_review.sh" <<'CLAUDE_STUB'
394
560
  #!/usr/bin/env bash
395
561
  set -u
396
562
  state="$REVIEW_GATE_TEST_STATE"
563
+ # Record this wrapper's process GROUP before anything else. The suite reaps and audits
564
+ # by group, not by pid: a descendant started from here can carry an argv that names no
565
+ # path under WORK (a bare `sleep`), and a pid-keyed record would name only the wrapper
566
+ # while the group is what actually holds everything this wrapper spawned.
567
+ review_pgid="$(ps -o pgid= -p $$ 2>/dev/null | tr -d ' ')"
568
+ case "$review_pgid" in
569
+ ''|*[!0-9]*) : ;;
570
+ *)
571
+ mkdir -p "$state/pgids" 2>/dev/null &&
572
+ ps -o lstart= -p "$review_pgid" 2>/dev/null >"$state/pgids/$review_pgid" 2>/dev/null
573
+ ;;
574
+ esac
397
575
  mode="$1"
398
576
  shift
399
577
  diff_file=""
@@ -473,7 +651,14 @@ case "$behavior" in
473
651
  (
474
652
  trap '' TERM
475
653
  printf '%s\n' "$BASHPID" >"$state/hang_child_pid"
476
- while :; do sleep 1; done
654
+ # Bounded, like the escaped-descendant fixture below: this must outlive every
655
+ # budget any case gives it (largest is --total-timeout 50) so the controller's
656
+ # kill is what ends it, but it must NOT be unbounded. When an abort removes the
657
+ # controller before its timeout path runs, nothing else can signal this process
658
+ # group — it is TERM-immune and lives in its own session — so an unbounded loop
659
+ # is a wrapper that survives for days. Measured before this bound: 39 hours.
660
+ hang_left="${REVIEW_GATE_TEST_HANG_SECONDS:-300}"
661
+ while [ "$hang_left" -gt 0 ]; do sleep 1; hang_left=$((hang_left-1)); done
477
662
  ) &
478
663
  wait "$!"
479
664
  ;;
@@ -561,6 +746,16 @@ cat >"$WORK/harness/scripts/candidate_stub.sh" <<'CLIENT_STUB'
561
746
  #!/usr/bin/env bash
562
747
  set -u
563
748
  state="$REVIEW_GATE_TEST_STATE"
749
+ # Same process-group record as the claude stub; the fallback clients run the same
750
+ # TERM-immune hang behaviors and leak the same way.
751
+ review_pgid="$(ps -o pgid= -p $$ 2>/dev/null | tr -d ' ')"
752
+ case "$review_pgid" in
753
+ ''|*[!0-9]*) : ;;
754
+ *)
755
+ mkdir -p "$state/pgids" 2>/dev/null &&
756
+ ps -o lstart= -p "$review_pgid" 2>/dev/null >"$state/pgids/$review_pgid" 2>/dev/null
757
+ ;;
758
+ esac
564
759
  client="$(basename "$0" _review.sh)"
565
760
  mode=review
566
761
  diff_file=""
@@ -618,7 +813,10 @@ case "$behavior" in
618
813
  oversize_inline) printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"packet_too_large_for_inline","reason_code":"capability_missing","cascade_eligible":true}\n' "$client" "$mode"; exit 2 ;;
619
814
  hang)
620
815
  trap '' TERM
621
- while :; do sleep 1; done
816
+ # Bounded for the same reason as the claude stub's hang: an abort that removes
817
+ # the controller leaves this TERM-immune process with no other reaper.
818
+ hang_left="${REVIEW_GATE_TEST_HANG_SECONDS:-300}"
819
+ while [ "$hang_left" -gt 0 ]; do sleep 1; hang_left=$((hang_left-1)); done
622
820
  ;;
623
821
  auth)
624
822
  if [ "$host_attempted" = 1 ]; then
@@ -794,7 +992,16 @@ cat >"$WORK/near-limit-plan.json" <<JSON
794
992
  JSON
795
993
 
796
994
  reset_case() {
797
- rm -f "$WORK/state"/*
995
+ # Files only: the process-group record is a directory and must survive case resets —
996
+ # a wrapper still running from the previous case would otherwise lose the only handle
997
+ # the suite has on its group.
998
+ find "$WORK/state" -maxdepth 1 -type f -exec rm -f {} +
999
+ # Drop records whose group has no live member. A group id is a recycled number, so a
1000
+ # record kept past its group's life can eventually name a stranger's group — and this
1001
+ # suite signals groups. Pruning between cases bounds that staleness to a single case,
1002
+ # a window in which the OS cannot plausibly have recycled the number, while a record
1003
+ # whose wrapper is still running (the leak this all exists for) is kept.
1004
+ prune_dead_pgid_records
798
1005
  printf '%s' "$1" >"$WORK/state/claude_behavior"
799
1006
  printf '%s' "$2" >"$WORK/state/kimi_behavior"
800
1007
  printf '%s' "$3" >"$WORK/state/opencode_behavior"
@@ -2141,6 +2348,14 @@ out="$(cat "$escaped_out_file" 2>/dev/null || true)"
2141
2348
  escaped_elapsed="$(( $(date +%s) - escaped_started_at ))"
2142
2349
  escaped_child_pid="$(cat "$WORK/state/escaped_hang_child_pid" 2>/dev/null || true)"
2143
2350
  escaped_child_trap_pid="$escaped_child_pid"
2351
+ # Audit copy, in a directory neither this case nor reset_case clears: the working pid
2352
+ # file below is removed as part of this case's own cleanup, so a mutation that disables
2353
+ # the kill while leaving the removal in place would erase the only handle the exit
2354
+ # assertion could have used.
2355
+ mkdir -p "$WORK/state/audit" 2>/dev/null || true
2356
+ printf '%s %s\n' "$escaped_child_pid" \
2357
+ "$(ps -o lstart= -p "$escaped_child_pid" 2>/dev/null || true)" \
2358
+ >"$WORK/state/audit/escaped_pid" 2>/dev/null || true
2144
2359
  # Read the beat BEFORE the liveness probe so a descendant that dies between the two
2145
2360
  # still reports how far it got; empty means it never reached its own setsid.
2146
2361
  # Wait for the descendant to TESTIFY, not for time to pass: a live holder stamps
@@ -2403,6 +2618,35 @@ check "base packet freezes tracked and untracked changes once" \
2403
2618
  check "owner selection source has one controller-owned definition" \
2404
2619
  '[ "$(grep -c '\''"implementer-declared"'\'' "$DIR/review_gate.py")" = 1 ]'
2405
2620
 
2621
+ # The acceptance object for wrapper cleanup is a clean process table, not a green
2622
+ # case: every case above can pass while a wrapper the controller failed to reap is
2623
+ # still running. Read the ledger last, so a wrapper any case abandoned is named here
2624
+ # instead of surviving the run silently.
2625
+ # Settle first: the last cases' wrappers may still be tearing down when this line is
2626
+ # reached, and flagging a process that is already exiting makes this check flaky
2627
+ # rather than load-bearing (measured: one pid reported, already gone by the time the
2628
+ # diagnostic below ran). The bound stays far under the fixture's own 300s hang bound,
2629
+ # so a wrapper nobody reaped still fails here.
2630
+ leaked_settle_deadline="$(( $(date +%s) + 10 ))"
2631
+ while :; do
2632
+ leaked_wrappers="$(review_harness_pids_alive | tr '\n' ' ')"
2633
+ [ -z "$leaked_wrappers" ] && break
2634
+ [ "$(date +%s)" -ge "$leaked_settle_deadline" ] && break
2635
+ sleep 1
2636
+ done
2637
+ [ -z "$leaked_wrappers" ] || {
2638
+ printf 'leaked reviewer wrappers: %s\n' "$leaked_wrappers" >&2
2639
+ # One `ps` per pid: a space-separated list after a single -p relies on BSD operand
2640
+ # parsing and does not carry across ps implementations. Redirection ORDER is
2641
+ # load-bearing: `2>/dev/null >&2` points fd2 at /dev/null and then duplicates fd1 onto
2642
+ # THAT, sending the diagnostic to /dev/null — measured, three times, as a failure that
2643
+ # named pids it could not describe.
2644
+ for leaked_pid in $leaked_wrappers; do
2645
+ ps -o pid=,ppid=,etime=,stat=,command= -p "$leaked_pid" >&2 2>/dev/null || true
2646
+ done
2647
+ }
2648
+ check "the suite leaves no reviewer wrapper running" '[ -z "$leaked_wrappers" ]'
2649
+
2406
2650
  printf '%s\n' '----'
2407
2651
  if [ "$fails" -eq 0 ]; then
2408
2652
  echo review_gate_tests_ok