@ccoalm/ccl-skills 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +8 -11
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +9 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +29 -31
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +32 -0
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
  33. package/dist/assets/release.json +79 -24
  34. package/package.json +1 -1
@@ -0,0 +1,431 @@
1
+ #!/usr/bin/env bash
2
+ # Tests for gate_receipt.py — candidate-SHA-bound receipts of deterministic-gate
3
+ # output. Proves the mint preconditions (clean committed tree, out-of-tree
4
+ # receipt, never overwrite), the structural validator, and — critically — that
5
+ # the re-run verifier goes RED for the RIGHT reason under applied mutations
6
+ # (tampered output hash, tampered exit code, foreign key), and stays infra (rc 2,
7
+ # no verdict) when the environment cannot judge (wrong checked-out candidate,
8
+ # dirty tree). RED-run receipts (non-zero recorded exit) are first-class: the
9
+ # pre-fix RED claim is the canonical use case.
10
+ set -euo pipefail
11
+
12
+ SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
13
+ GR="$SCRIPT_DIR/gate_receipt.py"
14
+ [ -f "$GR" ] || { echo "FAIL: gate_receipt.py not found: $GR" >&2; exit 1; }
15
+
16
+ fail() { echo "FAIL: $*" >&2; exit 1; }
17
+ assert_rc() { [ "$1" = "$2" ] || fail "expected rc=$2 got rc=$1${3:+ ($3)}"; }
18
+ assert_contains() { case "$2" in *"$1"*) : ;; *) fail "expected output to contain: $1${3:+ ($3)}; got: $2";; esac; }
19
+
20
+ TMP="$(mktemp -d "${TMPDIR:-/tmp}/gatereceipt.XXXXXX")"
21
+ trap 'rm -rf "$TMP"' EXIT
22
+ REPO="$TMP/repo"; LEDGER="$TMP/ledger"
23
+ mkdir -p "$REPO" "$LEDGER"
24
+ git -C "$REPO" init -q
25
+ git -C "$REPO" config user.email test@example.invalid
26
+ git -C "$REPO" config user.name test
27
+ echo seed > "$REPO/f.txt"
28
+ git -C "$REPO" add f.txt
29
+ git -C "$REPO" commit -qm init
30
+
31
+ run_gr() { (cd "$REPO" && python3 "$GR" "$@"); }
32
+
33
+ # (1) Mint on a clean tree, green command; receipt lands out of tree.
34
+ set +e
35
+ out="$(run_gr mint --out "$LEDGER/green.json" -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
36
+ set -e
37
+ assert_rc "$rc" 0 "mint green"
38
+ assert_contains "gate_receipt_minted:" "$out" "mint green token"
39
+ head_sha="$(git -C "$REPO" rev-parse HEAD)"
40
+ assert_contains "candidate=$head_sha" "$out" "mint records HEAD"
41
+
42
+ # (2) Never overwrite an existing receipt.
43
+ set +e
44
+ out="$(run_gr mint --out "$LEDGER/green.json" -- true 2>&1)"; rc=$?
45
+ set -e
46
+ assert_rc "$rc" 2 "mint overwrite refused"
47
+ assert_contains "never overwritten" "$out" "overwrite message"
48
+
49
+ # (3) Refuse a dirty tree (a receipt binds a committed candidate only).
50
+ echo drift >> "$REPO/f.txt"
51
+ set +e
52
+ out="$(run_gr mint --out "$LEDGER/dirty.json" -- true 2>&1)"; rc=$?
53
+ set -e
54
+ assert_rc "$rc" 2 "mint dirty tree refused"
55
+ assert_contains "tree not clean" "$out" "dirty message"
56
+ git -C "$REPO" checkout -q f.txt
57
+
58
+ # (4) A RED run mints fine — recorded exit code, not required success.
59
+ set +e
60
+ out="$(run_gr mint --out "$LEDGER/red.json" -- bash -c 'echo failing; exit 3' 2>&1)"; rc=$?
61
+ set -e
62
+ assert_rc "$rc" 0 "mint red run"
63
+ assert_contains "exit=3" "$out" "red exit recorded"
64
+
65
+ # (5) Structural verify passes for both receipts.
66
+ set +e
67
+ out="$(run_gr verify "$LEDGER/green.json" 2>&1)"; rc=$?
68
+ set -e
69
+ assert_rc "$rc" 0 "structural green"
70
+ assert_contains "gate_receipt_structural_ok" "$out" "structural token"
71
+
72
+ # (6) Re-run verify passes: green receipt and red-recorded receipt alike.
73
+ set +e
74
+ out="$(run_gr verify "$LEDGER/green.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
75
+ set -e
76
+ assert_rc "$rc" 0 "rerun green"
77
+ assert_contains "scope=full" "$out" "rerun full scope"
78
+ set +e
79
+ out="$(run_gr verify "$LEDGER/red.json" --rerun -- bash -c 'echo failing; exit 3' 2>&1)"; rc=$?
80
+ set -e
81
+ assert_rc "$rc" 0 "rerun red-recorded"
82
+
83
+ # (7) APPLIED MUTATION: tampered output hash -> rc 1, named output_hash_mismatch.
84
+ python3 - "$LEDGER" <<'EOF'
85
+ import json, sys
86
+ ledger = sys.argv[1]
87
+ d = json.load(open(f"{ledger}/green.json"))
88
+ d["output_sha256"] = "0" * 64
89
+ open(f"{ledger}/tampered-hash.json", "w").write(
90
+ json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
91
+ EOF
92
+ set +e
93
+ out="$(run_gr verify "$LEDGER/tampered-hash.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
94
+ set -e
95
+ assert_rc "$rc" 1 "tampered hash"
96
+ assert_contains "output_hash_mismatch" "$out" "tampered hash reason"
97
+
98
+ # (8) APPLIED MUTATION: tampered exit code -> rc 1, named exit_code_mismatch.
99
+ python3 - "$LEDGER" <<'EOF'
100
+ import json, sys
101
+ ledger = sys.argv[1]
102
+ d = json.load(open(f"{ledger}/red.json"))
103
+ d["exit_code"] = 0
104
+ open(f"{ledger}/tampered-exit.json", "w").write(
105
+ json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
106
+ EOF
107
+ set +e
108
+ out="$(run_gr verify "$LEDGER/tampered-exit.json" --rerun -- bash -c 'echo failing; exit 3' 2>&1)"; rc=$?
109
+ set -e
110
+ assert_rc "$rc" 1 "tampered exit"
111
+ assert_contains "exit_code_mismatch" "$out" "tampered exit reason"
112
+
113
+ # (9) APPLIED MUTATION: foreign key -> rc 1 structural (exact key set enforced).
114
+ python3 - "$LEDGER" <<'EOF'
115
+ import json, sys
116
+ ledger = sys.argv[1]
117
+ d = json.load(open(f"{ledger}/green.json"))
118
+ d["note"] = "smuggled"
119
+ open(f"{ledger}/extra-key.json", "w").write(
120
+ json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
121
+ EOF
122
+ set +e
123
+ out="$(run_gr verify "$LEDGER/extra-key.json" 2>&1)"; rc=$?
124
+ set -e
125
+ assert_rc "$rc" 1 "extra key"
126
+ assert_contains "exactly the gate-receipt key set" "$out" "extra key reason"
127
+
128
+ # (10) Wrong checked-out candidate -> rc 2 (no verdict), named wrong_candidate.
129
+ echo advance >> "$REPO/f.txt"
130
+ git -C "$REPO" commit -qam advance
131
+ set +e
132
+ out="$(run_gr verify "$LEDGER/green.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
133
+ set -e
134
+ assert_rc "$rc" 2 "wrong candidate"
135
+ assert_contains "wrong_candidate" "$out" "wrong candidate reason"
136
+ git -C "$REPO" reset -q --hard "$head_sha"
137
+
138
+ # (11) Nondeterministic gate output: full rerun red, --exit-only green.
139
+ # (python3 secrets, not `date +%N`: BSD/macOS date prints a literal N, which
140
+ # would be deterministic and silently invert this case on those hosts.)
141
+ set +e
142
+ out="$(run_gr mint --out "$LEDGER/nondet.json" -- python3 -c 'import secrets; print(secrets.token_hex())' 2>&1)"; rc=$?
143
+ set -e
144
+ assert_rc "$rc" 0 "mint nondet"
145
+ set +e
146
+ out="$(run_gr verify "$LEDGER/nondet.json" --rerun -- python3 -c 'import secrets; print(secrets.token_hex())' 2>&1)"; rc=$?
147
+ set -e
148
+ assert_rc "$rc" 1 "nondet full rerun"
149
+ assert_contains "output_hash_mismatch" "$out" "nondet reason"
150
+ set +e
151
+ out="$(run_gr verify "$LEDGER/nondet.json" --rerun --exit-only -- python3 -c 'import secrets; print(secrets.token_hex())' 2>&1)"; rc=$?
152
+ set -e
153
+ assert_rc "$rc" 0 "nondet exit-only"
154
+ assert_contains "scope=exit-only" "$out" "exit-only scope token"
155
+
156
+ # (12) --exit-only without --rerun is a usage error, not a verdict.
157
+ set +e
158
+ out="$(run_gr verify "$LEDGER/green.json" --exit-only 2>&1)"; rc=$?
159
+ set -e
160
+ assert_rc "$rc" 2 "exit-only without rerun"
161
+
162
+ # (13) Dirty tree at the CORRECT candidate: rerun verification is rc 2 named
163
+ # dirty_tree (no verdict) — dirty state must never surface as a false rc 0/1.
164
+ echo drift >> "$REPO/f.txt"
165
+ set +e
166
+ out="$(run_gr verify "$LEDGER/green.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
167
+ set -e
168
+ assert_rc "$rc" 2 "dirty rerun"
169
+ assert_contains "dirty_tree" "$out" "dirty rerun reason"
170
+ git -C "$REPO" checkout -q f.txt
171
+
172
+ # (14) Signal-killed gate: minted with the shell convention 128+N, structurally
173
+ # valid, and a re-run (which signals itself again) matches. A raw negative
174
+ # Python returncode would make a legitimate RED receipt unusable.
175
+ set +e
176
+ out="$(run_gr mint --out "$LEDGER/signal.json" -- bash -c 'kill -KILL $$' 2>&1)"; rc=$?
177
+ set -e
178
+ assert_rc "$rc" 0 "mint signal-killed"
179
+ assert_contains "exit=137" "$out" "signal exit normalized to 128+9"
180
+ set +e
181
+ out="$(run_gr verify "$LEDGER/signal.json" 2>&1)"; rc=$?
182
+ set -e
183
+ assert_rc "$rc" 0 "signal receipt structural"
184
+ set +e
185
+ out="$(run_gr verify "$LEDGER/signal.json" --rerun -- bash -c 'kill -KILL $$' 2>&1)"; rc=$?
186
+ set -e
187
+ assert_rc "$rc" 0 "signal receipt rerun"
188
+
189
+ # (15) Silent hanging gate: --timeout kills the process group and mints
190
+ # nothing (rc 2), instead of blocking forever on a pipe with no output.
191
+ set +e
192
+ out="$(run_gr mint --out "$LEDGER/hang.json" --timeout 3 -- bash -c 'sleep 60' 2>&1)"; rc=$?
193
+ set -e
194
+ assert_rc "$rc" 2 "hanging gate timeout"
195
+ assert_contains "no receipt minted" "$out" "timeout message"
196
+ [ ! -f "$LEDGER/hang.json" ] || fail "timeout must not leave a receipt"
197
+
198
+ # (16) Candidate moved mid-run (the gate itself commits): refuse, no receipt.
199
+ set +e
200
+ out="$(run_gr mint --out "$LEDGER/moved.json" -- git commit -q --allow-empty -m mid-run 2>&1)"; rc=$?
201
+ set -e
202
+ assert_rc "$rc" 2 "mid-run HEAD move"
203
+ assert_contains "candidate changed during the run" "$out" "mid-run move reason"
204
+ [ ! -f "$LEDGER/moved.json" ] || fail "mid-run move must not leave a receipt"
205
+ git -C "$REPO" reset -q --hard "$head_sha"
206
+
207
+ # (17) Repository-relative cwd is recorded and re-runs execute from it:
208
+ # mint from a subdirectory, verify --rerun from the repository root.
209
+ mkdir -p "$REPO/sub"
210
+ ( cd "$REPO/sub" && python3 "$GR" mint --out "$LEDGER/subdir.json" -- bash -c 'cat ../f.txt' ) \
211
+ || fail "mint from subdirectory"
212
+ python3 - "$LEDGER/subdir.json" <<'EOF'
213
+ import json, sys
214
+ d = json.load(open(sys.argv[1]))
215
+ assert d["cwd"] == "sub", d["cwd"]
216
+ assert d["output_tail"] == "", "tail must default to empty"
217
+ EOF
218
+ set +e
219
+ out="$(run_gr verify "$LEDGER/subdir.json" --rerun -- bash -c 'cat ../f.txt' 2>&1)"; rc=$?
220
+ set -e
221
+ assert_rc "$rc" 0 "subdir receipt rerun from repo root"
222
+
223
+ # (18) Receipts are created 0600, and the default tail is empty even for a
224
+ # gate that prints output (nothing verbatim is copied without opt-in).
225
+ perms="$(python3 -c "import os,sys,stat; print(oct(stat.S_IMODE(os.stat(sys.argv[1]).st_mode)))" "$LEDGER/green.json")"
226
+ [ "$perms" = "0o600" ] || fail "receipt permissions must be 0600, got $perms"
227
+ python3 - "$LEDGER/green.json" <<'EOF'
228
+ import json, sys
229
+ d = json.load(open(sys.argv[1]))
230
+ assert d["output_tail"] == "", "default mint must not embed a plaintext tail"
231
+ EOF
232
+
233
+ # (19) Opt-in tail with maximally invalid UTF-8: replacement decoding must not
234
+ # inflate the stored tail past the cap — the minted receipt stays structurally
235
+ # valid.
236
+ set +e
237
+ out="$(run_gr mint --out "$LEDGER/invalid-utf8.json" --tail-bytes 16384 -- python3 -c 'import sys; sys.stdout.buffer.write(b"\xff" * 16384)' 2>&1)"; rc=$?
238
+ set -e
239
+ assert_rc "$rc" 0 "mint invalid-utf8 tail"
240
+ set +e
241
+ out="$(run_gr verify "$LEDGER/invalid-utf8.json" 2>&1)"; rc=$?
242
+ set -e
243
+ assert_rc "$rc" 0 "invalid-utf8 tail receipt structural"
244
+
245
+ # (20) The verifier's command is mandatory and compared: --rerun without a
246
+ # command refuses (rc 2, the recorded argv is untrusted input), and a receipt
247
+ # whose recorded command differs from what the verifier typed is rc 1
248
+ # command_mismatch — the recorded argv is NEVER executed.
249
+ set +e
250
+ out="$(run_gr verify "$LEDGER/green.json" --rerun 2>&1)"; rc=$?
251
+ set -e
252
+ assert_rc "$rc" 2 "rerun without command"
253
+ assert_contains "candidate-controlled" "$out" "rerun-without-command reason"
254
+ set +e
255
+ out="$(run_gr verify "$LEDGER/green.json" --rerun -- bash -c 'echo attacker' 2>&1)"; rc=$?
256
+ set -e
257
+ assert_rc "$rc" 1 "command mismatch"
258
+ assert_contains "command_mismatch" "$out" "command mismatch reason"
259
+
260
+ # (21) Receipts may not be minted INSIDE the candidate repository (they would
261
+ # dirty the tree after the cleanliness check ran, or hide under .git).
262
+ set +e
263
+ out="$(run_gr mint --out "$REPO/in-tree.json" -- true 2>&1)"; rc=$?
264
+ set -e
265
+ assert_rc "$rc" 2 "in-tree receipt refused"
266
+ assert_contains "inside the candidate repository" "$out" "in-tree reason"
267
+ set +e
268
+ out="$(run_gr mint --out "$REPO/.git/hidden.json" -- true 2>&1)"; rc=$?
269
+ set -e
270
+ assert_rc "$rc" 2 "under-.git receipt refused"
271
+
272
+ # (22) A receipt the tool's own verifier would reject as oversized is never
273
+ # minted: huge argv refuses with rc 2 and no file.
274
+ set +e
275
+ big_arg="$(python3 -c 'print("x" * 70000)')"
276
+ out="$(run_gr mint --out "$LEDGER/huge.json" -- bash -c true bash "$big_arg" 2>&1)"; rc=$?
277
+ set -e
278
+ assert_rc "$rc" 2 "oversized receipt refused"
279
+ [ ! -f "$LEDGER/huge.json" ] || fail "oversized mint must not leave a receipt"
280
+
281
+ # (23) An EMPTY later argument is legal argv: mint, structural verify, and
282
+ # rerun must all round-trip (only command[0] must be non-empty).
283
+ set +e
284
+ out="$(run_gr mint --out "$LEDGER/empty-arg.json" -- printf '%s' '' 2>&1)"; rc=$?
285
+ set -e
286
+ assert_rc "$rc" 0 "mint empty-arg"
287
+ set +e
288
+ out="$(run_gr verify "$LEDGER/empty-arg.json" 2>&1)"; rc=$?
289
+ set -e
290
+ assert_rc "$rc" 0 "empty-arg structural"
291
+ set +e
292
+ out="$(run_gr verify "$LEDGER/empty-arg.json" --rerun -- printf '%s' '' 2>&1)"; rc=$?
293
+ set -e
294
+ assert_rc "$rc" 0 "empty-arg rerun"
295
+
296
+ # (24) An OS-level failure (permission denied on the gate binary) is rc 2
297
+ # no-verdict — never the rc 1 the contract reserves for a failed receipt.
298
+ : > "$TMP/noexec"
299
+ chmod -x "$TMP/noexec"
300
+ set +e
301
+ out="$(run_gr mint --out "$LEDGER/noexec.json" -- "$TMP/noexec" 2>&1)"; rc=$?
302
+ set -e
303
+ assert_rc "$rc" 2 "permission-denied gate"
304
+
305
+ # (25) status.showUntrackedFiles=no cannot hide untracked gate inputs: the
306
+ # explicit --untracked-files=all flag sees them and mint refuses.
307
+ git -C "$REPO" config status.showUntrackedFiles no
308
+ echo hidden > "$REPO/untracked-input.txt"
309
+ set +e
310
+ out="$(run_gr mint --out "$LEDGER/hidden-untracked.json" -- true 2>&1)"; rc=$?
311
+ set -e
312
+ assert_rc "$rc" 2 "hidden untracked state"
313
+ assert_contains "tree not clean" "$out" "hidden untracked reason"
314
+ rm "$REPO/untracked-input.txt"
315
+ git -C "$REPO" config --unset status.showUntrackedFiles
316
+
317
+ # (26) A forged cwd that is an in-repo symlink resolving OUTSIDE the
318
+ # repository is refused before anything executes.
319
+ ln -s "$TMP" "$REPO/escape"
320
+ git -C "$REPO" add escape
321
+ git -C "$REPO" commit -qm add-escape-symlink
322
+ python3 - "$LEDGER" "$(git -C "$REPO" rev-parse HEAD)" <<'EOF'
323
+ import json, sys
324
+ ledger, head = sys.argv[1], sys.argv[2]
325
+ d = json.load(open(f"{ledger}/green.json"))
326
+ d["cwd"] = "escape"
327
+ d["candidate_commit"] = head
328
+ open(f"{ledger}/forged-cwd.json", "w").write(
329
+ json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
330
+ EOF
331
+ set +e
332
+ out="$(run_gr verify "$LEDGER/forged-cwd.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
333
+ set -e
334
+ assert_rc "$rc" 2 "forged escaping cwd"
335
+ assert_contains "resolves outside the repository" "$out" "escaping cwd reason"
336
+ git -C "$REPO" reset -q --hard "$head_sha"
337
+
338
+ # (27) An untracked filename with invalid UTF-8 bytes (core.quotePath=false)
339
+ # must yield rc 2 tree-not-clean — never an uncaught decode traceback exiting
340
+ # with the verdict-reserved rc 1. APFS refuses such names; skip the leg (with
341
+ # a printed marker) where the filesystem cannot produce the state.
342
+ git -C "$REPO" config core.quotePath false
343
+ if python3 -c 'import os,sys; os.close(os.open(os.path.join(sys.argv[1].encode(), b"\xff\xfebad"), os.O_CREAT|os.O_WRONLY, 0o644))' "$REPO" 2>/dev/null; then
344
+ set +e
345
+ out="$(run_gr mint --out "$LEDGER/badname.json" -- true 2>&1)"; rc=$?
346
+ set -e
347
+ assert_rc "$rc" 2 "invalid-utf8 filename"
348
+ case "$out" in *Traceback*) fail "decode error surfaced as a traceback";; esac
349
+ python3 -c 'import os,sys; os.unlink(os.path.join(sys.argv[1].encode(), b"\xff\xfebad"))' "$REPO"
350
+ else
351
+ echo "note: filesystem refuses invalid-UTF-8 names; case 27 leg skipped"
352
+ fi
353
+ git -C "$REPO" config --unset core.quotePath
354
+
355
+ # (28) APPLIED MUTATION: forged output_bytes with genuine hash/exit -> rc 1
356
+ # named output_bytes_mismatch (every recorded field is compared on full rerun).
357
+ python3 - "$LEDGER" <<'EOF'
358
+ import json, sys
359
+ ledger = sys.argv[1]
360
+ d = json.load(open(f"{ledger}/green.json"))
361
+ d["output_bytes"] = d["output_bytes"] + 7
362
+ open(f"{ledger}/forged-bytes.json", "w").write(
363
+ json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
364
+ EOF
365
+ set +e
366
+ out="$(run_gr verify "$LEDGER/forged-bytes.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
367
+ set -e
368
+ assert_rc "$rc" 1 "forged bytes"
369
+ assert_contains "output_bytes_mismatch" "$out" "forged bytes reason"
370
+
371
+ # (29) APPLIED MUTATION: forged human-readable tail -> rc 1 output_tail_mismatch;
372
+ # a genuine opt-in tail round-trips on full rerun.
373
+ set +e
374
+ out="$(run_gr mint --out "$LEDGER/tailed.json" --tail-bytes 4096 -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
375
+ set -e
376
+ assert_rc "$rc" 0 "mint tailed"
377
+ set +e
378
+ out="$(run_gr verify "$LEDGER/tailed.json" --rerun -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
379
+ set -e
380
+ assert_rc "$rc" 0 "genuine tail rerun"
381
+ python3 - "$LEDGER" <<'EOF'
382
+ import json, sys
383
+ ledger = sys.argv[1]
384
+ d = json.load(open(f"{ledger}/tailed.json"))
385
+ d["output_tail"] = "forged human-readable excerpt\n"
386
+ open(f"{ledger}/forged-tail.json", "w").write(
387
+ json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
388
+ EOF
389
+ set +e
390
+ out="$(run_gr verify "$LEDGER/forged-tail.json" --rerun -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
391
+ set -e
392
+ assert_rc "$rc" 1 "forged tail"
393
+ assert_contains "output_tail_mismatch" "$out" "forged tail reason"
394
+
395
+ # (30) A failed publish never wedges the slot: after an overwrite refusal the
396
+ # temp file is gone and the original receipt is intact.
397
+ set +e
398
+ out="$(run_gr mint --out "$LEDGER/green.json" -- true 2>&1)"; rc=$?
399
+ set -e
400
+ assert_rc "$rc" 2 "republish refused"
401
+ ls "$LEDGER"/green.json.tmp.* 2>/dev/null && fail "temp file left behind"
402
+ python3 -c 'import json,sys; json.load(open(sys.argv[1]))' "$LEDGER/green.json" || fail "original receipt corrupted"
403
+
404
+ # (31) APPLIED MUTATIONS on the tail binding: an EMPTIED tail and a TRUNCATED
405
+ # genuine suffix must both go red on full rerun — the tail is reconstructed at
406
+ # the recorded tail_bytes and compared exactly, so weakening the excerpt while
407
+ # keeping exit/hash cannot ride a full-verification verdict.
408
+ python3 - "$LEDGER" <<'EOF'
409
+ import json, sys
410
+ ledger = sys.argv[1]
411
+ d = json.load(open(f"{ledger}/tailed.json"))
412
+ d["output_tail"] = ""
413
+ open(f"{ledger}/emptied-tail.json", "w").write(
414
+ json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
415
+ d2 = json.load(open(f"{ledger}/tailed.json"))
416
+ d2["output_tail"] = d2["output_tail"][-5:]
417
+ open(f"{ledger}/truncated-tail.json", "w").write(
418
+ json.dumps(d2, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
419
+ EOF
420
+ set +e
421
+ out="$(run_gr verify "$LEDGER/emptied-tail.json" --rerun -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
422
+ set -e
423
+ assert_rc "$rc" 1 "emptied tail"
424
+ assert_contains "output_tail_mismatch" "$out" "emptied tail reason"
425
+ set +e
426
+ out="$(run_gr verify "$LEDGER/truncated-tail.json" --rerun -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
427
+ set -e
428
+ assert_rc "$rc" 1 "truncated tail"
429
+ assert_contains "output_tail_mismatch" "$out" "truncated tail reason"
430
+
431
+ echo "test_gate_receipt: ok"
@@ -0,0 +1,151 @@
1
+ #!/usr/bin/env bash
2
+ # Oracle self-proof walk for the pinned-phrase gate families in
3
+ # check-ccl-skills.sh (heavy lane: each leg runs the full shipped gate against
4
+ # a fixture clone of the working tree).
5
+ #
6
+ # Why this exists (074): the pinned-phrase families are straight-line bash
7
+ # loops; an empty phrase list, a quoting regression, or a broken exit path
8
+ # would turn every one of them into an always-green check with no signal.
9
+ # None of the ~40 pinned phrases had ever been proven capable of going red.
10
+ # This walk applies one deletion mutation per gate family and requires the
11
+ # full gate to fail RED with that family's own token (differential
12
+ # attribution: the mutant leg must fail on the mutated family, and the
13
+ # unmutated control leg must show every family token green).
14
+ #
15
+ # MUST-HIT (one applied mutation per family -> expected red token):
16
+ # W1 existing-project-assessment-report.md loses "Assessment launch checklist"
17
+ # -> project_assessment_template_gate_missing
18
+ # W2 skill-extraction-workflow/SKILL.md loses "would other teammates hit this"
19
+ # -> skill_extraction_teammate_trigger_gate_missing
20
+ # W3 testing-strategy/SKILL.md loses "先写测试用例"
21
+ # -> testing_strategy_test_case_first_gate_missing
22
+ # W4 product-rd-workflow/SKILL.md loses "### Pre-Final Continuation Gate"
23
+ # -> product_rd_entrypoint_anchor_gate_missing
24
+ # W5 contract-anchored reference loses its pinned discriminator sentence
25
+ # -> contract_anchor_missing (via the delegated contract-anchor gate)
26
+ #
27
+ # MUST-NOT-HIT (control): the unmutated fixture run exits 0 and prints every
28
+ # family green token (project_assessment_template_gate_ok,
29
+ # task_retro_memory_escape_gate_ok, test_case_first_gate_ok,
30
+ # product_rd_entrypoint_anchor_gate_ok, contract_anchor_gate_ok).
31
+ #
32
+ # Coverage boundary (stated, not silent): one pin per family is mutated, so
33
+ # the walk attests each family's parse/exit path, not every individual phrase;
34
+ # per-family phrase-list non-vacuity is separately asserted by counting the
35
+ # `for required_phrase in` loops and requiring each mutated phrase to exist in
36
+ # the fixture before mutation. ALIAS_AUDIT_CMD is unset for determinism (the
37
+ # gate then takes the public-fallback R0 branch on every host).
38
+ set -u
39
+
40
+ script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)
41
+ repo_root=$(cd "$script_dir/../../.." && pwd -P)
42
+ gate_rel="skills/skill-extraction-workflow/scripts/check-ccl-skills.sh"
43
+ fail=0
44
+
45
+ note() { printf '%s\n' "$*"; }
46
+
47
+ tmp=$(mktemp -d)
48
+ trap 'rm -rf "$tmp"' EXIT
49
+
50
+ # Fixture: faithful copy of the working tree (tracked + untracked-unignored),
51
+ # committed once so BASE_REF=HEAD yields an empty diff.
52
+ fixture="$tmp/fixture"
53
+ mkdir -p "$fixture"
54
+ (cd "$repo_root" && git ls-files --cached --others --exclude-standard -z \
55
+ | tar --null -T - -cf - ) | tar -xf - -C "$fixture"
56
+ git -C "$fixture" init -q
57
+ git -C "$fixture" -c user.name=fixture -c user.email=fixture@invalid add -A
58
+ git -C "$fixture" -c user.name=fixture -c user.email=fixture@invalid commit -qm fixture
59
+ # The impact-chain gate resolves merge-bases against origin/main and origin/dev;
60
+ # point both at the fixture's single commit so diff-scoped gates see an empty
61
+ # scope instead of dying on a missing remote ref.
62
+ git -C "$fixture" update-ref refs/remotes/origin/main HEAD
63
+ git -C "$fixture" update-ref refs/remotes/origin/dev HEAD
64
+
65
+ # Vacuity guard: the shipped gate must still carry its pinned-phrase loops.
66
+ loop_count=$(grep -c 'for required_phrase in' "$fixture/$gate_rel")
67
+ if [[ "$loop_count" -lt 7 ]]; then
68
+ note "FAIL vacuity-guard: expected >=7 'for required_phrase in' loops, found $loop_count"
69
+ fail=1
70
+ fi
71
+
72
+ run_gate() { # run_gate -> sets got_rc/got_out (full gate inside the fixture)
73
+ # CCL_SKILL_BASE_REF is deliberately NOT set: it would leak into child
74
+ # validators' synthetic self-test repos (where HEAD always resolves and
75
+ # turns their "no base -> degraded" legs into false passes). The fixture has
76
+ # a single commit, so diff-scoped gates see an empty scope and the run lands
77
+ # interim — the pinned-phrase families under test are tree scans and run
78
+ # fully either way.
79
+ got_out=$(cd "$fixture" && env -u ALIAS_AUDIT_CMD -u CCL_SKILL_BASE_REF \
80
+ bash "$gate_rel" . 2>&1)
81
+ got_rc=$?
82
+ }
83
+
84
+ restore_fixture() {
85
+ git -C "$fixture" checkout -q -- .
86
+ }
87
+
88
+ mutate() { # mutate <repo-relative-file> <exact-phrase-to-delete>
89
+ local file="$fixture/$1" phrase="$2"
90
+ if ! grep -qF -- "$phrase" "$file"; then
91
+ note "FAIL pre-mutation: phrase not present (walk out of sync): $phrase"
92
+ fail=1
93
+ return 1
94
+ fi
95
+ PHRASE="$phrase" perl -pi -e 's/\Q$ENV{PHRASE}\E/mutated-away/g' "$file"
96
+ }
97
+
98
+ walk() { # walk <case> <file> <phrase> <expected-red-token>
99
+ local case_id="$1" file="$2" phrase="$3" token="$4"
100
+ mutate "$file" "$phrase" || return
101
+ run_gate
102
+ if [[ "$got_rc" -eq 0 ]]; then
103
+ note "FAIL $case_id: gate stayed green under applied mutation ($token never fired)"
104
+ fail=1
105
+ elif [[ "$got_out" != *"$token"* ]]; then
106
+ note "FAIL $case_id: gate red but wrong reason (wanted $token)"
107
+ note "$(tail -n 5 <<<"$got_out")"
108
+ fail=1
109
+ else
110
+ note "ok $case_id ($token)"
111
+ fi
112
+ restore_fixture
113
+ }
114
+
115
+ # Control leg first: unmutated fixture must be green with every family token.
116
+ run_gate
117
+ if [[ "$got_rc" -ne 0 ]]; then
118
+ note "FAIL control: unmutated fixture gate rc=$got_rc"
119
+ note "$(tail -n 10 <<<"$got_out")"
120
+ fail=1
121
+ else
122
+ for token in \
123
+ project_assessment_template_gate_ok \
124
+ task_retro_memory_escape_gate_ok \
125
+ test_case_first_gate_ok \
126
+ product_rd_entrypoint_anchor_gate_ok \
127
+ "contract_anchor_gate_ok ("; do
128
+ if [[ "$got_out" != *"$token"* ]]; then
129
+ note "FAIL control: green run missing family token $token"
130
+ fail=1
131
+ fi
132
+ done
133
+ [[ "$fail" -eq 0 ]] && note "ok control (all family tokens green)"
134
+ fi
135
+
136
+ walk W1 "skills/product-rd-workflow/references/existing-project-assessment-report.md" \
137
+ "Assessment launch checklist" "project_assessment_template_gate_missing"
138
+ walk W2 "skills/skill-extraction-workflow/SKILL.md" \
139
+ "would other teammates hit this" "skill_extraction_teammate_trigger_gate_missing"
140
+ walk W3 "skills/testing-strategy/SKILL.md" \
141
+ "先写测试用例" "testing_strategy_test_case_first_gate_missing"
142
+ walk W4 "skills/product-rd-workflow/SKILL.md" \
143
+ "### Pre-Final Continuation Gate" "product_rd_entrypoint_anchor_gate_missing"
144
+ walk W5 "skills/testing-strategy/references/ci-fixtures-and-flake-control.md" \
145
+ "One discriminating predicate decides the verdict" "contract_anchor_missing"
146
+
147
+ if [[ "$fail" -ne 0 ]]; then
148
+ echo "test_pinned_phrase_mutation_walk: FAIL"
149
+ exit 1
150
+ fi
151
+ echo "test_pinned_phrase_mutation_walk: ok"