@ccoalm/ccl-skills 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +8 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +29 -31
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +32 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
- package/dist/assets/release.json +79 -24
- package/package.json +1 -1
|
@@ -0,0 +1,431 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Tests for gate_receipt.py — candidate-SHA-bound receipts of deterministic-gate
|
|
3
|
+
# output. Proves the mint preconditions (clean committed tree, out-of-tree
|
|
4
|
+
# receipt, never overwrite), the structural validator, and — critically — that
|
|
5
|
+
# the re-run verifier goes RED for the RIGHT reason under applied mutations
|
|
6
|
+
# (tampered output hash, tampered exit code, foreign key), and stays infra (rc 2,
|
|
7
|
+
# no verdict) when the environment cannot judge (wrong checked-out candidate,
|
|
8
|
+
# dirty tree). RED-run receipts (non-zero recorded exit) are first-class: the
|
|
9
|
+
# pre-fix RED claim is the canonical use case.
|
|
10
|
+
set -euo pipefail
|
|
11
|
+
|
|
12
|
+
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
13
|
+
GR="$SCRIPT_DIR/gate_receipt.py"
|
|
14
|
+
[ -f "$GR" ] || { echo "FAIL: gate_receipt.py not found: $GR" >&2; exit 1; }
|
|
15
|
+
|
|
16
|
+
fail() { echo "FAIL: $*" >&2; exit 1; }
|
|
17
|
+
assert_rc() { [ "$1" = "$2" ] || fail "expected rc=$2 got rc=$1${3:+ ($3)}"; }
|
|
18
|
+
assert_contains() { case "$2" in *"$1"*) : ;; *) fail "expected output to contain: $1${3:+ ($3)}; got: $2";; esac; }
|
|
19
|
+
|
|
20
|
+
TMP="$(mktemp -d "${TMPDIR:-/tmp}/gatereceipt.XXXXXX")"
|
|
21
|
+
trap 'rm -rf "$TMP"' EXIT
|
|
22
|
+
REPO="$TMP/repo"; LEDGER="$TMP/ledger"
|
|
23
|
+
mkdir -p "$REPO" "$LEDGER"
|
|
24
|
+
git -C "$REPO" init -q
|
|
25
|
+
git -C "$REPO" config user.email test@example.invalid
|
|
26
|
+
git -C "$REPO" config user.name test
|
|
27
|
+
echo seed > "$REPO/f.txt"
|
|
28
|
+
git -C "$REPO" add f.txt
|
|
29
|
+
git -C "$REPO" commit -qm init
|
|
30
|
+
|
|
31
|
+
run_gr() { (cd "$REPO" && python3 "$GR" "$@"); }
|
|
32
|
+
|
|
33
|
+
# (1) Mint on a clean tree, green command; receipt lands out of tree.
|
|
34
|
+
set +e
|
|
35
|
+
out="$(run_gr mint --out "$LEDGER/green.json" -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
|
|
36
|
+
set -e
|
|
37
|
+
assert_rc "$rc" 0 "mint green"
|
|
38
|
+
assert_contains "gate_receipt_minted:" "$out" "mint green token"
|
|
39
|
+
head_sha="$(git -C "$REPO" rev-parse HEAD)"
|
|
40
|
+
assert_contains "candidate=$head_sha" "$out" "mint records HEAD"
|
|
41
|
+
|
|
42
|
+
# (2) Never overwrite an existing receipt.
|
|
43
|
+
set +e
|
|
44
|
+
out="$(run_gr mint --out "$LEDGER/green.json" -- true 2>&1)"; rc=$?
|
|
45
|
+
set -e
|
|
46
|
+
assert_rc "$rc" 2 "mint overwrite refused"
|
|
47
|
+
assert_contains "never overwritten" "$out" "overwrite message"
|
|
48
|
+
|
|
49
|
+
# (3) Refuse a dirty tree (a receipt binds a committed candidate only).
|
|
50
|
+
echo drift >> "$REPO/f.txt"
|
|
51
|
+
set +e
|
|
52
|
+
out="$(run_gr mint --out "$LEDGER/dirty.json" -- true 2>&1)"; rc=$?
|
|
53
|
+
set -e
|
|
54
|
+
assert_rc "$rc" 2 "mint dirty tree refused"
|
|
55
|
+
assert_contains "tree not clean" "$out" "dirty message"
|
|
56
|
+
git -C "$REPO" checkout -q f.txt
|
|
57
|
+
|
|
58
|
+
# (4) A RED run mints fine — recorded exit code, not required success.
|
|
59
|
+
set +e
|
|
60
|
+
out="$(run_gr mint --out "$LEDGER/red.json" -- bash -c 'echo failing; exit 3' 2>&1)"; rc=$?
|
|
61
|
+
set -e
|
|
62
|
+
assert_rc "$rc" 0 "mint red run"
|
|
63
|
+
assert_contains "exit=3" "$out" "red exit recorded"
|
|
64
|
+
|
|
65
|
+
# (5) Structural verify passes for both receipts.
|
|
66
|
+
set +e
|
|
67
|
+
out="$(run_gr verify "$LEDGER/green.json" 2>&1)"; rc=$?
|
|
68
|
+
set -e
|
|
69
|
+
assert_rc "$rc" 0 "structural green"
|
|
70
|
+
assert_contains "gate_receipt_structural_ok" "$out" "structural token"
|
|
71
|
+
|
|
72
|
+
# (6) Re-run verify passes: green receipt and red-recorded receipt alike.
|
|
73
|
+
set +e
|
|
74
|
+
out="$(run_gr verify "$LEDGER/green.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
|
|
75
|
+
set -e
|
|
76
|
+
assert_rc "$rc" 0 "rerun green"
|
|
77
|
+
assert_contains "scope=full" "$out" "rerun full scope"
|
|
78
|
+
set +e
|
|
79
|
+
out="$(run_gr verify "$LEDGER/red.json" --rerun -- bash -c 'echo failing; exit 3' 2>&1)"; rc=$?
|
|
80
|
+
set -e
|
|
81
|
+
assert_rc "$rc" 0 "rerun red-recorded"
|
|
82
|
+
|
|
83
|
+
# (7) APPLIED MUTATION: tampered output hash -> rc 1, named output_hash_mismatch.
|
|
84
|
+
python3 - "$LEDGER" <<'EOF'
|
|
85
|
+
import json, sys
|
|
86
|
+
ledger = sys.argv[1]
|
|
87
|
+
d = json.load(open(f"{ledger}/green.json"))
|
|
88
|
+
d["output_sha256"] = "0" * 64
|
|
89
|
+
open(f"{ledger}/tampered-hash.json", "w").write(
|
|
90
|
+
json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
|
|
91
|
+
EOF
|
|
92
|
+
set +e
|
|
93
|
+
out="$(run_gr verify "$LEDGER/tampered-hash.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
|
|
94
|
+
set -e
|
|
95
|
+
assert_rc "$rc" 1 "tampered hash"
|
|
96
|
+
assert_contains "output_hash_mismatch" "$out" "tampered hash reason"
|
|
97
|
+
|
|
98
|
+
# (8) APPLIED MUTATION: tampered exit code -> rc 1, named exit_code_mismatch.
|
|
99
|
+
python3 - "$LEDGER" <<'EOF'
|
|
100
|
+
import json, sys
|
|
101
|
+
ledger = sys.argv[1]
|
|
102
|
+
d = json.load(open(f"{ledger}/red.json"))
|
|
103
|
+
d["exit_code"] = 0
|
|
104
|
+
open(f"{ledger}/tampered-exit.json", "w").write(
|
|
105
|
+
json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
|
|
106
|
+
EOF
|
|
107
|
+
set +e
|
|
108
|
+
out="$(run_gr verify "$LEDGER/tampered-exit.json" --rerun -- bash -c 'echo failing; exit 3' 2>&1)"; rc=$?
|
|
109
|
+
set -e
|
|
110
|
+
assert_rc "$rc" 1 "tampered exit"
|
|
111
|
+
assert_contains "exit_code_mismatch" "$out" "tampered exit reason"
|
|
112
|
+
|
|
113
|
+
# (9) APPLIED MUTATION: foreign key -> rc 1 structural (exact key set enforced).
|
|
114
|
+
python3 - "$LEDGER" <<'EOF'
|
|
115
|
+
import json, sys
|
|
116
|
+
ledger = sys.argv[1]
|
|
117
|
+
d = json.load(open(f"{ledger}/green.json"))
|
|
118
|
+
d["note"] = "smuggled"
|
|
119
|
+
open(f"{ledger}/extra-key.json", "w").write(
|
|
120
|
+
json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
|
|
121
|
+
EOF
|
|
122
|
+
set +e
|
|
123
|
+
out="$(run_gr verify "$LEDGER/extra-key.json" 2>&1)"; rc=$?
|
|
124
|
+
set -e
|
|
125
|
+
assert_rc "$rc" 1 "extra key"
|
|
126
|
+
assert_contains "exactly the gate-receipt key set" "$out" "extra key reason"
|
|
127
|
+
|
|
128
|
+
# (10) Wrong checked-out candidate -> rc 2 (no verdict), named wrong_candidate.
|
|
129
|
+
echo advance >> "$REPO/f.txt"
|
|
130
|
+
git -C "$REPO" commit -qam advance
|
|
131
|
+
set +e
|
|
132
|
+
out="$(run_gr verify "$LEDGER/green.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
|
|
133
|
+
set -e
|
|
134
|
+
assert_rc "$rc" 2 "wrong candidate"
|
|
135
|
+
assert_contains "wrong_candidate" "$out" "wrong candidate reason"
|
|
136
|
+
git -C "$REPO" reset -q --hard "$head_sha"
|
|
137
|
+
|
|
138
|
+
# (11) Nondeterministic gate output: full rerun red, --exit-only green.
|
|
139
|
+
# (python3 secrets, not `date +%N`: BSD/macOS date prints a literal N, which
|
|
140
|
+
# would be deterministic and silently invert this case on those hosts.)
|
|
141
|
+
set +e
|
|
142
|
+
out="$(run_gr mint --out "$LEDGER/nondet.json" -- python3 -c 'import secrets; print(secrets.token_hex())' 2>&1)"; rc=$?
|
|
143
|
+
set -e
|
|
144
|
+
assert_rc "$rc" 0 "mint nondet"
|
|
145
|
+
set +e
|
|
146
|
+
out="$(run_gr verify "$LEDGER/nondet.json" --rerun -- python3 -c 'import secrets; print(secrets.token_hex())' 2>&1)"; rc=$?
|
|
147
|
+
set -e
|
|
148
|
+
assert_rc "$rc" 1 "nondet full rerun"
|
|
149
|
+
assert_contains "output_hash_mismatch" "$out" "nondet reason"
|
|
150
|
+
set +e
|
|
151
|
+
out="$(run_gr verify "$LEDGER/nondet.json" --rerun --exit-only -- python3 -c 'import secrets; print(secrets.token_hex())' 2>&1)"; rc=$?
|
|
152
|
+
set -e
|
|
153
|
+
assert_rc "$rc" 0 "nondet exit-only"
|
|
154
|
+
assert_contains "scope=exit-only" "$out" "exit-only scope token"
|
|
155
|
+
|
|
156
|
+
# (12) --exit-only without --rerun is a usage error, not a verdict.
|
|
157
|
+
set +e
|
|
158
|
+
out="$(run_gr verify "$LEDGER/green.json" --exit-only 2>&1)"; rc=$?
|
|
159
|
+
set -e
|
|
160
|
+
assert_rc "$rc" 2 "exit-only without rerun"
|
|
161
|
+
|
|
162
|
+
# (13) Dirty tree at the CORRECT candidate: rerun verification is rc 2 named
|
|
163
|
+
# dirty_tree (no verdict) — dirty state must never surface as a false rc 0/1.
|
|
164
|
+
echo drift >> "$REPO/f.txt"
|
|
165
|
+
set +e
|
|
166
|
+
out="$(run_gr verify "$LEDGER/green.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
|
|
167
|
+
set -e
|
|
168
|
+
assert_rc "$rc" 2 "dirty rerun"
|
|
169
|
+
assert_contains "dirty_tree" "$out" "dirty rerun reason"
|
|
170
|
+
git -C "$REPO" checkout -q f.txt
|
|
171
|
+
|
|
172
|
+
# (14) Signal-killed gate: minted with the shell convention 128+N, structurally
|
|
173
|
+
# valid, and a re-run (which signals itself again) matches. A raw negative
|
|
174
|
+
# Python returncode would make a legitimate RED receipt unusable.
|
|
175
|
+
set +e
|
|
176
|
+
out="$(run_gr mint --out "$LEDGER/signal.json" -- bash -c 'kill -KILL $$' 2>&1)"; rc=$?
|
|
177
|
+
set -e
|
|
178
|
+
assert_rc "$rc" 0 "mint signal-killed"
|
|
179
|
+
assert_contains "exit=137" "$out" "signal exit normalized to 128+9"
|
|
180
|
+
set +e
|
|
181
|
+
out="$(run_gr verify "$LEDGER/signal.json" 2>&1)"; rc=$?
|
|
182
|
+
set -e
|
|
183
|
+
assert_rc "$rc" 0 "signal receipt structural"
|
|
184
|
+
set +e
|
|
185
|
+
out="$(run_gr verify "$LEDGER/signal.json" --rerun -- bash -c 'kill -KILL $$' 2>&1)"; rc=$?
|
|
186
|
+
set -e
|
|
187
|
+
assert_rc "$rc" 0 "signal receipt rerun"
|
|
188
|
+
|
|
189
|
+
# (15) Silent hanging gate: --timeout kills the process group and mints
|
|
190
|
+
# nothing (rc 2), instead of blocking forever on a pipe with no output.
|
|
191
|
+
set +e
|
|
192
|
+
out="$(run_gr mint --out "$LEDGER/hang.json" --timeout 3 -- bash -c 'sleep 60' 2>&1)"; rc=$?
|
|
193
|
+
set -e
|
|
194
|
+
assert_rc "$rc" 2 "hanging gate timeout"
|
|
195
|
+
assert_contains "no receipt minted" "$out" "timeout message"
|
|
196
|
+
[ ! -f "$LEDGER/hang.json" ] || fail "timeout must not leave a receipt"
|
|
197
|
+
|
|
198
|
+
# (16) Candidate moved mid-run (the gate itself commits): refuse, no receipt.
|
|
199
|
+
set +e
|
|
200
|
+
out="$(run_gr mint --out "$LEDGER/moved.json" -- git commit -q --allow-empty -m mid-run 2>&1)"; rc=$?
|
|
201
|
+
set -e
|
|
202
|
+
assert_rc "$rc" 2 "mid-run HEAD move"
|
|
203
|
+
assert_contains "candidate changed during the run" "$out" "mid-run move reason"
|
|
204
|
+
[ ! -f "$LEDGER/moved.json" ] || fail "mid-run move must not leave a receipt"
|
|
205
|
+
git -C "$REPO" reset -q --hard "$head_sha"
|
|
206
|
+
|
|
207
|
+
# (17) Repository-relative cwd is recorded and re-runs execute from it:
|
|
208
|
+
# mint from a subdirectory, verify --rerun from the repository root.
|
|
209
|
+
mkdir -p "$REPO/sub"
|
|
210
|
+
( cd "$REPO/sub" && python3 "$GR" mint --out "$LEDGER/subdir.json" -- bash -c 'cat ../f.txt' ) \
|
|
211
|
+
|| fail "mint from subdirectory"
|
|
212
|
+
python3 - "$LEDGER/subdir.json" <<'EOF'
|
|
213
|
+
import json, sys
|
|
214
|
+
d = json.load(open(sys.argv[1]))
|
|
215
|
+
assert d["cwd"] == "sub", d["cwd"]
|
|
216
|
+
assert d["output_tail"] == "", "tail must default to empty"
|
|
217
|
+
EOF
|
|
218
|
+
set +e
|
|
219
|
+
out="$(run_gr verify "$LEDGER/subdir.json" --rerun -- bash -c 'cat ../f.txt' 2>&1)"; rc=$?
|
|
220
|
+
set -e
|
|
221
|
+
assert_rc "$rc" 0 "subdir receipt rerun from repo root"
|
|
222
|
+
|
|
223
|
+
# (18) Receipts are created 0600, and the default tail is empty even for a
|
|
224
|
+
# gate that prints output (nothing verbatim is copied without opt-in).
|
|
225
|
+
perms="$(python3 -c "import os,sys,stat; print(oct(stat.S_IMODE(os.stat(sys.argv[1]).st_mode)))" "$LEDGER/green.json")"
|
|
226
|
+
[ "$perms" = "0o600" ] || fail "receipt permissions must be 0600, got $perms"
|
|
227
|
+
python3 - "$LEDGER/green.json" <<'EOF'
|
|
228
|
+
import json, sys
|
|
229
|
+
d = json.load(open(sys.argv[1]))
|
|
230
|
+
assert d["output_tail"] == "", "default mint must not embed a plaintext tail"
|
|
231
|
+
EOF
|
|
232
|
+
|
|
233
|
+
# (19) Opt-in tail with maximally invalid UTF-8: replacement decoding must not
|
|
234
|
+
# inflate the stored tail past the cap — the minted receipt stays structurally
|
|
235
|
+
# valid.
|
|
236
|
+
set +e
|
|
237
|
+
out="$(run_gr mint --out "$LEDGER/invalid-utf8.json" --tail-bytes 16384 -- python3 -c 'import sys; sys.stdout.buffer.write(b"\xff" * 16384)' 2>&1)"; rc=$?
|
|
238
|
+
set -e
|
|
239
|
+
assert_rc "$rc" 0 "mint invalid-utf8 tail"
|
|
240
|
+
set +e
|
|
241
|
+
out="$(run_gr verify "$LEDGER/invalid-utf8.json" 2>&1)"; rc=$?
|
|
242
|
+
set -e
|
|
243
|
+
assert_rc "$rc" 0 "invalid-utf8 tail receipt structural"
|
|
244
|
+
|
|
245
|
+
# (20) The verifier's command is mandatory and compared: --rerun without a
|
|
246
|
+
# command refuses (rc 2, the recorded argv is untrusted input), and a receipt
|
|
247
|
+
# whose recorded command differs from what the verifier typed is rc 1
|
|
248
|
+
# command_mismatch — the recorded argv is NEVER executed.
|
|
249
|
+
set +e
|
|
250
|
+
out="$(run_gr verify "$LEDGER/green.json" --rerun 2>&1)"; rc=$?
|
|
251
|
+
set -e
|
|
252
|
+
assert_rc "$rc" 2 "rerun without command"
|
|
253
|
+
assert_contains "candidate-controlled" "$out" "rerun-without-command reason"
|
|
254
|
+
set +e
|
|
255
|
+
out="$(run_gr verify "$LEDGER/green.json" --rerun -- bash -c 'echo attacker' 2>&1)"; rc=$?
|
|
256
|
+
set -e
|
|
257
|
+
assert_rc "$rc" 1 "command mismatch"
|
|
258
|
+
assert_contains "command_mismatch" "$out" "command mismatch reason"
|
|
259
|
+
|
|
260
|
+
# (21) Receipts may not be minted INSIDE the candidate repository (they would
|
|
261
|
+
# dirty the tree after the cleanliness check ran, or hide under .git).
|
|
262
|
+
set +e
|
|
263
|
+
out="$(run_gr mint --out "$REPO/in-tree.json" -- true 2>&1)"; rc=$?
|
|
264
|
+
set -e
|
|
265
|
+
assert_rc "$rc" 2 "in-tree receipt refused"
|
|
266
|
+
assert_contains "inside the candidate repository" "$out" "in-tree reason"
|
|
267
|
+
set +e
|
|
268
|
+
out="$(run_gr mint --out "$REPO/.git/hidden.json" -- true 2>&1)"; rc=$?
|
|
269
|
+
set -e
|
|
270
|
+
assert_rc "$rc" 2 "under-.git receipt refused"
|
|
271
|
+
|
|
272
|
+
# (22) A receipt the tool's own verifier would reject as oversized is never
|
|
273
|
+
# minted: huge argv refuses with rc 2 and no file.
|
|
274
|
+
set +e
|
|
275
|
+
big_arg="$(python3 -c 'print("x" * 70000)')"
|
|
276
|
+
out="$(run_gr mint --out "$LEDGER/huge.json" -- bash -c true bash "$big_arg" 2>&1)"; rc=$?
|
|
277
|
+
set -e
|
|
278
|
+
assert_rc "$rc" 2 "oversized receipt refused"
|
|
279
|
+
[ ! -f "$LEDGER/huge.json" ] || fail "oversized mint must not leave a receipt"
|
|
280
|
+
|
|
281
|
+
# (23) An EMPTY later argument is legal argv: mint, structural verify, and
|
|
282
|
+
# rerun must all round-trip (only command[0] must be non-empty).
|
|
283
|
+
set +e
|
|
284
|
+
out="$(run_gr mint --out "$LEDGER/empty-arg.json" -- printf '%s' '' 2>&1)"; rc=$?
|
|
285
|
+
set -e
|
|
286
|
+
assert_rc "$rc" 0 "mint empty-arg"
|
|
287
|
+
set +e
|
|
288
|
+
out="$(run_gr verify "$LEDGER/empty-arg.json" 2>&1)"; rc=$?
|
|
289
|
+
set -e
|
|
290
|
+
assert_rc "$rc" 0 "empty-arg structural"
|
|
291
|
+
set +e
|
|
292
|
+
out="$(run_gr verify "$LEDGER/empty-arg.json" --rerun -- printf '%s' '' 2>&1)"; rc=$?
|
|
293
|
+
set -e
|
|
294
|
+
assert_rc "$rc" 0 "empty-arg rerun"
|
|
295
|
+
|
|
296
|
+
# (24) An OS-level failure (permission denied on the gate binary) is rc 2
|
|
297
|
+
# no-verdict — never the rc 1 the contract reserves for a failed receipt.
|
|
298
|
+
: > "$TMP/noexec"
|
|
299
|
+
chmod -x "$TMP/noexec"
|
|
300
|
+
set +e
|
|
301
|
+
out="$(run_gr mint --out "$LEDGER/noexec.json" -- "$TMP/noexec" 2>&1)"; rc=$?
|
|
302
|
+
set -e
|
|
303
|
+
assert_rc "$rc" 2 "permission-denied gate"
|
|
304
|
+
|
|
305
|
+
# (25) status.showUntrackedFiles=no cannot hide untracked gate inputs: the
|
|
306
|
+
# explicit --untracked-files=all flag sees them and mint refuses.
|
|
307
|
+
git -C "$REPO" config status.showUntrackedFiles no
|
|
308
|
+
echo hidden > "$REPO/untracked-input.txt"
|
|
309
|
+
set +e
|
|
310
|
+
out="$(run_gr mint --out "$LEDGER/hidden-untracked.json" -- true 2>&1)"; rc=$?
|
|
311
|
+
set -e
|
|
312
|
+
assert_rc "$rc" 2 "hidden untracked state"
|
|
313
|
+
assert_contains "tree not clean" "$out" "hidden untracked reason"
|
|
314
|
+
rm "$REPO/untracked-input.txt"
|
|
315
|
+
git -C "$REPO" config --unset status.showUntrackedFiles
|
|
316
|
+
|
|
317
|
+
# (26) A forged cwd that is an in-repo symlink resolving OUTSIDE the
|
|
318
|
+
# repository is refused before anything executes.
|
|
319
|
+
ln -s "$TMP" "$REPO/escape"
|
|
320
|
+
git -C "$REPO" add escape
|
|
321
|
+
git -C "$REPO" commit -qm add-escape-symlink
|
|
322
|
+
python3 - "$LEDGER" "$(git -C "$REPO" rev-parse HEAD)" <<'EOF'
|
|
323
|
+
import json, sys
|
|
324
|
+
ledger, head = sys.argv[1], sys.argv[2]
|
|
325
|
+
d = json.load(open(f"{ledger}/green.json"))
|
|
326
|
+
d["cwd"] = "escape"
|
|
327
|
+
d["candidate_commit"] = head
|
|
328
|
+
open(f"{ledger}/forged-cwd.json", "w").write(
|
|
329
|
+
json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
|
|
330
|
+
EOF
|
|
331
|
+
set +e
|
|
332
|
+
out="$(run_gr verify "$LEDGER/forged-cwd.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
|
|
333
|
+
set -e
|
|
334
|
+
assert_rc "$rc" 2 "forged escaping cwd"
|
|
335
|
+
assert_contains "resolves outside the repository" "$out" "escaping cwd reason"
|
|
336
|
+
git -C "$REPO" reset -q --hard "$head_sha"
|
|
337
|
+
|
|
338
|
+
# (27) An untracked filename with invalid UTF-8 bytes (core.quotePath=false)
|
|
339
|
+
# must yield rc 2 tree-not-clean — never an uncaught decode traceback exiting
|
|
340
|
+
# with the verdict-reserved rc 1. APFS refuses such names; skip the leg (with
|
|
341
|
+
# a printed marker) where the filesystem cannot produce the state.
|
|
342
|
+
git -C "$REPO" config core.quotePath false
|
|
343
|
+
if python3 -c 'import os,sys; os.close(os.open(os.path.join(sys.argv[1].encode(), b"\xff\xfebad"), os.O_CREAT|os.O_WRONLY, 0o644))' "$REPO" 2>/dev/null; then
|
|
344
|
+
set +e
|
|
345
|
+
out="$(run_gr mint --out "$LEDGER/badname.json" -- true 2>&1)"; rc=$?
|
|
346
|
+
set -e
|
|
347
|
+
assert_rc "$rc" 2 "invalid-utf8 filename"
|
|
348
|
+
case "$out" in *Traceback*) fail "decode error surfaced as a traceback";; esac
|
|
349
|
+
python3 -c 'import os,sys; os.unlink(os.path.join(sys.argv[1].encode(), b"\xff\xfebad"))' "$REPO"
|
|
350
|
+
else
|
|
351
|
+
echo "note: filesystem refuses invalid-UTF-8 names; case 27 leg skipped"
|
|
352
|
+
fi
|
|
353
|
+
git -C "$REPO" config --unset core.quotePath
|
|
354
|
+
|
|
355
|
+
# (28) APPLIED MUTATION: forged output_bytes with genuine hash/exit -> rc 1
|
|
356
|
+
# named output_bytes_mismatch (every recorded field is compared on full rerun).
|
|
357
|
+
python3 - "$LEDGER" <<'EOF'
|
|
358
|
+
import json, sys
|
|
359
|
+
ledger = sys.argv[1]
|
|
360
|
+
d = json.load(open(f"{ledger}/green.json"))
|
|
361
|
+
d["output_bytes"] = d["output_bytes"] + 7
|
|
362
|
+
open(f"{ledger}/forged-bytes.json", "w").write(
|
|
363
|
+
json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
|
|
364
|
+
EOF
|
|
365
|
+
set +e
|
|
366
|
+
out="$(run_gr verify "$LEDGER/forged-bytes.json" --rerun -- bash -c 'echo gate-output; exit 0' 2>&1)"; rc=$?
|
|
367
|
+
set -e
|
|
368
|
+
assert_rc "$rc" 1 "forged bytes"
|
|
369
|
+
assert_contains "output_bytes_mismatch" "$out" "forged bytes reason"
|
|
370
|
+
|
|
371
|
+
# (29) APPLIED MUTATION: forged human-readable tail -> rc 1 output_tail_mismatch;
|
|
372
|
+
# a genuine opt-in tail round-trips on full rerun.
|
|
373
|
+
set +e
|
|
374
|
+
out="$(run_gr mint --out "$LEDGER/tailed.json" --tail-bytes 4096 -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
|
|
375
|
+
set -e
|
|
376
|
+
assert_rc "$rc" 0 "mint tailed"
|
|
377
|
+
set +e
|
|
378
|
+
out="$(run_gr verify "$LEDGER/tailed.json" --rerun -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
|
|
379
|
+
set -e
|
|
380
|
+
assert_rc "$rc" 0 "genuine tail rerun"
|
|
381
|
+
python3 - "$LEDGER" <<'EOF'
|
|
382
|
+
import json, sys
|
|
383
|
+
ledger = sys.argv[1]
|
|
384
|
+
d = json.load(open(f"{ledger}/tailed.json"))
|
|
385
|
+
d["output_tail"] = "forged human-readable excerpt\n"
|
|
386
|
+
open(f"{ledger}/forged-tail.json", "w").write(
|
|
387
|
+
json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
|
|
388
|
+
EOF
|
|
389
|
+
set +e
|
|
390
|
+
out="$(run_gr verify "$LEDGER/forged-tail.json" --rerun -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
|
|
391
|
+
set -e
|
|
392
|
+
assert_rc "$rc" 1 "forged tail"
|
|
393
|
+
assert_contains "output_tail_mismatch" "$out" "forged tail reason"
|
|
394
|
+
|
|
395
|
+
# (30) A failed publish never wedges the slot: after an overwrite refusal the
|
|
396
|
+
# temp file is gone and the original receipt is intact.
|
|
397
|
+
set +e
|
|
398
|
+
out="$(run_gr mint --out "$LEDGER/green.json" -- true 2>&1)"; rc=$?
|
|
399
|
+
set -e
|
|
400
|
+
assert_rc "$rc" 2 "republish refused"
|
|
401
|
+
ls "$LEDGER"/green.json.tmp.* 2>/dev/null && fail "temp file left behind"
|
|
402
|
+
python3 -c 'import json,sys; json.load(open(sys.argv[1]))' "$LEDGER/green.json" || fail "original receipt corrupted"
|
|
403
|
+
|
|
404
|
+
# (31) APPLIED MUTATIONS on the tail binding: an EMPTIED tail and a TRUNCATED
|
|
405
|
+
# genuine suffix must both go red on full rerun — the tail is reconstructed at
|
|
406
|
+
# the recorded tail_bytes and compared exactly, so weakening the excerpt while
|
|
407
|
+
# keeping exit/hash cannot ride a full-verification verdict.
|
|
408
|
+
python3 - "$LEDGER" <<'EOF'
|
|
409
|
+
import json, sys
|
|
410
|
+
ledger = sys.argv[1]
|
|
411
|
+
d = json.load(open(f"{ledger}/tailed.json"))
|
|
412
|
+
d["output_tail"] = ""
|
|
413
|
+
open(f"{ledger}/emptied-tail.json", "w").write(
|
|
414
|
+
json.dumps(d, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
|
|
415
|
+
d2 = json.load(open(f"{ledger}/tailed.json"))
|
|
416
|
+
d2["output_tail"] = d2["output_tail"][-5:]
|
|
417
|
+
open(f"{ledger}/truncated-tail.json", "w").write(
|
|
418
|
+
json.dumps(d2, ensure_ascii=False, sort_keys=True, indent=1) + "\n")
|
|
419
|
+
EOF
|
|
420
|
+
set +e
|
|
421
|
+
out="$(run_gr verify "$LEDGER/emptied-tail.json" --rerun -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
|
|
422
|
+
set -e
|
|
423
|
+
assert_rc "$rc" 1 "emptied tail"
|
|
424
|
+
assert_contains "output_tail_mismatch" "$out" "emptied tail reason"
|
|
425
|
+
set +e
|
|
426
|
+
out="$(run_gr verify "$LEDGER/truncated-tail.json" --rerun -- bash -c 'echo tail-content; exit 0' 2>&1)"; rc=$?
|
|
427
|
+
set -e
|
|
428
|
+
assert_rc "$rc" 1 "truncated tail"
|
|
429
|
+
assert_contains "output_tail_mismatch" "$out" "truncated tail reason"
|
|
430
|
+
|
|
431
|
+
echo "test_gate_receipt: ok"
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Oracle self-proof walk for the pinned-phrase gate families in
|
|
3
|
+
# check-ccl-skills.sh (heavy lane: each leg runs the full shipped gate against
|
|
4
|
+
# a fixture clone of the working tree).
|
|
5
|
+
#
|
|
6
|
+
# Why this exists (074): the pinned-phrase families are straight-line bash
|
|
7
|
+
# loops; an empty phrase list, a quoting regression, or a broken exit path
|
|
8
|
+
# would turn every one of them into an always-green check with no signal.
|
|
9
|
+
# None of the ~40 pinned phrases had ever been proven capable of going red.
|
|
10
|
+
# This walk applies one deletion mutation per gate family and requires the
|
|
11
|
+
# full gate to fail RED with that family's own token (differential
|
|
12
|
+
# attribution: the mutant leg must fail on the mutated family, and the
|
|
13
|
+
# unmutated control leg must show every family token green).
|
|
14
|
+
#
|
|
15
|
+
# MUST-HIT (one applied mutation per family -> expected red token):
|
|
16
|
+
# W1 existing-project-assessment-report.md loses "Assessment launch checklist"
|
|
17
|
+
# -> project_assessment_template_gate_missing
|
|
18
|
+
# W2 skill-extraction-workflow/SKILL.md loses "would other teammates hit this"
|
|
19
|
+
# -> skill_extraction_teammate_trigger_gate_missing
|
|
20
|
+
# W3 testing-strategy/SKILL.md loses "先写测试用例"
|
|
21
|
+
# -> testing_strategy_test_case_first_gate_missing
|
|
22
|
+
# W4 product-rd-workflow/SKILL.md loses "### Pre-Final Continuation Gate"
|
|
23
|
+
# -> product_rd_entrypoint_anchor_gate_missing
|
|
24
|
+
# W5 contract-anchored reference loses its pinned discriminator sentence
|
|
25
|
+
# -> contract_anchor_missing (via the delegated contract-anchor gate)
|
|
26
|
+
#
|
|
27
|
+
# MUST-NOT-HIT (control): the unmutated fixture run exits 0 and prints every
|
|
28
|
+
# family green token (project_assessment_template_gate_ok,
|
|
29
|
+
# task_retro_memory_escape_gate_ok, test_case_first_gate_ok,
|
|
30
|
+
# product_rd_entrypoint_anchor_gate_ok, contract_anchor_gate_ok).
|
|
31
|
+
#
|
|
32
|
+
# Coverage boundary (stated, not silent): one pin per family is mutated, so
|
|
33
|
+
# the walk attests each family's parse/exit path, not every individual phrase;
|
|
34
|
+
# per-family phrase-list non-vacuity is separately asserted by counting the
|
|
35
|
+
# `for required_phrase in` loops and requiring each mutated phrase to exist in
|
|
36
|
+
# the fixture before mutation. ALIAS_AUDIT_CMD is unset for determinism (the
|
|
37
|
+
# gate then takes the public-fallback R0 branch on every host).
|
|
38
|
+
set -u
|
|
39
|
+
|
|
40
|
+
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)
|
|
41
|
+
repo_root=$(cd "$script_dir/../../.." && pwd -P)
|
|
42
|
+
gate_rel="skills/skill-extraction-workflow/scripts/check-ccl-skills.sh"
|
|
43
|
+
fail=0
|
|
44
|
+
|
|
45
|
+
note() { printf '%s\n' "$*"; }
|
|
46
|
+
|
|
47
|
+
tmp=$(mktemp -d)
|
|
48
|
+
trap 'rm -rf "$tmp"' EXIT
|
|
49
|
+
|
|
50
|
+
# Fixture: faithful copy of the working tree (tracked + untracked-unignored),
|
|
51
|
+
# committed once so BASE_REF=HEAD yields an empty diff.
|
|
52
|
+
fixture="$tmp/fixture"
|
|
53
|
+
mkdir -p "$fixture"
|
|
54
|
+
(cd "$repo_root" && git ls-files --cached --others --exclude-standard -z \
|
|
55
|
+
| tar --null -T - -cf - ) | tar -xf - -C "$fixture"
|
|
56
|
+
git -C "$fixture" init -q
|
|
57
|
+
git -C "$fixture" -c user.name=fixture -c user.email=fixture@invalid add -A
|
|
58
|
+
git -C "$fixture" -c user.name=fixture -c user.email=fixture@invalid commit -qm fixture
|
|
59
|
+
# The impact-chain gate resolves merge-bases against origin/main and origin/dev;
|
|
60
|
+
# point both at the fixture's single commit so diff-scoped gates see an empty
|
|
61
|
+
# scope instead of dying on a missing remote ref.
|
|
62
|
+
git -C "$fixture" update-ref refs/remotes/origin/main HEAD
|
|
63
|
+
git -C "$fixture" update-ref refs/remotes/origin/dev HEAD
|
|
64
|
+
|
|
65
|
+
# Vacuity guard: the shipped gate must still carry its pinned-phrase loops.
|
|
66
|
+
loop_count=$(grep -c 'for required_phrase in' "$fixture/$gate_rel")
|
|
67
|
+
if [[ "$loop_count" -lt 7 ]]; then
|
|
68
|
+
note "FAIL vacuity-guard: expected >=7 'for required_phrase in' loops, found $loop_count"
|
|
69
|
+
fail=1
|
|
70
|
+
fi
|
|
71
|
+
|
|
72
|
+
run_gate() { # run_gate -> sets got_rc/got_out (full gate inside the fixture)
|
|
73
|
+
# CCL_SKILL_BASE_REF is deliberately NOT set: it would leak into child
|
|
74
|
+
# validators' synthetic self-test repos (where HEAD always resolves and
|
|
75
|
+
# turns their "no base -> degraded" legs into false passes). The fixture has
|
|
76
|
+
# a single commit, so diff-scoped gates see an empty scope and the run lands
|
|
77
|
+
# interim — the pinned-phrase families under test are tree scans and run
|
|
78
|
+
# fully either way.
|
|
79
|
+
got_out=$(cd "$fixture" && env -u ALIAS_AUDIT_CMD -u CCL_SKILL_BASE_REF \
|
|
80
|
+
bash "$gate_rel" . 2>&1)
|
|
81
|
+
got_rc=$?
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
restore_fixture() {
|
|
85
|
+
git -C "$fixture" checkout -q -- .
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
mutate() { # mutate <repo-relative-file> <exact-phrase-to-delete>
|
|
89
|
+
local file="$fixture/$1" phrase="$2"
|
|
90
|
+
if ! grep -qF -- "$phrase" "$file"; then
|
|
91
|
+
note "FAIL pre-mutation: phrase not present (walk out of sync): $phrase"
|
|
92
|
+
fail=1
|
|
93
|
+
return 1
|
|
94
|
+
fi
|
|
95
|
+
PHRASE="$phrase" perl -pi -e 's/\Q$ENV{PHRASE}\E/mutated-away/g' "$file"
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
walk() { # walk <case> <file> <phrase> <expected-red-token>
|
|
99
|
+
local case_id="$1" file="$2" phrase="$3" token="$4"
|
|
100
|
+
mutate "$file" "$phrase" || return
|
|
101
|
+
run_gate
|
|
102
|
+
if [[ "$got_rc" -eq 0 ]]; then
|
|
103
|
+
note "FAIL $case_id: gate stayed green under applied mutation ($token never fired)"
|
|
104
|
+
fail=1
|
|
105
|
+
elif [[ "$got_out" != *"$token"* ]]; then
|
|
106
|
+
note "FAIL $case_id: gate red but wrong reason (wanted $token)"
|
|
107
|
+
note "$(tail -n 5 <<<"$got_out")"
|
|
108
|
+
fail=1
|
|
109
|
+
else
|
|
110
|
+
note "ok $case_id ($token)"
|
|
111
|
+
fi
|
|
112
|
+
restore_fixture
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
# Control leg first: unmutated fixture must be green with every family token.
|
|
116
|
+
run_gate
|
|
117
|
+
if [[ "$got_rc" -ne 0 ]]; then
|
|
118
|
+
note "FAIL control: unmutated fixture gate rc=$got_rc"
|
|
119
|
+
note "$(tail -n 10 <<<"$got_out")"
|
|
120
|
+
fail=1
|
|
121
|
+
else
|
|
122
|
+
for token in \
|
|
123
|
+
project_assessment_template_gate_ok \
|
|
124
|
+
task_retro_memory_escape_gate_ok \
|
|
125
|
+
test_case_first_gate_ok \
|
|
126
|
+
product_rd_entrypoint_anchor_gate_ok \
|
|
127
|
+
"contract_anchor_gate_ok ("; do
|
|
128
|
+
if [[ "$got_out" != *"$token"* ]]; then
|
|
129
|
+
note "FAIL control: green run missing family token $token"
|
|
130
|
+
fail=1
|
|
131
|
+
fi
|
|
132
|
+
done
|
|
133
|
+
[[ "$fail" -eq 0 ]] && note "ok control (all family tokens green)"
|
|
134
|
+
fi
|
|
135
|
+
|
|
136
|
+
walk W1 "skills/product-rd-workflow/references/existing-project-assessment-report.md" \
|
|
137
|
+
"Assessment launch checklist" "project_assessment_template_gate_missing"
|
|
138
|
+
walk W2 "skills/skill-extraction-workflow/SKILL.md" \
|
|
139
|
+
"would other teammates hit this" "skill_extraction_teammate_trigger_gate_missing"
|
|
140
|
+
walk W3 "skills/testing-strategy/SKILL.md" \
|
|
141
|
+
"先写测试用例" "testing_strategy_test_case_first_gate_missing"
|
|
142
|
+
walk W4 "skills/product-rd-workflow/SKILL.md" \
|
|
143
|
+
"### Pre-Final Continuation Gate" "product_rd_entrypoint_anchor_gate_missing"
|
|
144
|
+
walk W5 "skills/testing-strategy/references/ci-fixtures-and-flake-control.md" \
|
|
145
|
+
"One discriminating predicate decides the verdict" "contract_anchor_missing"
|
|
146
|
+
|
|
147
|
+
if [[ "$fail" -ne 0 ]]; then
|
|
148
|
+
echo "test_pinned_phrase_mutation_walk: FAIL"
|
|
149
|
+
exit 1
|
|
150
|
+
fi
|
|
151
|
+
echo "test_pinned_phrase_mutation_walk: ok"
|