@ccoalm/ccl-skills 0.16.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +8 -7
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-review-covers-head.sh +128 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-untracked-background.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_review_covers_head.sh +104 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_untracked_background.sh +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +13 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +78 -126
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/wording-only-review.md +136 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +521 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +25 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_order.sh +30 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +439 -19
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +14 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +6 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +48 -119
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +9 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check_review_evidence_present.py +122 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +4 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +37 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +102 -20
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +16 -27
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +3 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_review_evidence_present.sh +81 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +137 -279
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +41 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +49 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/closeout-reread.md +40 -0
- package/dist/assets/release.json +72 -52
- package/package.json +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +0 -1242
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +0 -978
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +0 -1477
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +0 -1183
|
@@ -1,15 +1,14 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
|
-
# Regression for the extraction
|
|
2
|
+
# Regression for the extraction review wrapper: every call is one single-shot
|
|
3
|
+
# pass, and no caller option can reopen a tracked review chain.
|
|
3
4
|
set -euo pipefail
|
|
4
5
|
|
|
5
6
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
6
7
|
WRAPPER="$SCRIPT_DIR/extraction_review_gate.sh"
|
|
7
8
|
REAL_CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
|
|
8
|
-
REAL_VALIDATOR="$SCRIPT_DIR/validate_extraction_review_state.py"
|
|
9
9
|
ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd -P)"
|
|
10
10
|
[ -x "$WRAPPER" ] || { echo "FAIL: wrapper missing or not executable: $WRAPPER" >&2; exit 1; }
|
|
11
11
|
[ -x "$REAL_CONTROLLER" ] || { echo "FAIL: real controller missing or not executable: $REAL_CONTROLLER" >&2; exit 1; }
|
|
12
|
-
[ -f "$REAL_VALIDATOR" ] || { echo "FAIL: real validator missing: $REAL_VALIDATOR" >&2; exit 1; }
|
|
13
12
|
|
|
14
13
|
TMP="$(mktemp -d "${TMPDIR:-/tmp}/extraction-review-gate.XXXXXX")"
|
|
15
14
|
trap 'rm -rf "$TMP"' EXIT
|
|
@@ -17,326 +16,185 @@ trap 'rm -rf "$TMP"' EXIT
|
|
|
17
16
|
fail() { echo "FAIL: $*" >&2; exit 1; }
|
|
18
17
|
assert_rc() { [ "$1" = "$2" ] || fail "expected rc=$2 got rc=$1 ($3)"; }
|
|
19
18
|
assert_contains() { case "$2" in *"$1"*) : ;; *) fail "expected '$1' ($3): $2";; esac; }
|
|
19
|
+
assert_not_contains() { case "$2" in *"$1"*) fail "unexpected '$1' ($3): $2";; *) : ;; esac; }
|
|
20
20
|
|
|
21
|
+
# A fake controller that records its argv, so the fixed options are observable.
|
|
21
22
|
mkdir -p "$TMP/skills/skill-extraction-workflow/scripts" "$TMP/skills/code-review/scripts"
|
|
22
|
-
|
|
23
|
+
FAKE_WRAPPER="$TMP/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh"
|
|
24
|
+
cp "$WRAPPER" "$FAKE_WRAPPER"
|
|
23
25
|
printf '%s\n' \
|
|
24
26
|
'#!/usr/bin/env bash' \
|
|
25
27
|
'printf '\''%s\0'\'' "$@" >"$CAPTURE_PATH"' \
|
|
26
28
|
'exit "${FAKE_RC:-0}"' >"$TMP/skills/code-review/scripts/review_gate.sh"
|
|
27
|
-
chmod +x "$
|
|
29
|
+
chmod +x "$FAKE_WRAPPER" "$TMP/skills/code-review/scripts/review_gate.sh"
|
|
28
30
|
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
python3 - "$TMP/args" <<'PY'
|
|
31
|
+
captured() {
|
|
32
|
+
python3 - "$TMP/args" <<'PY'
|
|
32
33
|
import sys
|
|
33
34
|
from pathlib import Path
|
|
34
|
-
|
|
35
|
-
args = [item.decode() for item in Path(sys.argv[1]).read_bytes().split(b"\0") if item]
|
|
36
|
-
assert args[:2] == ["--challenge-budget", "1"], args
|
|
37
|
-
assert args.count("--challenge-budget") == 1, args
|
|
38
|
-
PY
|
|
39
|
-
|
|
40
|
-
# A caller option with a missing operand must not consume the wrapper-owned
|
|
41
|
-
# budget flag. The real controller will reject the missing operand, but only
|
|
42
|
-
# after seeing the fixed extraction budget as its own option/value pair.
|
|
43
|
-
CAPTURE_PATH="$TMP/args" "$TMP/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh" \
|
|
44
|
-
--mode review --cwd /synthetic --implementer-family openai --base
|
|
45
|
-
python3 - "$TMP/args" <<'PY'
|
|
46
|
-
import sys
|
|
47
|
-
from pathlib import Path
|
|
48
|
-
|
|
49
|
-
args = [item.decode() for item in Path(sys.argv[1]).read_bytes().split(b"\0") if item]
|
|
50
|
-
assert args[:2] == ["--challenge-budget", "1"], args
|
|
51
|
-
assert args[-1] == "--base", args
|
|
52
|
-
assert args.count("--challenge-budget") == 1, args
|
|
35
|
+
print(" ".join(i.decode() for i in Path(sys.argv[1]).read_bytes().split(b"\0") if i))
|
|
53
36
|
PY
|
|
54
|
-
|
|
55
|
-
# Exercise the installed wrapper/controller pair without invoking a model. The
|
|
56
|
-
# same real controller defaults to budget 0 when called directly, while the
|
|
57
|
-
# extraction wrapper must make the emitted receipt report budget 1.
|
|
58
|
-
python3 - "$TMP/real-controller.diff" "$TMP/real-controller-plan.json" <<'PY'
|
|
59
|
-
import json
|
|
60
|
-
import sys
|
|
61
|
-
from pathlib import Path
|
|
62
|
-
|
|
63
|
-
diff_path, plan_path = map(Path, sys.argv[1:])
|
|
64
|
-
diff_path.write_text(
|
|
65
|
-
"diff --git a/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh "
|
|
66
|
-
"b/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh\n"
|
|
67
|
-
"--- a/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh\n"
|
|
68
|
-
"+++ b/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh\n"
|
|
69
|
-
"@@ -1 +1 @@\n-old wrapper\n+new wrapper\n",
|
|
70
|
-
encoding="utf-8",
|
|
71
|
-
)
|
|
72
|
-
conclusions = {
|
|
73
|
-
"correctness": "The real controller receipt exposes the effective extraction budget.",
|
|
74
|
-
"safety": "The probe selects only the implementer family and invokes no external reviewer.",
|
|
75
|
-
"failure_paths": "The no-independent-reviewer boundary remains structured and fail closed.",
|
|
76
|
-
"tests_evidence": "Direct and wrapped calls provide a differential budget assertion.",
|
|
77
|
-
"compatibility": "The generic controller default remains zero while extraction fixes one.",
|
|
78
37
|
}
|
|
79
|
-
skills = {
|
|
80
|
-
"correctness": "skill-extraction-workflow",
|
|
81
|
-
"safety": "code-review",
|
|
82
|
-
"failure_paths": "python-service-dev",
|
|
83
|
-
"tests_evidence": "testing-strategy",
|
|
84
|
-
"compatibility": "terminal-cli-dev",
|
|
85
|
-
}
|
|
86
|
-
plan = {
|
|
87
|
-
"intent": "Prove the extraction wrapper and real review controller agree on budget one.",
|
|
88
|
-
"acceptance": ["The wrapped real-controller receipt reports challenge_budget one."],
|
|
89
|
-
"self_review": [
|
|
90
|
-
{
|
|
91
|
-
"concern": concern,
|
|
92
|
-
"skill": skills[concern],
|
|
93
|
-
"conclusion": conclusion,
|
|
94
|
-
"evidence_refs": ["real-controller-differential"],
|
|
95
|
-
}
|
|
96
|
-
for concern, conclusion in conclusions.items()
|
|
97
|
-
],
|
|
98
|
-
"evidence": [
|
|
99
|
-
{
|
|
100
|
-
"id": "real-controller-differential",
|
|
101
|
-
"result": "The same real controller is invoked directly and through the extraction wrapper.",
|
|
102
|
-
}
|
|
103
|
-
],
|
|
104
|
-
}
|
|
105
|
-
plan_path.write_text(json.dumps(plan, indent=2) + "\n", encoding="utf-8")
|
|
106
|
-
PY
|
|
107
38
|
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
--implementer-family openai --review-plan-file "$TMP/real-controller-plan.json"
|
|
111
|
-
--stage build --review-harness --timeout 5 --total-timeout 5
|
|
112
|
-
)
|
|
113
|
-
# Make a Codex executable visibly available, independent of whichever CLI
|
|
114
|
-
# version the host carries. Because the implementer is OpenAI-family, the real
|
|
115
|
-
# controller must reject Codex as same-family before probing or invoking it.
|
|
116
|
-
mkdir -p "$TMP/fake-bin"
|
|
117
|
-
printf '%s\n' \
|
|
118
|
-
'#!/usr/bin/env bash' \
|
|
119
|
-
'printf invoked >"$CODEX_MARKER"' \
|
|
120
|
-
'exit 99' >"$TMP/fake-bin/codex"
|
|
121
|
-
chmod +x "$TMP/fake-bin/codex"
|
|
122
|
-
set +e
|
|
123
|
-
direct_out="$(CODEX_MARKER="$TMP/codex-invoked" PATH="$TMP/fake-bin:$PATH" \
|
|
124
|
-
CODE_REVIEW_CLIENT_ORDER=codex bash "$REAL_CONTROLLER" "${real_args[@]}" 2>&1)"
|
|
125
|
-
direct_rc=$?
|
|
126
|
-
wrapped_out="$(CODEX_MARKER="$TMP/codex-invoked" PATH="$TMP/fake-bin:$PATH" \
|
|
127
|
-
CODE_REVIEW_CLIENT_ORDER=codex "$WRAPPER" "${real_args[@]}" \
|
|
128
|
-
--review-chain-id extraction-wrapper-real-controller \
|
|
129
|
-
--autonomous-review-index 1 2>&1)"
|
|
130
|
-
wrapped_rc=$?
|
|
131
|
-
set -e
|
|
132
|
-
assert_rc "$direct_rc" 2 "direct real controller must stop before model inference"
|
|
133
|
-
assert_rc "$wrapped_rc" 2 "wrapped real controller must stop before model inference"
|
|
134
|
-
assert_contains '"reason_code":"no_independent_reviewer_available"' "$direct_out" "direct real-controller boundary"
|
|
135
|
-
assert_contains '"reason_code":"no_independent_reviewer_available"' "$wrapped_out" "wrapped real-controller boundary"
|
|
136
|
-
assert_contains '"challenge_budget":0' "$direct_out" "generic controller default budget"
|
|
137
|
-
assert_contains '"challenge_budget":1' "$wrapped_out" "wrapper-enforced real-controller budget"
|
|
138
|
-
[ ! -e "$TMP/codex-invoked" ] || fail "same-family Codex executable was invoked"
|
|
39
|
+
CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode review --cwd /synthetic --implementer-family openai
|
|
40
|
+
assert_contains "--challenge-budget 0 --mode review" "$(captured)" "review is single-shot with no challenge capacity"
|
|
139
41
|
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
# the terminal disposition fields so validation must walk the producer's real
|
|
143
|
-
# chain/scope shape before stopping at the intentionally omitted base evidence.
|
|
144
|
-
printf '%s\n' "$wrapped_out" >"$TMP/wrapped-real-controller.out"
|
|
145
|
-
python3 - "$TMP/wrapped-real-controller.out" "$TMP" <<'PY'
|
|
146
|
-
import copy
|
|
147
|
-
import hashlib
|
|
148
|
-
import json
|
|
149
|
-
import sys
|
|
150
|
-
from pathlib import Path
|
|
42
|
+
CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode challenge --cwd /synthetic --implementer-family openai --focus f
|
|
43
|
+
assert_contains "--challenge-budget 1 --challenge-index 1 --mode challenge" "$(captured)" "challenge is the single untracked challenge"
|
|
151
44
|
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
receipt = None
|
|
155
|
-
for line in reversed(output_path.read_text(encoding="utf-8").splitlines()):
|
|
156
|
-
try:
|
|
157
|
-
value = json.loads(line)
|
|
158
|
-
except json.JSONDecodeError:
|
|
159
|
-
continue
|
|
160
|
-
if isinstance(value, dict) and value.get("schema_version") == 3:
|
|
161
|
-
receipt = value
|
|
162
|
-
break
|
|
163
|
-
assert receipt is not None
|
|
45
|
+
CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode=challenge --cwd /synthetic --focus f
|
|
46
|
+
assert_contains "--challenge-budget 1 --challenge-index 1 --mode=challenge" "$(captured)" "--mode=VALUE spelling"
|
|
164
47
|
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
receipt_name = f"{stem}-receipt.json"
|
|
169
|
-
(root / receipt_name).write_bytes(encoded)
|
|
170
|
-
digest = hashlib.sha256(encoded).hexdigest()
|
|
171
|
-
ledger = {
|
|
172
|
-
"schema_version": 3,
|
|
173
|
-
"candidate_sha256": value["candidate_sha256"],
|
|
174
|
-
"controller_receipts": [
|
|
175
|
-
{"sequence": 1, "file": receipt_name, "sha256": digest}
|
|
176
|
-
],
|
|
177
|
-
"completion_receipt": None,
|
|
178
|
-
"base_attestations": [],
|
|
179
|
-
"autonomous_round": 1,
|
|
180
|
-
"controller_review_state": controller_state,
|
|
181
|
-
"finding_classes": [],
|
|
182
|
-
"unreviewed_delta": [],
|
|
183
|
-
"closeout_state": "ready_for_human_decision",
|
|
184
|
-
}
|
|
185
|
-
(root / f"{stem}-ledger.json").write_text(
|
|
186
|
-
json.dumps(ledger, ensure_ascii=False, indent=2) + "\n",
|
|
187
|
-
encoding="utf-8",
|
|
188
|
-
)
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
write_receipt_and_ledger("real-emitted", receipt, receipt.get("review_state", "inconclusive"))
|
|
192
|
-
normalized = copy.deepcopy(receipt)
|
|
193
|
-
scope_before = json.dumps(normalized["review_scope"], sort_keys=True)
|
|
194
|
-
normalized.update(
|
|
195
|
-
status="passed",
|
|
196
|
-
findings=[],
|
|
197
|
-
review_state="reviewed",
|
|
198
|
-
human_decision_required=False,
|
|
199
|
-
)
|
|
200
|
-
assert json.dumps(normalized["review_scope"], sort_keys=True) == scope_before
|
|
201
|
-
write_receipt_and_ledger("real-scope", normalized, "reviewed")
|
|
202
|
-
PY
|
|
203
|
-
|
|
204
|
-
set +e
|
|
205
|
-
emitted_state_out="$(python3 "$REAL_VALIDATOR" "$TMP/real-emitted-ledger.json" 2>&1)"
|
|
206
|
-
emitted_state_rc=$?
|
|
207
|
-
scope_state_out="$(python3 "$REAL_VALIDATOR" "$TMP/real-scope-ledger.json" 2>&1)"
|
|
208
|
-
scope_state_rc=$?
|
|
209
|
-
set -e
|
|
210
|
-
assert_rc "$emitted_state_rc" 1 "exact real receipt reaches semantic status validation"
|
|
211
|
-
assert_contains "must have status passed or findings" "$emitted_state_out" "exact real receipt semantic boundary"
|
|
212
|
-
assert_rc "$scope_state_rc" 1 "producer-derived receipt reaches post-scope validation"
|
|
213
|
-
assert_contains "base_attestations must be a non-empty array" "$scope_state_out" "real producer scope shape"
|
|
214
|
-
|
|
215
|
-
# Shorter spellings that do not start with --challenge-b are ambiguous among
|
|
216
|
-
# the controller's budget/index/classes options, so they fail closed instead of
|
|
217
|
-
# becoming a hidden budget override.
|
|
218
|
-
for abbreviated in --challeng --challenge=4; do
|
|
48
|
+
# Missing or unsupported modes fail closed before the controller runs.
|
|
49
|
+
for bad in "" "--mode complete" "--mode=complete" "--mo challenge"; do
|
|
50
|
+
: >"$TMP/args"
|
|
219
51
|
set +e
|
|
220
|
-
|
|
52
|
+
# shellcheck disable=SC2086
|
|
53
|
+
out="$(CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" $bad --cwd /synthetic 2>&1)"
|
|
221
54
|
rc=$?
|
|
222
55
|
set -e
|
|
223
|
-
assert_rc "$rc" 2 "
|
|
224
|
-
assert_contains "
|
|
56
|
+
assert_rc "$rc" 2 "mode '$bad' must be refused"
|
|
57
|
+
assert_contains "extraction_review_gate_error" "$out" "mode '$bad' reason"
|
|
58
|
+
[ ! -s "$TMP/args" ] || fail "controller ran for mode '$bad'"
|
|
225
59
|
done
|
|
226
60
|
|
|
227
|
-
|
|
61
|
+
# Every chain option, full or abbreviated, with or without =VALUE, is refused.
|
|
62
|
+
for spelling in \
|
|
63
|
+
--review-chain-id --review-chain-id=x --review-c \
|
|
64
|
+
--autonomous-review-index --autonomous-review-index=2 --au \
|
|
65
|
+
--prior-review-result-file --prior-review-result-file=/r.json --prio \
|
|
66
|
+
--predecessor-chain-result-file --pre \
|
|
67
|
+
--completion-review-result-file --com \
|
|
68
|
+
--challenge-budget --challenge-budget=4 --challenge-b=4 \
|
|
69
|
+
--challenge-index --challenge-index=2 --challenge-i; do
|
|
228
70
|
: >"$TMP/args"
|
|
229
71
|
set +e
|
|
230
|
-
out="$(CAPTURE_PATH="$TMP/args" "$
|
|
72
|
+
out="$(CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode challenge --focus f "$spelling" 1 2>&1)"
|
|
231
73
|
rc=$?
|
|
232
74
|
set -e
|
|
233
|
-
assert_rc "$rc" 2 "
|
|
234
|
-
assert_contains "
|
|
235
|
-
[ ! -s "$TMP/args" ] || fail "controller ran after
|
|
75
|
+
assert_rc "$rc" 2 "chain option $spelling must be refused"
|
|
76
|
+
assert_contains "extraction_review_gate_error" "$out" "$spelling reason"
|
|
77
|
+
[ ! -s "$TMP/args" ] || fail "controller ran after $spelling"
|
|
236
78
|
done
|
|
237
79
|
|
|
238
80
|
set +e
|
|
239
|
-
CAPTURE_PATH="$TMP/args" FAKE_RC=7 "$
|
|
240
|
-
--mode review --cwd /synthetic --base base --implementer-family openai >/dev/null 2>&1
|
|
81
|
+
CAPTURE_PATH="$TMP/args" FAKE_RC=7 "$FAKE_WRAPPER" --mode review --cwd /synthetic >/dev/null 2>&1
|
|
241
82
|
rc=$?
|
|
242
83
|
set -e
|
|
243
84
|
assert_rc "$rc" 7 "wrapper must preserve controller exit status"
|
|
244
85
|
|
|
245
86
|
mv "$TMP/skills/code-review/scripts/review_gate.sh" "$TMP/controller-away"
|
|
246
87
|
set +e
|
|
247
|
-
out="$(CAPTURE_PATH="$TMP/args" "$
|
|
248
|
-
--mode review --cwd /synthetic --base base --implementer-family openai 2>&1)"
|
|
88
|
+
out="$(CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode review --cwd /synthetic 2>&1)"
|
|
249
89
|
rc=$?
|
|
250
90
|
set -e
|
|
251
91
|
assert_rc "$rc" 2 "missing controller must fail closed"
|
|
252
92
|
assert_contains "controller is unavailable" "$out" "missing-controller reason"
|
|
253
93
|
|
|
254
|
-
# The
|
|
255
|
-
#
|
|
256
|
-
#
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
import
|
|
94
|
+
# The real controller must accept both single-shot shapes without a chain id.
|
|
95
|
+
# The implementer is OpenAI-family and the only client offered is Codex, so the
|
|
96
|
+
# controller must refuse it as same-family and stop before any model inference.
|
|
97
|
+
python3 - "$TMP/real.diff" "$TMP/real-plan.json" "$ROOT" <<'PY'
|
|
98
|
+
import json
|
|
99
|
+
import subprocess
|
|
260
100
|
import sys
|
|
261
101
|
from pathlib import Path
|
|
262
102
|
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
103
|
+
diff_path, plan_path = map(Path, sys.argv[1:3])
|
|
104
|
+
root = Path(sys.argv[3])
|
|
105
|
+
diff_path.write_text(
|
|
106
|
+
"diff --git a/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh "
|
|
107
|
+
"b/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh\n"
|
|
108
|
+
"--- a/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh\n"
|
|
109
|
+
"+++ b/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh\n"
|
|
110
|
+
"@@ -1 +1 @@\n-old wrapper\n+new wrapper\n",
|
|
111
|
+
encoding="utf-8",
|
|
112
|
+
)
|
|
113
|
+
# The required concern set has one owner, the controller; derive it rather than
|
|
114
|
+
# keeping a copy that drifts when the set changes.
|
|
115
|
+
required = subprocess.run(
|
|
116
|
+
[sys.executable, str(root / "skills/code-review/scripts/review_gate.py"),
|
|
117
|
+
"--print-required-concerns", "--stage", "build"],
|
|
118
|
+
capture_output=True, text=True, check=True,
|
|
119
|
+
).stdout.split()
|
|
120
|
+
assert required, "the controller printed no required concerns"
|
|
121
|
+
owners = ["skill-extraction-workflow", "code-review", "python-service-dev",
|
|
122
|
+
"testing-strategy", "terminal-cli-dev"]
|
|
123
|
+
rows = [
|
|
124
|
+
{"concern": c, "skill": owners[i % len(owners)],
|
|
125
|
+
"conclusion": f"The single-shot wrapper probe covers {c} within its fixture.",
|
|
126
|
+
"evidence_refs": ["wrapper-probe"]}
|
|
127
|
+
for i, c in enumerate(required)
|
|
128
|
+
]
|
|
129
|
+
covered = {r["skill"] for r in rows}
|
|
130
|
+
rows += [
|
|
131
|
+
{"concern": required[0], "skill": o,
|
|
132
|
+
"conclusion": f"{o} is covered for {required[0]} by the same probe.",
|
|
133
|
+
"evidence_refs": ["wrapper-probe"]}
|
|
134
|
+
for o in owners if o not in covered
|
|
135
|
+
]
|
|
136
|
+
plan_path.write_text(json.dumps({
|
|
137
|
+
"intent": "Prove the extraction wrapper drives the real controller single-shot.",
|
|
138
|
+
"acceptance": ["Both passes reach the reviewer-selection boundary untracked."],
|
|
139
|
+
"self_review": rows,
|
|
140
|
+
"evidence": [{"id": "wrapper-probe", "result": "The real controller is invoked through the wrapper."}],
|
|
141
|
+
}, indent=2) + "\n", encoding="utf-8")
|
|
142
|
+
PY
|
|
278
143
|
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
144
|
+
mkdir -p "$TMP/fake-bin"
|
|
145
|
+
printf '%s\n' '#!/usr/bin/env bash' 'printf invoked >"$CODEX_MARKER"' 'exit 99' >"$TMP/fake-bin/codex"
|
|
146
|
+
chmod +x "$TMP/fake-bin/codex"
|
|
147
|
+
real_args=(
|
|
148
|
+
--cwd "$ROOT" --diff-file "$TMP/real.diff" --implementer-family openai
|
|
149
|
+
--review-plan-file "$TMP/real-plan.json" --stage build --review-harness
|
|
150
|
+
--timeout 5 --total-timeout 5
|
|
151
|
+
)
|
|
152
|
+
for pass in review challenge; do
|
|
153
|
+
extra=()
|
|
154
|
+
[ "$pass" = challenge ] && extra=(--focus "single-shot probe")
|
|
155
|
+
set +e
|
|
156
|
+
out="$(CODEX_MARKER="$TMP/codex-invoked" PATH="$TMP/fake-bin:$PATH" CODE_REVIEW_CLIENT_ORDER=codex \
|
|
157
|
+
"$WRAPPER" --mode "$pass" "${real_args[@]}" "${extra[@]}" 2>&1)"
|
|
158
|
+
rc=$?
|
|
159
|
+
set -e
|
|
160
|
+
assert_rc "$rc" 2 "real $pass must stop before model inference"
|
|
161
|
+
assert_contains '"reason_code":"no_independent_reviewer_available"' "$out" "real $pass reaches reviewer selection"
|
|
162
|
+
assert_not_contains 'review_chain_required' "$out" "real $pass needs no chain"
|
|
163
|
+
assert_not_contains 'review_chain_invalid' "$out" "real $pass needs no chain"
|
|
164
|
+
done
|
|
165
|
+
[ ! -e "$TMP/codex-invoked" ] || fail "same-family Codex executable was invoked"
|
|
288
166
|
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
167
|
+
# The owner documents must route non-wording work through this wrapper and must
|
|
168
|
+
# no longer send anyone to the retired chain ledger or merge-side binder.
|
|
169
|
+
python3 - "$ROOT" <<'PY'
|
|
170
|
+
import re
|
|
171
|
+
import sys
|
|
172
|
+
from pathlib import Path
|
|
293
173
|
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
for
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
assert "
|
|
315
|
-
|
|
316
|
-
assert "
|
|
317
|
-
assert "
|
|
318
|
-
assert "codex review --base" not in quickstart
|
|
319
|
-
assert "codex exec adversarial" not in quickstart
|
|
320
|
-
assert "--challenge-budget" not in quickstart, (
|
|
321
|
-
"quickstart must not let callers override the extraction review budget"
|
|
322
|
-
)
|
|
323
|
-
assert re.search(
|
|
324
|
-
r"[Rr]ound 2 challenge.{0,500}ready_for_human_decision",
|
|
325
|
-
quickstart,
|
|
326
|
-
re.DOTALL,
|
|
327
|
-
), "quickstart does not validate an early-clean round-2 terminal checkpoint"
|
|
328
|
-
assert re.search(
|
|
329
|
-
r"[Rr]ound 2 findings.{0,200}continuation_authorization_required",
|
|
330
|
-
quickstart,
|
|
331
|
-
re.DOTALL,
|
|
332
|
-
), "quickstart does not validate the exhausted-budget terminal checkpoint"
|
|
333
|
-
assert "baseline_race" in quickstart
|
|
334
|
-
assert "at most two challenges" not in quickstart, (
|
|
335
|
-
"quickstart still advertises the retired two-challenge budget"
|
|
336
|
-
)
|
|
337
|
-
assert "At the third Agent-autonomous round" not in dual, (
|
|
338
|
-
"dual-track still gates the lane at the retired third round"
|
|
339
|
-
)
|
|
174
|
+
root = Path(sys.argv[1])
|
|
175
|
+
ref = root / "skills/skill-extraction-workflow"
|
|
176
|
+
docs = {
|
|
177
|
+
"SKILL": (ref / "SKILL.md").read_text(encoding="utf-8"),
|
|
178
|
+
"quickstart": (ref / "references/extraction-quickstart.md").read_text(encoding="utf-8"),
|
|
179
|
+
"dual-track": (ref / "references/dual-track-review-gate.md").read_text(encoding="utf-8"),
|
|
180
|
+
"validation-and-landing": (ref / "references/validation-and-landing.md").read_text(encoding="utf-8"),
|
|
181
|
+
}
|
|
182
|
+
for label, text in docs.items():
|
|
183
|
+
assert "scripts/extraction_review_gate.sh" in text, f"{label} does not route through the wrapper"
|
|
184
|
+
for retired in ("validate_extraction_review_state.py", "review_ledger_binding.py",
|
|
185
|
+
"challenge_budget=1", "succession challenge"):
|
|
186
|
+
assert retired not in text, f"{label} still points at the retired {retired}"
|
|
187
|
+
assert re.search(r"[Nn]on-wording.{0,240}scripts/extraction_review_gate\.sh",
|
|
188
|
+
docs["SKILL"], re.DOTALL), "SKILL does not scope the wrapper to non-wording work"
|
|
189
|
+
assert re.search(r"[Ww]ording-only.{0,500}(?:single|one)[- ](?:round|review|pass)",
|
|
190
|
+
docs["quickstart"], re.DOTALL), "quickstart lost the wording-only single-review exception"
|
|
191
|
+
assert "--challenge-budget" not in docs["quickstart"], "quickstart must not hand callers the budget flag"
|
|
192
|
+
dual = docs["dual-track"]
|
|
193
|
+
for pinned in ("post-review delta", "Every post-review delta gets a delta pass", "After five delta passes", "never left to a human reader"):
|
|
194
|
+
assert pinned in dual, f"dual-track lost '{pinned}'"
|
|
195
|
+
wording_only = (root / "skills/code-review/references/wording-only-review.md").read_text(encoding="utf-8")
|
|
196
|
+
assert "--wording-only-proof-file" in wording_only
|
|
197
|
+
assert "--challenge-budget 0" in wording_only
|
|
340
198
|
PY
|
|
341
199
|
|
|
342
200
|
echo "test_extraction_review_gate: ok"
|
|
@@ -720,5 +720,45 @@ run_gate "$FIX"
|
|
|
720
720
|
assert_rc "$rc" 0 "a fenced example row is an illustration, not a live declaration"
|
|
721
721
|
pass "fenced declaration-looking example does not red a register with no live declarations"
|
|
722
722
|
|
|
723
|
-
|
|
723
|
+
# ── N. RED: the anchored list rule DELETED outright ──────────────────────────
|
|
724
|
+
# A prose rule pinned for a documentation round is normally removed, not
|
|
725
|
+
# reworded: the round that added this case pinned five rules whose only
|
|
726
|
+
# mechanical protection is this gate, and its walk deleted each in turn.
|
|
727
|
+
new_fixture deleted_rule "$LITERAL"
|
|
728
|
+
# A mutation that does not apply proves nothing, so each edit below is checked
|
|
729
|
+
# both before and after: the target must be present first, and the edit must
|
|
730
|
+
# have changed the file. Without this a fixture drift turns these cases into
|
|
731
|
+
# assertions about an unmutated fixture that still pass.
|
|
732
|
+
grep -qF 'Never bypass the demo isolation boundary' "$FIX/skills/demo-skill/SKILL.md" \
|
|
733
|
+
|| fail "fixture drift: the rule this case deletes is not in the fixture"
|
|
734
|
+
grep -v 'Never bypass the demo isolation boundary' "$FIX/skills/demo-skill/SKILL.md" \
|
|
735
|
+
> "$FIX/skills/demo-skill/SKILL.md.tmp"
|
|
736
|
+
mv "$FIX/skills/demo-skill/SKILL.md.tmp" "$FIX/skills/demo-skill/SKILL.md"
|
|
737
|
+
grep -qF 'Never bypass the demo isolation boundary' "$FIX/skills/demo-skill/SKILL.md" \
|
|
738
|
+
&& fail "mutation did not apply: the rule is still present"
|
|
739
|
+
run_gate "$FIX"
|
|
740
|
+
assert_rc "$rc" 1 "deleting the anchored rule outright must be caught"
|
|
741
|
+
assert_contains "anchor text absent from target" "$out" "deleted anchored rule"
|
|
742
|
+
assert_contains "Never bypass the demo isolation boundary" "$out" "names the dead locator"
|
|
743
|
+
pass "deleting an anchored rule turns the gate RED"
|
|
744
|
+
|
|
745
|
+
# ── N+1. GREEN by design: text OUTSIDE the anchor may be gutted ──────────────
|
|
746
|
+
# The stated boundary, asserted rather than promised: a substring locator binds
|
|
747
|
+
# the letters it names and nothing else. Here the anchored clause survives while
|
|
748
|
+
# the rest of its line is replaced, and the gate is silent — which is why the
|
|
749
|
+
# ledger rows that rely on it must not claim semantic protection.
|
|
750
|
+
new_fixture unanchored_clause_gutted "$LITERAL"
|
|
751
|
+
grep -qF ' when dispatching work.' "$FIX/skills/demo-skill/SKILL.md" \
|
|
752
|
+
|| fail "fixture drift: the clause this case guts is not in the fixture"
|
|
753
|
+
sed -i.bak 's/ when dispatching work\./ — every safeguard around it removed./' \
|
|
754
|
+
"$FIX/skills/demo-skill/SKILL.md"
|
|
755
|
+
grep -qF ' — every safeguard around it removed.' "$FIX/skills/demo-skill/SKILL.md" \
|
|
756
|
+
|| fail "mutation did not apply: the unanchored clause is unchanged"
|
|
757
|
+
run_gate "$FIX"
|
|
758
|
+
assert_rc "$rc" 0 "gutting text outside the anchor is invisible to this gate"
|
|
759
|
+
assert_contains "register_firing_path_resolution_ok" "$out" "boundary control"
|
|
760
|
+
assert_not_contains "anchor text absent" "$out" "no false RED on unanchored text"
|
|
761
|
+
pass "text outside the anchor can be gutted while the gate stays green (declared boundary)"
|
|
762
|
+
|
|
763
|
+
[ "$passed" -eq 59 ] || fail "expected 59 assertions, saw $passed (a case was skipped or misplaced)"
|
|
724
764
|
echo "register_firing_path_resolution_tests_ok ($passed assertions)"
|
|
@@ -29,8 +29,8 @@ pass() { passed=$((passed + 1)); echo "PASS: $*"; }
|
|
|
29
29
|
# guard silently stale again (the observed drift shape: a 16 guarding 13
|
|
30
30
|
# executed cases, accumulated while this suite could not run).
|
|
31
31
|
static_pass_calls="$(grep -c '^pass "' "$0")"
|
|
32
|
-
[ "$static_pass_calls" = "
|
|
33
|
-
|| fail "pass-call inventory drifted: counted $static_pass_calls, guard expects
|
|
32
|
+
[ "$static_pass_calls" = "21" ] \
|
|
33
|
+
|| fail "pass-call inventory drifted: counted $static_pass_calls, guard expects 21"
|
|
34
34
|
pass "executed-count guard matches the script's own pass-call inventory"
|
|
35
35
|
|
|
36
36
|
REPO="$TMP/repo"
|
|
@@ -478,6 +478,52 @@ case "$out" in
|
|
|
478
478
|
esac
|
|
479
479
|
pass "a second row cannot inherit a one-row historical locator waiver"
|
|
480
480
|
|
|
481
|
+
# A retired locator that several historical rows cited is waived as ONE entry
|
|
482
|
+
# binding every one of those rows by digest. Each row stays individually pinned:
|
|
483
|
+
# removing one, rewriting one, or adding one more citation must each red.
|
|
484
|
+
MULTI_LOCATOR="command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh"
|
|
485
|
+
multi_row_case() {
|
|
486
|
+
multi_case="$1"
|
|
487
|
+
git -C "$REPO" checkout -- . >/dev/null 2>&1 \
|
|
488
|
+
|| fail "could not restore the clone before the multi-row waiver $multi_case case"
|
|
489
|
+
python3 - "$REGISTER" "$MULTI_LOCATOR" "$multi_case" <<'PY'
|
|
490
|
+
from pathlib import Path
|
|
491
|
+
import sys
|
|
492
|
+
|
|
493
|
+
path = Path(sys.argv[1])
|
|
494
|
+
locator, case = sys.argv[2], sys.argv[3]
|
|
495
|
+
lines = path.read_text(encoding="utf-8").splitlines(keepends=True)
|
|
496
|
+
matches = [i for i, line in enumerate(lines) if line.startswith("|") and locator in line]
|
|
497
|
+
assert len(matches) > 1, f"expected several real waived rows, found {len(matches)}"
|
|
498
|
+
index = matches[len(matches) // 2]
|
|
499
|
+
if case == "delete":
|
|
500
|
+
del lines[index]
|
|
501
|
+
elif case == "rewrite":
|
|
502
|
+
lines[index] = lines[index].replace("| ", "| Fixture-rewritten claim: ", 1)
|
|
503
|
+
else:
|
|
504
|
+
lines.insert(index + 1, lines[index])
|
|
505
|
+
path.write_text("".join(lines), encoding="utf-8")
|
|
506
|
+
PY
|
|
507
|
+
run_check
|
|
508
|
+
[ "$rc" = "1" ] || { dump; fail "multi-row waiver $multi_case must be rc=1, got rc=$rc"; }
|
|
509
|
+
case "$multi_case:$out" in
|
|
510
|
+
delete:*"EXEMPT entry has no citing row in the ledger for 1 of"*) : ;;
|
|
511
|
+
rewrite:*"EXEMPT citing row does not match the waived row"*) : ;;
|
|
512
|
+
duplicate:*"EXEMPT locator cited by"*"rows (allowance"*) : ;;
|
|
513
|
+
*) dump; fail "multi-row waiver $multi_case failed, but not via its exact diagnostic" ;;
|
|
514
|
+
esac
|
|
515
|
+
case "$out" in
|
|
516
|
+
*ccl_skill_check_clean_ok*) dump; fail "a multi-row waiver $multi_case must never yield a clean-landing token" ;;
|
|
517
|
+
*) : ;;
|
|
518
|
+
esac
|
|
519
|
+
}
|
|
520
|
+
multi_row_case delete
|
|
521
|
+
pass "multi-row waiver: deleting one citing row fails closed"
|
|
522
|
+
multi_row_case rewrite
|
|
523
|
+
pass "multi-row waiver: rewriting one citing row fails closed"
|
|
524
|
+
multi_row_case duplicate
|
|
525
|
+
pass "multi-row waiver: a further citing row fails closed"
|
|
526
|
+
|
|
481
527
|
# RED: adding an EXEMPT key without a row digest silently downgrades identity
|
|
482
528
|
# binding unless the production gate rejects the incomplete waiver definition.
|
|
483
529
|
# The checker resolves the gate beside itself, so mutate STUB_GATE — never the
|
|
@@ -547,5 +593,5 @@ case "$out" in
|
|
|
547
593
|
esac
|
|
548
594
|
pass "a failure inside the cleanup-trap window keeps its exit status"
|
|
549
595
|
|
|
550
|
-
[ "$passed" -eq
|
|
596
|
+
[ "$passed" -eq 21 ] || fail "expected 21 assertions, saw $passed"
|
|
551
597
|
echo "register_firing_path_wiring_tests_ok ($passed assertions)"
|