@ccoalm/ccl-skills 0.17.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +8 -7
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-review-covers-head.sh +128 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-untracked-background.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_review_covers_head.sh +104 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_untracked_background.sh +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +13 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +401 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +8 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +230 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +6 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +48 -119
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +9 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check_review_evidence_present.py +122 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +0 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +37 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +102 -20
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +16 -27
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +3 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_review_evidence_present.sh +81 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +130 -310
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +49 -3
- package/dist/assets/release.json +55 -45
- package/package.json +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +0 -1242
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +0 -978
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +0 -1477
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +0 -1183
|
@@ -1,15 +1,14 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
|
-
# Regression for the extraction
|
|
2
|
+
# Regression for the extraction review wrapper: every call is one single-shot
|
|
3
|
+
# pass, and no caller option can reopen a tracked review chain.
|
|
3
4
|
set -euo pipefail
|
|
4
5
|
|
|
5
6
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
6
7
|
WRAPPER="$SCRIPT_DIR/extraction_review_gate.sh"
|
|
7
8
|
REAL_CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
|
|
8
|
-
REAL_VALIDATOR="$SCRIPT_DIR/validate_extraction_review_state.py"
|
|
9
9
|
ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd -P)"
|
|
10
10
|
[ -x "$WRAPPER" ] || { echo "FAIL: wrapper missing or not executable: $WRAPPER" >&2; exit 1; }
|
|
11
11
|
[ -x "$REAL_CONTROLLER" ] || { echo "FAIL: real controller missing or not executable: $REAL_CONTROLLER" >&2; exit 1; }
|
|
12
|
-
[ -f "$REAL_VALIDATOR" ] || { echo "FAIL: real validator missing: $REAL_VALIDATOR" >&2; exit 1; }
|
|
13
12
|
|
|
14
13
|
TMP="$(mktemp -d "${TMPDIR:-/tmp}/extraction-review-gate.XXXXXX")"
|
|
15
14
|
trap 'rm -rf "$TMP"' EXIT
|
|
@@ -17,45 +16,85 @@ trap 'rm -rf "$TMP"' EXIT
|
|
|
17
16
|
fail() { echo "FAIL: $*" >&2; exit 1; }
|
|
18
17
|
assert_rc() { [ "$1" = "$2" ] || fail "expected rc=$2 got rc=$1 ($3)"; }
|
|
19
18
|
assert_contains() { case "$2" in *"$1"*) : ;; *) fail "expected '$1' ($3): $2";; esac; }
|
|
19
|
+
assert_not_contains() { case "$2" in *"$1"*) fail "unexpected '$1' ($3): $2";; *) : ;; esac; }
|
|
20
20
|
|
|
21
|
+
# A fake controller that records its argv, so the fixed options are observable.
|
|
21
22
|
mkdir -p "$TMP/skills/skill-extraction-workflow/scripts" "$TMP/skills/code-review/scripts"
|
|
22
|
-
|
|
23
|
+
FAKE_WRAPPER="$TMP/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh"
|
|
24
|
+
cp "$WRAPPER" "$FAKE_WRAPPER"
|
|
23
25
|
printf '%s\n' \
|
|
24
26
|
'#!/usr/bin/env bash' \
|
|
25
27
|
'printf '\''%s\0'\'' "$@" >"$CAPTURE_PATH"' \
|
|
26
28
|
'exit "${FAKE_RC:-0}"' >"$TMP/skills/code-review/scripts/review_gate.sh"
|
|
27
|
-
chmod +x "$
|
|
29
|
+
chmod +x "$FAKE_WRAPPER" "$TMP/skills/code-review/scripts/review_gate.sh"
|
|
28
30
|
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
python3 - "$TMP/args" <<'PY'
|
|
31
|
+
captured() {
|
|
32
|
+
python3 - "$TMP/args" <<'PY'
|
|
32
33
|
import sys
|
|
33
34
|
from pathlib import Path
|
|
34
|
-
|
|
35
|
-
args = [item.decode() for item in Path(sys.argv[1]).read_bytes().split(b"\0") if item]
|
|
36
|
-
assert args[:2] == ["--challenge-budget", "1"], args
|
|
37
|
-
assert args.count("--challenge-budget") == 1, args
|
|
35
|
+
print(" ".join(i.decode() for i in Path(sys.argv[1]).read_bytes().split(b"\0") if i))
|
|
38
36
|
PY
|
|
37
|
+
}
|
|
39
38
|
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
# after seeing the fixed extraction budget as its own option/value pair.
|
|
43
|
-
CAPTURE_PATH="$TMP/args" "$TMP/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh" \
|
|
44
|
-
--mode review --cwd /synthetic --implementer-family openai --base
|
|
45
|
-
python3 - "$TMP/args" <<'PY'
|
|
46
|
-
import sys
|
|
47
|
-
from pathlib import Path
|
|
39
|
+
CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode review --cwd /synthetic --implementer-family openai
|
|
40
|
+
assert_contains "--challenge-budget 0 --mode review" "$(captured)" "review is single-shot with no challenge capacity"
|
|
48
41
|
|
|
49
|
-
args
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
42
|
+
CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode challenge --cwd /synthetic --implementer-family openai --focus f
|
|
43
|
+
assert_contains "--challenge-budget 1 --challenge-index 1 --mode challenge" "$(captured)" "challenge is the single untracked challenge"
|
|
44
|
+
|
|
45
|
+
CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode=challenge --cwd /synthetic --focus f
|
|
46
|
+
assert_contains "--challenge-budget 1 --challenge-index 1 --mode=challenge" "$(captured)" "--mode=VALUE spelling"
|
|
47
|
+
|
|
48
|
+
# Missing or unsupported modes fail closed before the controller runs.
|
|
49
|
+
for bad in "" "--mode complete" "--mode=complete" "--mo challenge"; do
|
|
50
|
+
: >"$TMP/args"
|
|
51
|
+
set +e
|
|
52
|
+
# shellcheck disable=SC2086
|
|
53
|
+
out="$(CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" $bad --cwd /synthetic 2>&1)"
|
|
54
|
+
rc=$?
|
|
55
|
+
set -e
|
|
56
|
+
assert_rc "$rc" 2 "mode '$bad' must be refused"
|
|
57
|
+
assert_contains "extraction_review_gate_error" "$out" "mode '$bad' reason"
|
|
58
|
+
[ ! -s "$TMP/args" ] || fail "controller ran for mode '$bad'"
|
|
59
|
+
done
|
|
60
|
+
|
|
61
|
+
# Every chain option, full or abbreviated, with or without =VALUE, is refused.
|
|
62
|
+
for spelling in \
|
|
63
|
+
--review-chain-id --review-chain-id=x --review-c \
|
|
64
|
+
--autonomous-review-index --autonomous-review-index=2 --au \
|
|
65
|
+
--prior-review-result-file --prior-review-result-file=/r.json --prio \
|
|
66
|
+
--predecessor-chain-result-file --pre \
|
|
67
|
+
--completion-review-result-file --com \
|
|
68
|
+
--challenge-budget --challenge-budget=4 --challenge-b=4 \
|
|
69
|
+
--challenge-index --challenge-index=2 --challenge-i; do
|
|
70
|
+
: >"$TMP/args"
|
|
71
|
+
set +e
|
|
72
|
+
out="$(CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode challenge --focus f "$spelling" 1 2>&1)"
|
|
73
|
+
rc=$?
|
|
74
|
+
set -e
|
|
75
|
+
assert_rc "$rc" 2 "chain option $spelling must be refused"
|
|
76
|
+
assert_contains "extraction_review_gate_error" "$out" "$spelling reason"
|
|
77
|
+
[ ! -s "$TMP/args" ] || fail "controller ran after $spelling"
|
|
78
|
+
done
|
|
79
|
+
|
|
80
|
+
set +e
|
|
81
|
+
CAPTURE_PATH="$TMP/args" FAKE_RC=7 "$FAKE_WRAPPER" --mode review --cwd /synthetic >/dev/null 2>&1
|
|
82
|
+
rc=$?
|
|
83
|
+
set -e
|
|
84
|
+
assert_rc "$rc" 7 "wrapper must preserve controller exit status"
|
|
85
|
+
|
|
86
|
+
mv "$TMP/skills/code-review/scripts/review_gate.sh" "$TMP/controller-away"
|
|
87
|
+
set +e
|
|
88
|
+
out="$(CAPTURE_PATH="$TMP/args" "$FAKE_WRAPPER" --mode review --cwd /synthetic 2>&1)"
|
|
89
|
+
rc=$?
|
|
90
|
+
set -e
|
|
91
|
+
assert_rc "$rc" 2 "missing controller must fail closed"
|
|
92
|
+
assert_contains "controller is unavailable" "$out" "missing-controller reason"
|
|
54
93
|
|
|
55
|
-
#
|
|
56
|
-
#
|
|
57
|
-
#
|
|
58
|
-
python3 - "$TMP/real
|
|
94
|
+
# The real controller must accept both single-shot shapes without a chain id.
|
|
95
|
+
# The implementer is OpenAI-family and the only client offered is Codex, so the
|
|
96
|
+
# controller must refuse it as same-family and stop before any model inference.
|
|
97
|
+
python3 - "$TMP/real.diff" "$TMP/real-plan.json" "$ROOT" <<'PY'
|
|
59
98
|
import json
|
|
60
99
|
import subprocess
|
|
61
100
|
import sys
|
|
@@ -71,310 +110,91 @@ diff_path.write_text(
|
|
|
71
110
|
"@@ -1 +1 @@\n-old wrapper\n+new wrapper\n",
|
|
72
111
|
encoding="utf-8",
|
|
73
112
|
)
|
|
74
|
-
# The required set has
|
|
75
|
-
#
|
|
76
|
-
# first failing target so the later shards never report it -- five suites drifted
|
|
77
|
-
# that way in one round. Derive it instead.
|
|
113
|
+
# The required concern set has one owner, the controller; derive it rather than
|
|
114
|
+
# keeping a copy that drifts when the set changes.
|
|
78
115
|
required = subprocess.run(
|
|
79
|
-
[
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
"--print-required-concerns",
|
|
83
|
-
"--stage",
|
|
84
|
-
"build",
|
|
85
|
-
],
|
|
86
|
-
capture_output=True,
|
|
87
|
-
text=True,
|
|
88
|
-
check=True,
|
|
116
|
+
[sys.executable, str(root / "skills/code-review/scripts/review_gate.py"),
|
|
117
|
+
"--print-required-concerns", "--stage", "build"],
|
|
118
|
+
capture_output=True, text=True, check=True,
|
|
89
119
|
).stdout.split()
|
|
90
120
|
assert required, "the controller printed no required concerns"
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
owners = [
|
|
99
|
-
"skill-extraction-workflow",
|
|
100
|
-
"code-review",
|
|
101
|
-
"python-service-dev",
|
|
102
|
-
"testing-strategy",
|
|
103
|
-
"terminal-cli-dev",
|
|
121
|
+
owners = ["skill-extraction-workflow", "code-review", "python-service-dev",
|
|
122
|
+
"testing-strategy", "terminal-cli-dev"]
|
|
123
|
+
rows = [
|
|
124
|
+
{"concern": c, "skill": owners[i % len(owners)],
|
|
125
|
+
"conclusion": f"The single-shot wrapper probe covers {c} within its fixture.",
|
|
126
|
+
"evidence_refs": ["wrapper-probe"]}
|
|
127
|
+
for i, c in enumerate(required)
|
|
104
128
|
]
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
129
|
+
covered = {r["skill"] for r in rows}
|
|
130
|
+
rows += [
|
|
131
|
+
{"concern": required[0], "skill": o,
|
|
132
|
+
"conclusion": f"{o} is covered for {required[0]} by the same probe.",
|
|
133
|
+
"evidence_refs": ["wrapper-probe"]}
|
|
134
|
+
for o in owners if o not in covered
|
|
108
135
|
]
|
|
109
|
-
|
|
110
|
-
"intent": "Prove the extraction wrapper
|
|
111
|
-
"acceptance": ["
|
|
112
|
-
"self_review":
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
"skill": skills[concern],
|
|
116
|
-
"conclusion": conclusion,
|
|
117
|
-
"evidence_refs": ["real-controller-differential"],
|
|
118
|
-
}
|
|
119
|
-
for concern, conclusion in conclusions.items()
|
|
120
|
-
]
|
|
121
|
-
+ [
|
|
122
|
-
{
|
|
123
|
-
"concern": concern,
|
|
124
|
-
"skill": owner,
|
|
125
|
-
"conclusion": f"{owner} is covered for {concern} by the same differential probe.",
|
|
126
|
-
"evidence_refs": ["real-controller-differential"],
|
|
127
|
-
}
|
|
128
|
-
for concern, owner in extra_rows
|
|
129
|
-
],
|
|
130
|
-
"evidence": [
|
|
131
|
-
{
|
|
132
|
-
"id": "real-controller-differential",
|
|
133
|
-
"result": "The same real controller is invoked directly and through the extraction wrapper.",
|
|
134
|
-
}
|
|
135
|
-
],
|
|
136
|
-
}
|
|
137
|
-
plan_path.write_text(json.dumps(plan, indent=2) + "\n", encoding="utf-8")
|
|
136
|
+
plan_path.write_text(json.dumps({
|
|
137
|
+
"intent": "Prove the extraction wrapper drives the real controller single-shot.",
|
|
138
|
+
"acceptance": ["Both passes reach the reviewer-selection boundary untracked."],
|
|
139
|
+
"self_review": rows,
|
|
140
|
+
"evidence": [{"id": "wrapper-probe", "result": "The real controller is invoked through the wrapper."}],
|
|
141
|
+
}, indent=2) + "\n", encoding="utf-8")
|
|
138
142
|
PY
|
|
139
143
|
|
|
140
|
-
real_args=(
|
|
141
|
-
--mode review --cwd "$ROOT" --diff-file "$TMP/real-controller.diff"
|
|
142
|
-
--implementer-family openai --review-plan-file "$TMP/real-controller-plan.json"
|
|
143
|
-
--stage build --review-harness --timeout 5 --total-timeout 5
|
|
144
|
-
)
|
|
145
|
-
# Make a Codex executable visibly available, independent of whichever CLI
|
|
146
|
-
# version the host carries. Because the implementer is OpenAI-family, the real
|
|
147
|
-
# controller must reject Codex as same-family before probing or invoking it.
|
|
148
144
|
mkdir -p "$TMP/fake-bin"
|
|
149
|
-
printf '%s\n'
|
|
150
|
-
'#!/usr/bin/env bash' \
|
|
151
|
-
'printf invoked >"$CODEX_MARKER"' \
|
|
152
|
-
'exit 99' >"$TMP/fake-bin/codex"
|
|
145
|
+
printf '%s\n' '#!/usr/bin/env bash' 'printf invoked >"$CODEX_MARKER"' 'exit 99' >"$TMP/fake-bin/codex"
|
|
153
146
|
chmod +x "$TMP/fake-bin/codex"
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
wrapped_out="$(CODEX_MARKER="$TMP/codex-invoked" PATH="$TMP/fake-bin:$PATH" \
|
|
159
|
-
CODE_REVIEW_CLIENT_ORDER=codex "$WRAPPER" "${real_args[@]}" \
|
|
160
|
-
--review-chain-id extraction-wrapper-real-controller \
|
|
161
|
-
--autonomous-review-index 1 2>&1)"
|
|
162
|
-
wrapped_rc=$?
|
|
163
|
-
set -e
|
|
164
|
-
assert_rc "$direct_rc" 2 "direct real controller must stop before model inference"
|
|
165
|
-
assert_rc "$wrapped_rc" 2 "wrapped real controller must stop before model inference"
|
|
166
|
-
assert_contains '"reason_code":"no_independent_reviewer_available"' "$direct_out" "direct real-controller boundary"
|
|
167
|
-
assert_contains '"reason_code":"no_independent_reviewer_available"' "$wrapped_out" "wrapped real-controller boundary"
|
|
168
|
-
assert_contains '"challenge_budget":0' "$direct_out" "generic controller default budget"
|
|
169
|
-
assert_contains '"challenge_budget":1' "$wrapped_out" "wrapper-enforced real-controller budget"
|
|
170
|
-
[ ! -e "$TMP/codex-invoked" ] || fail "same-family Codex executable was invoked"
|
|
171
|
-
|
|
172
|
-
# Join the producer and consumer contracts. First feed the exact real receipt
|
|
173
|
-
# to the validator and reach its semantic status boundary. Then normalize only
|
|
174
|
-
# the terminal disposition fields so validation must walk the producer's real
|
|
175
|
-
# chain/scope shape before stopping at the intentionally omitted base evidence.
|
|
176
|
-
printf '%s\n' "$wrapped_out" >"$TMP/wrapped-real-controller.out"
|
|
177
|
-
python3 - "$TMP/wrapped-real-controller.out" "$TMP" <<'PY'
|
|
178
|
-
import copy
|
|
179
|
-
import hashlib
|
|
180
|
-
import json
|
|
181
|
-
import sys
|
|
182
|
-
from pathlib import Path
|
|
183
|
-
|
|
184
|
-
output_path = Path(sys.argv[1])
|
|
185
|
-
root = Path(sys.argv[2])
|
|
186
|
-
receipt = None
|
|
187
|
-
for line in reversed(output_path.read_text(encoding="utf-8").splitlines()):
|
|
188
|
-
try:
|
|
189
|
-
value = json.loads(line)
|
|
190
|
-
except json.JSONDecodeError:
|
|
191
|
-
continue
|
|
192
|
-
if isinstance(value, dict) and value.get("schema_version") == 3:
|
|
193
|
-
receipt = value
|
|
194
|
-
break
|
|
195
|
-
assert receipt is not None
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
def write_receipt_and_ledger(stem, value, controller_state):
|
|
199
|
-
encoded = (json.dumps(value, ensure_ascii=False, indent=2) + "\n").encode()
|
|
200
|
-
receipt_name = f"{stem}-receipt.json"
|
|
201
|
-
(root / receipt_name).write_bytes(encoded)
|
|
202
|
-
digest = hashlib.sha256(encoded).hexdigest()
|
|
203
|
-
ledger = {
|
|
204
|
-
"schema_version": 3,
|
|
205
|
-
"candidate_sha256": value["candidate_sha256"],
|
|
206
|
-
"controller_receipts": [
|
|
207
|
-
{"sequence": 1, "file": receipt_name, "sha256": digest}
|
|
208
|
-
],
|
|
209
|
-
"completion_receipt": None,
|
|
210
|
-
"base_attestations": [],
|
|
211
|
-
"autonomous_round": 1,
|
|
212
|
-
"controller_review_state": controller_state,
|
|
213
|
-
"finding_classes": [],
|
|
214
|
-
"unreviewed_delta": [],
|
|
215
|
-
"closeout_state": "ready_for_human_decision",
|
|
216
|
-
}
|
|
217
|
-
(root / f"{stem}-ledger.json").write_text(
|
|
218
|
-
json.dumps(ledger, ensure_ascii=False, indent=2) + "\n",
|
|
219
|
-
encoding="utf-8",
|
|
220
|
-
)
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
write_receipt_and_ledger("real-emitted", receipt, receipt.get("review_state", "inconclusive"))
|
|
224
|
-
normalized = copy.deepcopy(receipt)
|
|
225
|
-
scope_before = json.dumps(normalized["review_scope"], sort_keys=True)
|
|
226
|
-
normalized.update(
|
|
227
|
-
status="passed",
|
|
228
|
-
findings=[],
|
|
229
|
-
review_state="reviewed",
|
|
230
|
-
human_decision_required=False,
|
|
147
|
+
real_args=(
|
|
148
|
+
--cwd "$ROOT" --diff-file "$TMP/real.diff" --implementer-family openai
|
|
149
|
+
--review-plan-file "$TMP/real-plan.json" --stage build --review-harness
|
|
150
|
+
--timeout 5 --total-timeout 5
|
|
231
151
|
)
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
set +e
|
|
237
|
-
emitted_state_out="$(python3 "$REAL_VALIDATOR" "$TMP/real-emitted-ledger.json" 2>&1)"
|
|
238
|
-
emitted_state_rc=$?
|
|
239
|
-
scope_state_out="$(python3 "$REAL_VALIDATOR" "$TMP/real-scope-ledger.json" 2>&1)"
|
|
240
|
-
scope_state_rc=$?
|
|
241
|
-
set -e
|
|
242
|
-
assert_rc "$emitted_state_rc" 1 "exact real receipt reaches semantic status validation"
|
|
243
|
-
assert_contains "must have status passed or findings" "$emitted_state_out" "exact real receipt semantic boundary"
|
|
244
|
-
assert_rc "$scope_state_rc" 1 "producer-derived receipt reaches post-scope validation"
|
|
245
|
-
assert_contains "base_attestations must be a non-empty array" "$scope_state_out" "real producer scope shape"
|
|
246
|
-
|
|
247
|
-
# Shorter spellings that do not start with --challenge-b are ambiguous among
|
|
248
|
-
# the controller's budget/index/classes options, so they fail closed instead of
|
|
249
|
-
# becoming a hidden budget override.
|
|
250
|
-
for abbreviated in --challeng --challenge=4; do
|
|
251
|
-
set +e
|
|
252
|
-
out="$(CODE_REVIEW_CLIENT_ORDER=codex "$WRAPPER" "${real_args[@]}" "$abbreviated" 4 2>&1)"
|
|
253
|
-
rc=$?
|
|
254
|
-
set -e
|
|
255
|
-
assert_rc "$rc" 2 "ambiguous budget abbreviation must fail closed"
|
|
256
|
-
assert_contains "ambiguous option" "$out" "ambiguous abbreviation reason"
|
|
257
|
-
done
|
|
258
|
-
|
|
259
|
-
for spelling in --challenge-budget --challenge-budget=4 --challenge-b=4; do
|
|
260
|
-
: >"$TMP/args"
|
|
152
|
+
for pass in review challenge; do
|
|
153
|
+
extra=()
|
|
154
|
+
[ "$pass" = challenge ] && extra=(--focus "single-shot probe")
|
|
261
155
|
set +e
|
|
262
|
-
out="$(
|
|
156
|
+
out="$(CODEX_MARKER="$TMP/codex-invoked" PATH="$TMP/fake-bin:$PATH" CODE_REVIEW_CLIENT_ORDER=codex \
|
|
157
|
+
"$WRAPPER" --mode "$pass" "${real_args[@]}" "${extra[@]}" 2>&1)"
|
|
263
158
|
rc=$?
|
|
264
159
|
set -e
|
|
265
|
-
assert_rc "$rc" 2 "
|
|
266
|
-
assert_contains "
|
|
267
|
-
|
|
160
|
+
assert_rc "$rc" 2 "real $pass must stop before model inference"
|
|
161
|
+
assert_contains '"reason_code":"no_independent_reviewer_available"' "$out" "real $pass reaches reviewer selection"
|
|
162
|
+
assert_not_contains 'review_chain_required' "$out" "real $pass needs no chain"
|
|
163
|
+
assert_not_contains 'review_chain_invalid' "$out" "real $pass needs no chain"
|
|
268
164
|
done
|
|
165
|
+
[ ! -e "$TMP/codex-invoked" ] || fail "same-family Codex executable was invoked"
|
|
269
166
|
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
--mode review --cwd /synthetic --base base --implementer-family openai >/dev/null 2>&1
|
|
273
|
-
rc=$?
|
|
274
|
-
set -e
|
|
275
|
-
assert_rc "$rc" 7 "wrapper must preserve controller exit status"
|
|
276
|
-
|
|
277
|
-
mv "$TMP/skills/code-review/scripts/review_gate.sh" "$TMP/controller-away"
|
|
278
|
-
set +e
|
|
279
|
-
out="$(CAPTURE_PATH="$TMP/args" "$TMP/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh" \
|
|
280
|
-
--mode review --cwd /synthetic --base base --implementer-family openai 2>&1)"
|
|
281
|
-
rc=$?
|
|
282
|
-
set -e
|
|
283
|
-
assert_rc "$rc" 2 "missing controller must fail closed"
|
|
284
|
-
assert_contains "controller is unavailable" "$out" "missing-controller reason"
|
|
285
|
-
|
|
286
|
-
# The owner-facing start-here documents must route non-wording extraction work
|
|
287
|
-
# through the fixed-budget wrapper and its terminal validator. Strict
|
|
288
|
-
# wording-only changes retain the documented single-review exception; they must
|
|
289
|
-
# not be accidentally pulled into the multi-round ledger contract.
|
|
167
|
+
# The owner documents must route non-wording work through this wrapper and must
|
|
168
|
+
# no longer send anyone to the retired chain ledger or merge-side binder.
|
|
290
169
|
python3 - "$ROOT" <<'PY'
|
|
291
170
|
import re
|
|
292
171
|
import sys
|
|
293
172
|
from pathlib import Path
|
|
294
173
|
|
|
295
174
|
root = Path(sys.argv[1])
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
).read_text(encoding="utf-8")
|
|
300
|
-
dual
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
"SKILL": skill,
|
|
318
|
-
"quickstart": quickstart,
|
|
319
|
-
"dual-track": dual,
|
|
320
|
-
"validation-and-landing": landing,
|
|
321
|
-
}.items():
|
|
322
|
-
assert "scripts/extraction_review_gate.sh" in text, (
|
|
323
|
-
f"{label} does not route the non-wording lane through the owner wrapper"
|
|
324
|
-
)
|
|
325
|
-
|
|
326
|
-
for label, text in {"quickstart": quickstart, "dual-track": dual}.items():
|
|
327
|
-
assert "scripts/validate_extraction_review_state.py <closeout.json>" in text, (
|
|
328
|
-
f"{label} omits the terminal closeout validator"
|
|
329
|
-
)
|
|
330
|
-
|
|
331
|
-
assert re.search(
|
|
332
|
-
r"[Nn]on-wording.{0,240}scripts/extraction_review_gate\.sh",
|
|
333
|
-
skill,
|
|
334
|
-
re.DOTALL,
|
|
335
|
-
), "SKILL does not scope the fixed-budget wrapper to non-wording changes"
|
|
336
|
-
assert re.search(
|
|
337
|
-
r"[Ww]ording-only.{0,500}(?:single|one)[- ](?:round|review)",
|
|
338
|
-
quickstart,
|
|
339
|
-
re.DOTALL,
|
|
340
|
-
), "quickstart lost the strict wording-only single-review exception"
|
|
341
|
-
for label, text in {
|
|
342
|
-
"quickstart": quickstart,
|
|
343
|
-
"dual-track": dual,
|
|
344
|
-
"validation-and-landing": landing,
|
|
345
|
-
"code-review": code_review,
|
|
346
|
-
}.items():
|
|
347
|
-
assert "wording_only_boundary" in text, (
|
|
348
|
-
f"{label} does not require independent wording-only semantic confirmation"
|
|
349
|
-
)
|
|
350
|
-
assert "wording-only-review.md" in staged
|
|
175
|
+
ref = root / "skills/skill-extraction-workflow"
|
|
176
|
+
docs = {
|
|
177
|
+
"SKILL": (ref / "SKILL.md").read_text(encoding="utf-8"),
|
|
178
|
+
"quickstart": (ref / "references/extraction-quickstart.md").read_text(encoding="utf-8"),
|
|
179
|
+
"dual-track": (ref / "references/dual-track-review-gate.md").read_text(encoding="utf-8"),
|
|
180
|
+
"validation-and-landing": (ref / "references/validation-and-landing.md").read_text(encoding="utf-8"),
|
|
181
|
+
}
|
|
182
|
+
for label, text in docs.items():
|
|
183
|
+
assert "scripts/extraction_review_gate.sh" in text, f"{label} does not route through the wrapper"
|
|
184
|
+
for retired in ("validate_extraction_review_state.py", "review_ledger_binding.py",
|
|
185
|
+
"challenge_budget=1", "succession challenge"):
|
|
186
|
+
assert retired not in text, f"{label} still points at the retired {retired}"
|
|
187
|
+
assert re.search(r"[Nn]on-wording.{0,240}scripts/extraction_review_gate\.sh",
|
|
188
|
+
docs["SKILL"], re.DOTALL), "SKILL does not scope the wrapper to non-wording work"
|
|
189
|
+
assert re.search(r"[Ww]ording-only.{0,500}(?:single|one)[- ](?:round|review|pass)",
|
|
190
|
+
docs["quickstart"], re.DOTALL), "quickstart lost the wording-only single-review exception"
|
|
191
|
+
assert "--challenge-budget" not in docs["quickstart"], "quickstart must not hand callers the budget flag"
|
|
192
|
+
dual = docs["dual-track"]
|
|
193
|
+
for pinned in ("post-review delta", "Every post-review delta gets a delta pass", "After five delta passes", "never left to a human reader"):
|
|
194
|
+
assert pinned in dual, f"dual-track lost '{pinned}'"
|
|
195
|
+
wording_only = (root / "skills/code-review/references/wording-only-review.md").read_text(encoding="utf-8")
|
|
351
196
|
assert "--wording-only-proof-file" in wording_only
|
|
352
197
|
assert "--challenge-budget 0" in wording_only
|
|
353
|
-
assert "markdown-punctuation-only" in wording_only
|
|
354
|
-
assert "markdown-token-replacement" in wording_only
|
|
355
|
-
assert "opens no challenge chain or `complete` checkpoint" in dual
|
|
356
|
-
assert "codex review --base" not in quickstart
|
|
357
|
-
assert "codex exec adversarial" not in quickstart
|
|
358
|
-
assert "--challenge-budget" not in quickstart, (
|
|
359
|
-
"quickstart must not let callers override the extraction review budget"
|
|
360
|
-
)
|
|
361
|
-
assert re.search(
|
|
362
|
-
r"[Rr]ound 2 challenge.{0,500}ready_for_human_decision",
|
|
363
|
-
quickstart,
|
|
364
|
-
re.DOTALL,
|
|
365
|
-
), "quickstart does not validate an early-clean round-2 terminal checkpoint"
|
|
366
|
-
assert re.search(
|
|
367
|
-
r"[Rr]ound 2 findings.{0,200}continuation_authorization_required",
|
|
368
|
-
quickstart,
|
|
369
|
-
re.DOTALL,
|
|
370
|
-
), "quickstart does not validate the exhausted-budget terminal checkpoint"
|
|
371
|
-
assert "baseline_race" in quickstart
|
|
372
|
-
assert "at most two challenges" not in quickstart, (
|
|
373
|
-
"quickstart still advertises the retired two-challenge budget"
|
|
374
|
-
)
|
|
375
|
-
assert "At the third Agent-autonomous round" not in dual, (
|
|
376
|
-
"dual-track still gates the lane at the retired third round"
|
|
377
|
-
)
|
|
378
198
|
PY
|
|
379
199
|
|
|
380
200
|
echo "test_extraction_review_gate: ok"
|
|
@@ -29,8 +29,8 @@ pass() { passed=$((passed + 1)); echo "PASS: $*"; }
|
|
|
29
29
|
# guard silently stale again (the observed drift shape: a 16 guarding 13
|
|
30
30
|
# executed cases, accumulated while this suite could not run).
|
|
31
31
|
static_pass_calls="$(grep -c '^pass "' "$0")"
|
|
32
|
-
[ "$static_pass_calls" = "
|
|
33
|
-
|| fail "pass-call inventory drifted: counted $static_pass_calls, guard expects
|
|
32
|
+
[ "$static_pass_calls" = "21" ] \
|
|
33
|
+
|| fail "pass-call inventory drifted: counted $static_pass_calls, guard expects 21"
|
|
34
34
|
pass "executed-count guard matches the script's own pass-call inventory"
|
|
35
35
|
|
|
36
36
|
REPO="$TMP/repo"
|
|
@@ -478,6 +478,52 @@ case "$out" in
|
|
|
478
478
|
esac
|
|
479
479
|
pass "a second row cannot inherit a one-row historical locator waiver"
|
|
480
480
|
|
|
481
|
+
# A retired locator that several historical rows cited is waived as ONE entry
|
|
482
|
+
# binding every one of those rows by digest. Each row stays individually pinned:
|
|
483
|
+
# removing one, rewriting one, or adding one more citation must each red.
|
|
484
|
+
MULTI_LOCATOR="command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh"
|
|
485
|
+
multi_row_case() {
|
|
486
|
+
multi_case="$1"
|
|
487
|
+
git -C "$REPO" checkout -- . >/dev/null 2>&1 \
|
|
488
|
+
|| fail "could not restore the clone before the multi-row waiver $multi_case case"
|
|
489
|
+
python3 - "$REGISTER" "$MULTI_LOCATOR" "$multi_case" <<'PY'
|
|
490
|
+
from pathlib import Path
|
|
491
|
+
import sys
|
|
492
|
+
|
|
493
|
+
path = Path(sys.argv[1])
|
|
494
|
+
locator, case = sys.argv[2], sys.argv[3]
|
|
495
|
+
lines = path.read_text(encoding="utf-8").splitlines(keepends=True)
|
|
496
|
+
matches = [i for i, line in enumerate(lines) if line.startswith("|") and locator in line]
|
|
497
|
+
assert len(matches) > 1, f"expected several real waived rows, found {len(matches)}"
|
|
498
|
+
index = matches[len(matches) // 2]
|
|
499
|
+
if case == "delete":
|
|
500
|
+
del lines[index]
|
|
501
|
+
elif case == "rewrite":
|
|
502
|
+
lines[index] = lines[index].replace("| ", "| Fixture-rewritten claim: ", 1)
|
|
503
|
+
else:
|
|
504
|
+
lines.insert(index + 1, lines[index])
|
|
505
|
+
path.write_text("".join(lines), encoding="utf-8")
|
|
506
|
+
PY
|
|
507
|
+
run_check
|
|
508
|
+
[ "$rc" = "1" ] || { dump; fail "multi-row waiver $multi_case must be rc=1, got rc=$rc"; }
|
|
509
|
+
case "$multi_case:$out" in
|
|
510
|
+
delete:*"EXEMPT entry has no citing row in the ledger for 1 of"*) : ;;
|
|
511
|
+
rewrite:*"EXEMPT citing row does not match the waived row"*) : ;;
|
|
512
|
+
duplicate:*"EXEMPT locator cited by"*"rows (allowance"*) : ;;
|
|
513
|
+
*) dump; fail "multi-row waiver $multi_case failed, but not via its exact diagnostic" ;;
|
|
514
|
+
esac
|
|
515
|
+
case "$out" in
|
|
516
|
+
*ccl_skill_check_clean_ok*) dump; fail "a multi-row waiver $multi_case must never yield a clean-landing token" ;;
|
|
517
|
+
*) : ;;
|
|
518
|
+
esac
|
|
519
|
+
}
|
|
520
|
+
multi_row_case delete
|
|
521
|
+
pass "multi-row waiver: deleting one citing row fails closed"
|
|
522
|
+
multi_row_case rewrite
|
|
523
|
+
pass "multi-row waiver: rewriting one citing row fails closed"
|
|
524
|
+
multi_row_case duplicate
|
|
525
|
+
pass "multi-row waiver: a further citing row fails closed"
|
|
526
|
+
|
|
481
527
|
# RED: adding an EXEMPT key without a row digest silently downgrades identity
|
|
482
528
|
# binding unless the production gate rejects the incomplete waiver definition.
|
|
483
529
|
# The checker resolves the gate beside itself, so mutate STUB_GATE — never the
|
|
@@ -547,5 +593,5 @@ case "$out" in
|
|
|
547
593
|
esac
|
|
548
594
|
pass "a failure inside the cleanup-trap window keeps its exit status"
|
|
549
595
|
|
|
550
|
-
[ "$passed" -eq
|
|
596
|
+
[ "$passed" -eq 21 ] || fail "expected 21 assertions, saw $passed"
|
|
551
597
|
echo "register_firing_path_wiring_tests_ok ($passed assertions)"
|