@ccoalm/ccl-skills 0.15.5 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +15 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +40 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +99 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +177 -89
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +162 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +236 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +68 -0
- package/dist/assets/release.json +12 -12
- package/package.json +1 -1
|
@@ -134,17 +134,16 @@ echo "code_review_skill_dir=$CODE_REVIEW_SKILL_DIR" >&2
|
|
|
134
134
|
: "${REVIEW_CHAIN_ID:?set REVIEW_CHAIN_ID to a task-scoped chain id: letters, digits, dot, underscore, hyphen only}"
|
|
135
135
|
: "${REVIEW_STAGE:?set REVIEW_STAGE to the stage this candidate is actually at: explore, build, or release}"
|
|
136
136
|
: "${REVIEW_EVIDENCE_DIR:?set REVIEW_EVIDENCE_DIR to a durable directory you control for the per-round result rows}"
|
|
137
|
-
#
|
|
138
|
-
#
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
elif [ -n "${REVIEW_BASE:-}" ]; then
|
|
144
|
-
PACKET_ARGS=(--base "$REVIEW_BASE")
|
|
145
|
-
else
|
|
146
|
-
echo "set exactly one of REVIEW_DIFF_FILE or REVIEW_BASE" >&2; exit 1
|
|
137
|
+
# REVIEW_BASE names the candidate, REVIEW_DIFF_FILE widens what the reviewer reads,
|
|
138
|
+
# and BOTH is a widened packet that must BEGIN with the candidate (composition
|
|
139
|
+
# rules: references/staged-review-contract.md). if-blocks, not `[ -n ... ] && ...`: a
|
|
140
|
+
# trailing false test returns non-zero and `set -e` would kill the caller.
|
|
141
|
+
if [ -z "${REVIEW_DIFF_FILE:-}" ] && [ -z "${REVIEW_BASE:-}" ]; then
|
|
142
|
+
echo "set REVIEW_BASE, REVIEW_DIFF_FILE, or both" >&2; exit 1
|
|
147
143
|
fi
|
|
144
|
+
PACKET_ARGS=()
|
|
145
|
+
if [ -n "${REVIEW_BASE:-}" ]; then PACKET_ARGS+=(--base "$REVIEW_BASE"); fi
|
|
146
|
+
if [ -n "${REVIEW_DIFF_FILE:-}" ]; then PACKET_ARGS+=(--diff-file "$REVIEW_DIFF_FILE"); fi
|
|
148
147
|
# REVIEW_RUN_DIR holds the raw round-1 result only for the chain handoff; the durable
|
|
149
148
|
# per-round evidence is persisted to REVIEW_EVIDENCE_DIR, whose confidentiality you own.
|
|
150
149
|
REVIEW_RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/review-run.XXXXXX")" || exit 1
|
|
@@ -199,10 +198,11 @@ bash "$CODE_REVIEW_SKILL_DIR/scripts/review_gate.sh" \
|
|
|
199
198
|
>"$REVIEW_RUN_DIR/round2.json"
|
|
200
199
|
require_tracked_result "$REVIEW_RUN_DIR/round2.json" challenge 2
|
|
201
200
|
# Both rounds must bind the SAME candidate: the chain accepts older candidate hashes,
|
|
202
|
-
# so a
|
|
201
|
+
# so a candidate edited between rounds would otherwise be persisted as one coherent
|
|
202
|
+
# pair. Their packets may differ; each receipt records the packet it actually read.
|
|
203
203
|
python3 -c 'import json,sys; a=json.load(open(sys.argv[1])); b=json.load(open(sys.argv[2])); h=a.get("candidate_sha256"); sys.exit(0 if h and h==b.get("candidate_sha256") else 1)' \
|
|
204
204
|
"$ROUND1_RESULT_FILE" "$REVIEW_RUN_DIR/round2.json" \
|
|
205
|
-
|| { echo "round 2
|
|
205
|
+
|| { echo "round 2 bound a different candidate than round 1; rerun the pair on one frozen candidate" >&2; exit 1; }
|
|
206
206
|
cp "$REVIEW_RUN_DIR/round2.json" "$EVIDENCE_RUN_DIR/round2-challenge.json" || exit 1
|
|
207
207
|
cat "$EVIDENCE_RUN_DIR/round2-challenge.json"
|
|
208
208
|
# A secret-free diff egresses to non-Claude reviewers automatically; add
|
|
@@ -216,10 +216,10 @@ Run the script by path while keeping `--cwd` pointed at the product repository u
|
|
|
216
216
|
**The packet is the reviewer's whole world — compose it deliberately.** Review and challenge are built packet-bounded — Claude runs `--tools ""` with no `--add-dir`, and the other wrappers run in an isolated run workspace or a packet-only read surface. Treat the packet as the reviewer's whole world when deciding coverage: it is the only content bound by the packet hash and scanned before egress, so anything outside it is neither reliably visible to the reviewer nor covered by the verdict; a diff-only packet surfaces defects visible inside the changed lines and little else, and `--paths` only narrows it further. Whatever is absent from the packet is unreachable, not merely missed: a contradiction with an unchanged sibling clause, drift against a carrier outside the diff, or a silent weakening of upstream wording cannot be found by a reviewer who never saw the other side — that is the packet's shape, not the reviewer's weakness.
|
|
217
217
|
|
|
218
218
|
- Codex permits frozen-packet read/search; see [tool boundaries](references/development-completion.md#review-tools).
|
|
219
|
-
- To widen the packet, assemble it yourself and pass `--diff-file
|
|
220
|
-
- A verdict covers exactly the packet it was taken on
|
|
219
|
+
- To widen the packet, assemble it yourself and pass `--diff-file`, plus `--base` whenever the round must bind a landing candidate ([composition rules](references/staged-review-contract.md#the-packet-and-the-candidate)). It must name a regular file (no symlink or hardlink) holding text without NUL bytes. The gate hard-caps a packet at 200,000 bytes; split a larger candidate as described in the next bullet.
|
|
220
|
+
- A verdict covers exactly the packet it was taken on; the receipt records `packet_sha256` for those bytes and `candidate_sha256` for the base-derived candidate that will land, equal unless the packet was widened. A candidate too large for one packet is split by file group or risk class into a partition that still covers the whole candidate — every part in some packet, none dropped — each partition's verdict recorded against its own packet hash, and the candidate-wide claim withheld until every partition is conclusive; one partition's `no blocking findings` is never a verdict on the landing candidate. Cross-partition contradictions are unreachable by construction, so repeat the shared canonical context in every partition's packet and review anything that spans partitions as its own packet.
|
|
221
221
|
- Added context egresses to the selected reviewer exactly like the diff does, through the same credential tripwire — which catches machine-detectable secrets only. Paste rule text, carriers, and tool output; never paste credentials or material you would not send to that provider.
|
|
222
|
-
- A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane rather than editing the candidate to satisfy it.
|
|
222
|
+
- A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane, keeping `--base` so it still binds the same candidate, rather than editing the candidate to satisfy it.
|
|
223
223
|
|
|
224
224
|
When intentionally reviewing `code-review` itself, override the resolver from the ccl-skills repo under review before invoking the gate:
|
|
225
225
|
|
|
@@ -184,6 +184,46 @@ commit an in-scope path when Git should represent it; or compose complete
|
|
|
184
184
|
`--diff-file` partitions when the candidate must be split. Never omit a path
|
|
185
185
|
and report the remaining packet as the whole candidate.
|
|
186
186
|
|
|
187
|
+
## The packet and the candidate
|
|
188
|
+
|
|
189
|
+
They are two objects. The **packet** is what the reviewer reads; the **candidate**
|
|
190
|
+
is what will land and what `review_ledger_binding.py` recomputes at merge time.
|
|
191
|
+
A receipt records both hashes.
|
|
192
|
+
|
|
193
|
+
They hold the same value when the packet came from `--base` alone. Pass
|
|
194
|
+
`--diff-file` **with** `--base`/`--paths` to widen what the reviewer reads while
|
|
195
|
+
the round still binds the landing candidate — the shape an evidence-gap finding
|
|
196
|
+
needs, since editing the candidate would answer an input defect with a candidate
|
|
197
|
+
change. `--diff-file` alone binds no landing; only the combined form rejects a
|
|
198
|
+
`--wording-only-proof-file`.
|
|
199
|
+
|
|
200
|
+
What makes the widened form safe is a **prefix requirement**: the packet begins
|
|
201
|
+
with the base-derived candidate, byte for byte, so nothing lands unread.
|
|
202
|
+
|
|
203
|
+
- **Append context after the candidate diff.** Putting anything before the
|
|
204
|
+
candidate fails, and that is not cosmetic: a packet preceding it with a decoy
|
|
205
|
+
diff would read as the change while the real candidate read as context.
|
|
206
|
+
Interleaving context inside the candidate fails, as does dropping any part of
|
|
207
|
+
it. The reviewer is told where the candidate ends — the profile carries
|
|
208
|
+
`candidate_bytes` and states that exactly the first N packet bytes land — so
|
|
209
|
+
appended hunks that continue or seem to revert the diff cannot pass as it.
|
|
210
|
+
- **Keep the packet file outside the repository, and put nothing else in the
|
|
211
|
+
tree while the rounds run.** The controller counts every untracked path into
|
|
212
|
+
the candidate; the binder counts only committed content minus the receipt JSON
|
|
213
|
+
a round adds. A packet file, a superseded round's receipt, or any scratch
|
|
214
|
+
artifact in the worktree therefore moves the candidate the rounds bind and the
|
|
215
|
+
binder never computes it — the mirror of committing a plain-text attestation
|
|
216
|
+
after the rounds. Bound evidence lands before the rounds, receipts after,
|
|
217
|
+
nothing else present.
|
|
218
|
+
- **Read the candidate identity, do not reconstruct it.** `--print-candidate`
|
|
219
|
+
is the authority: its base is a fork point, its paths carry the round's
|
|
220
|
+
exclusions, and it refuses an uncommitted tree.
|
|
221
|
+
- Worth adding beyond the diff — the canonical rule the changed lines must not
|
|
222
|
+
contradict, sibling clauses, the carriers restating the change, gate output.
|
|
223
|
+
|
|
224
|
+
Rounds in one chain agree on the **candidate**, not the packet, which lets a
|
|
225
|
+
later round read more than an earlier one.
|
|
226
|
+
|
|
187
227
|
## Proof-bound wording-only single review
|
|
188
228
|
|
|
189
229
|
The wording-only exception is one untracked `review` with
|
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh
CHANGED
|
@@ -16,7 +16,7 @@ MAX_PROMPT_BYTES=245000
|
|
|
16
16
|
CHALLENGE_CLASSES="race conditions, data loss, security holes, auth bypass, lost or duplicated work, operational footguns"
|
|
17
17
|
|
|
18
18
|
emit_inconclusive() {
|
|
19
|
-
python3 - "$MODE" "$1" "${2:-invalid_input}" "${3:-false}" "${4:-}" <<'PY'
|
|
19
|
+
python3 - "$MODE" "$1" "${2:-invalid_input}" "${3:-false}" "${4:-}" "${5:-}" "${6:-}" <<'PY'
|
|
20
20
|
import json, sys
|
|
21
21
|
payload = {
|
|
22
22
|
"reviewer": "codex",
|
|
@@ -31,6 +31,10 @@ payload = {
|
|
|
31
31
|
}
|
|
32
32
|
if sys.argv[5]:
|
|
33
33
|
payload["transport_exit_code"] = int(sys.argv[5]) if sys.argv[5].isdigit() else sys.argv[5]
|
|
34
|
+
if sys.argv[6]:
|
|
35
|
+
payload["transport_diagnostic"] = sys.argv[6]
|
|
36
|
+
if sys.argv[7]:
|
|
37
|
+
payload["transport_run_dir"] = sys.argv[7]
|
|
34
38
|
print(json.dumps(payload, ensure_ascii=False, separators=(",", ":")))
|
|
35
39
|
PY
|
|
36
40
|
}
|
|
@@ -179,7 +183,15 @@ if [ "$REVIEW_SKILL_COUNT" -gt 0 ]; then
|
|
|
179
183
|
|| die_inconclusive codex_installed_skill_binding_invalid binding_mismatch false
|
|
180
184
|
fi
|
|
181
185
|
RUN_ROOT="$(mktemp -d "${TMPDIR:-/tmp}/codex-review.XXXXXX")"
|
|
182
|
-
|
|
186
|
+
# A failure that deletes its own evidence is the defect this round started from:
|
|
187
|
+
# rounds 122 and 123 left six receipts and no account of why the lane failed,
|
|
188
|
+
# because both captured streams went out with the run directory. On a transport
|
|
189
|
+
# failure the directory stays, and the receipt names it. It is mode 0700 under
|
|
190
|
+
# TMPDIR and holds exactly what it held while the run was in flight, so nothing
|
|
191
|
+
# is exposed that was not already; reclaiming it is the platform's temp-directory
|
|
192
|
+
# lifetime, as it is for every other run directory here.
|
|
193
|
+
PRESERVE_RUN_ROOT=0
|
|
194
|
+
cleanup() { [ "$PRESERVE_RUN_ROOT" = 1 ] || rm -rf "$RUN_ROOT"; }
|
|
183
195
|
trap cleanup EXIT
|
|
184
196
|
signal_inconclusive() {
|
|
185
197
|
emit_inconclusive codex_review_terminated operator_interrupt false
|
|
@@ -515,26 +527,102 @@ if [ -n "$AUTH_LINK_TARGET" ]; then
|
|
|
515
527
|
|| die_inconclusive codex_runtime_home_credential_moved binding_mismatch false
|
|
516
528
|
fi
|
|
517
529
|
if [ "$run_rc" != 0 ]; then
|
|
530
|
+
# `codex exec --json` reports supply and credential failures as structured
|
|
531
|
+
# events on stdout, not on stderr, so a classifier reading only stderr sees a
|
|
532
|
+
# quota exhaustion as an unclassifiable failure and stops the reviewer lane
|
|
533
|
+
# instead of cascading.
|
|
534
|
+
#
|
|
535
|
+
# Only TOP-LEVEL error events are read. Model-authored content arrives nested
|
|
536
|
+
# under `item`, and the model quotes the packet, which is untrusted candidate
|
|
537
|
+
# data -- grepping the raw stream would let a reviewed diff pick the verdict
|
|
538
|
+
# for this lane by writing quota vocabulary into itself.
|
|
539
|
+
TRANSPORT_ERRORS="$RUN_ROOT/transport-errors.txt"
|
|
540
|
+
: >"$TRANSPORT_ERRORS"
|
|
541
|
+
python3 - "$EVENTS" >"$TRANSPORT_ERRORS" 2>/dev/null <<'PY_TRANSPORT_ERRORS'
|
|
542
|
+
import json, sys
|
|
543
|
+
from pathlib import Path
|
|
544
|
+
|
|
545
|
+
try:
|
|
546
|
+
lines = Path(sys.argv[1]).read_text(encoding="utf-8", errors="replace").splitlines()
|
|
547
|
+
except OSError:
|
|
548
|
+
sys.exit(0)
|
|
549
|
+
seen = set()
|
|
550
|
+
for line in lines:
|
|
551
|
+
try:
|
|
552
|
+
event = json.loads(line)
|
|
553
|
+
except ValueError:
|
|
554
|
+
continue
|
|
555
|
+
if not isinstance(event, dict):
|
|
556
|
+
continue
|
|
557
|
+
kind = event.get("type")
|
|
558
|
+
if not isinstance(kind, str) or not (kind == "error" or kind.endswith(".failed")):
|
|
559
|
+
continue
|
|
560
|
+
message = event.get("message")
|
|
561
|
+
if not isinstance(message, str):
|
|
562
|
+
nested = event.get("error")
|
|
563
|
+
message = nested.get("message") if isinstance(nested, dict) else None
|
|
564
|
+
if not isinstance(message, str) or not message:
|
|
565
|
+
continue
|
|
566
|
+
# A failing turn repeats the error event verbatim, and the diagnostic is
|
|
567
|
+
# bounded: relaying both would spend half the budget on one sentence.
|
|
568
|
+
message = " ".join(message.split())
|
|
569
|
+
if message not in seen:
|
|
570
|
+
seen.add(message)
|
|
571
|
+
print(message)
|
|
572
|
+
PY_TRANSPORT_ERRORS
|
|
573
|
+
PRESERVE_RUN_ROOT=1
|
|
574
|
+
# Physical paths on both sides, not the literal `$HOME` string: a home spelled
|
|
575
|
+
# with a trailing slash, or reached through a symlink, is the same directory
|
|
576
|
+
# and must elide the same way. Comparing the raw variable would put the
|
|
577
|
+
# username into a committed receipt on exactly those hosts.
|
|
578
|
+
TRANSPORT_RUN_DIR="$(cd "$RUN_ROOT" 2>/dev/null && pwd -P)" || TRANSPORT_RUN_DIR="$RUN_ROOT"
|
|
579
|
+
[ -n "$TRANSPORT_RUN_DIR" ] || TRANSPORT_RUN_DIR="$RUN_ROOT"
|
|
580
|
+
transport_home_real=""
|
|
581
|
+
if [ -n "${HOME:-}" ]; then
|
|
582
|
+
transport_home_real="$(cd "$HOME" 2>/dev/null && pwd -P)" || transport_home_real=""
|
|
583
|
+
fi
|
|
584
|
+
case "$transport_home_real" in
|
|
585
|
+
"" | */) transport_home_real="" ;;
|
|
586
|
+
esac
|
|
587
|
+
if [ -n "$transport_home_real" ]; then
|
|
588
|
+
case "$TRANSPORT_RUN_DIR" in
|
|
589
|
+
"$transport_home_real"/*)
|
|
590
|
+
TRANSPORT_RUN_DIR="~${TRANSPORT_RUN_DIR#"$transport_home_real"}" ;;
|
|
591
|
+
esac
|
|
592
|
+
fi
|
|
593
|
+
# The receipt carries NO text derived from the run. Eight review chains each
|
|
594
|
+
# found a different escape from a filter over that text -- an unlisted key
|
|
595
|
+
# name, an assignment form, URL userinfo, a password containing the separator,
|
|
596
|
+
# an escaped quote, an uppercase scheme -- because "nothing secret-shaped
|
|
597
|
+
# survives" is not a decidable property of free text, and an adversarial
|
|
598
|
+
# reviewer can always spell one more. So the free text is gone: what the
|
|
599
|
+
# transport said stays in the preserved run directory, and the receipt says
|
|
600
|
+
# where that is. The classifier still reads the extracted error messages
|
|
601
|
+
# above; those are matched against fixed patterns and never persisted.
|
|
602
|
+
TRANSPORT_DIAGNOSTIC="the transport output for this failure is in transport_run_dir"
|
|
603
|
+
|
|
518
604
|
if bash "$TIMEOUT_CLASSIFIER" "$run_rc" "$run_elapsed" "$TIMEOUT"; then
|
|
519
|
-
die_inconclusive codex_timeout timeout true "$run_rc"
|
|
605
|
+
die_inconclusive codex_timeout timeout true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
520
606
|
fi
|
|
521
607
|
case "$run_rc" in
|
|
522
|
-
129|130|137|143) die_inconclusive codex_process_interrupted operator_interrupt false "$run_rc" ;;
|
|
608
|
+
129|130|137|143) die_inconclusive codex_process_interrupted operator_interrupt false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR" ;;
|
|
523
609
|
esac
|
|
524
|
-
|
|
525
|
-
|
|
610
|
+
# `usage limit` is the wording the CLI actually uses for an exhausted account;
|
|
611
|
+
# none of the older patterns match it.
|
|
612
|
+
if grep -qiE '429|rate.?limit|quota|usage limit' "$STDERR_FILE" "$TRANSPORT_ERRORS"; then
|
|
613
|
+
die_inconclusive codex_quota quota true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
526
614
|
fi
|
|
527
|
-
if grep -qiE 'unauthori[sz]ed|authentication|login|api key' "$STDERR_FILE"; then
|
|
528
|
-
die_inconclusive codex_auth_unavailable provider_unavailable true "$run_rc"
|
|
615
|
+
if grep -qiE 'unauthori[sz]ed|authentication|login|api key' "$STDERR_FILE" "$TRANSPORT_ERRORS"; then
|
|
616
|
+
die_inconclusive codex_auth_unavailable provider_unavailable true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
529
617
|
fi
|
|
530
618
|
if [ ! -s "$EVENTS" ] && [ ! -s "$RESULT_FILE" ] \
|
|
531
619
|
&& grep -qiE 'failed to initialize in-process app-server client: Operation not permitted' "$STDERR_FILE"; then
|
|
532
620
|
if [ "$HOST_REMEDIATION_ATTEMPTED" -eq 1 ]; then
|
|
533
|
-
die_inconclusive codex_host_path_unavailable_after_host_retry host_path_unavailable_after_host_retry true "$run_rc"
|
|
621
|
+
die_inconclusive codex_host_path_unavailable_after_host_retry host_path_unavailable_after_host_retry true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
534
622
|
fi
|
|
535
|
-
die_inconclusive codex_host_path_unavailable host_path_unavailable false "$run_rc"
|
|
623
|
+
die_inconclusive codex_host_path_unavailable host_path_unavailable false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
536
624
|
fi
|
|
537
|
-
die_inconclusive codex_run_failed unknown_client_failure false "$run_rc"
|
|
625
|
+
die_inconclusive codex_run_failed unknown_client_failure false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
538
626
|
fi
|
|
539
627
|
|
|
540
628
|
python3 "$PARSER" --client codex --mode "$MODE" --implementer-family "$IMPL_FAMILY" \
|