@ccoalm/ccl-skills 0.15.5 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +21 -19
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +113 -123
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/wording-only-review.md +136 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +99 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +297 -100
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +162 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +17 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_order.sh +30 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +445 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +14 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +56 -18
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +41 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +68 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/closeout-reread.md +40 -0
- package/dist/assets/release.json +31 -21
- package/package.json +1 -1
|
@@ -873,6 +873,76 @@ if [ "$behavior" = "no_events" ]; then
|
|
|
873
873
|
fi
|
|
874
874
|
printf '%s\n' '{"type":"thread.started","thread_id":"test-thread"}'
|
|
875
875
|
printf '%s\n' '{"type":"turn.started"}'
|
|
876
|
+
# Transport failures whose only account of themselves is the event stream. The
|
|
877
|
+
# CLI reports these on stdout, so a wrapper that classifies from stderr alone
|
|
878
|
+
# cannot see them.
|
|
879
|
+
if [ "$behavior" = "usage_limit_event" ]; then
|
|
880
|
+
printf '%s\n' '{"type":"error","message":"You'"'"'ve hit your usage limit. Visit https://example.invalid/settings/usage?A1B2C3QUERYSECRET to purchase more credits."}'
|
|
881
|
+
printf '%s\n' '{"type":"turn.failed","error":{"message":"You'"'"'ve hit your usage limit."}}'
|
|
882
|
+
exit 1
|
|
883
|
+
fi
|
|
884
|
+
if [ "$behavior" = "usage_limit_turn_failed" ]; then
|
|
885
|
+
printf '%s\n' '{"type":"turn.failed","error":{"message":"request rejected: rate limit exceeded for this account"}}'
|
|
886
|
+
exit 1
|
|
887
|
+
fi
|
|
888
|
+
if [ "$behavior" = "auth_event" ]; then
|
|
889
|
+
printf '%s\n' '{"type":"error","message":"unauthorized: the stored credential was rejected"}'
|
|
890
|
+
exit 1
|
|
891
|
+
fi
|
|
892
|
+
# The packet is untrusted, and the model quotes it. Quota vocabulary reaching
|
|
893
|
+
# the classifier from here would let a reviewed diff choose its own reviewer.
|
|
894
|
+
if [ "$behavior" = "quota_in_model_output" ]; then
|
|
895
|
+
printf '%s\n' '{"type":"item.completed","item":{"type":"agent_message","text":"the diff mentions 429 rate limit and quota handling CODEXLEAKMARKER7f3a"}}'
|
|
896
|
+
printf '%s\n' '{"type":"item.completed","item":{"type":"error","message":"429 quota rate limit CODEXLEAKMARKER7f3a"}}'
|
|
897
|
+
printf '%s\n' 'stub failed for an unrelated reason' >&2
|
|
898
|
+
exit 1
|
|
899
|
+
fi
|
|
900
|
+
if [ "$behavior" = "sensitive_streams" ]; then
|
|
901
|
+
# The credential-shaped values are assembled at runtime: written whole they
|
|
902
|
+
# would put a real `sk-` token shape in the repository, which the credential
|
|
903
|
+
# scanner in validate-skill.sh refuses -- correctly.
|
|
904
|
+
#
|
|
905
|
+
# They travel on the EVENT stream, because that is the stream the receipt
|
|
906
|
+
# reads. A fixture that put them on stderr would leave every redaction
|
|
907
|
+
# assertion below vacuously green.
|
|
908
|
+
python3 - "$HOME" 'livetoken00000000000000' 'urlsecretvalue' <<'PY_SENSITIVE'
|
|
909
|
+
import json, sys
|
|
910
|
+
# Both credential shapes are assembled here rather than written literally. A
|
|
911
|
+
# `sk-` token is refused by validate-skill.sh, and a credentialed URL trips the
|
|
912
|
+
# review gate's own egress tripwire on every later review of this repository --
|
|
913
|
+
# both scanners behaving correctly on a fixture that only looks real.
|
|
914
|
+
# The password carries a literal "@": a userinfo rule that stops at the first
|
|
915
|
+
# one leaves the tail of the password in the receipt.
|
|
916
|
+
# The host carries no dot: with one, the userinfo tail plus host reads as an
|
|
917
|
+
# email address, and a reviewer quoting the finding puts that shape into a
|
|
918
|
+
# receipt the public-sanitization gate then refuses.
|
|
919
|
+
credentialed_url = "https://proxyuser:p" + "@" + "ss" + sys.argv[3] + "@" + "localhost/path"
|
|
920
|
+
print(json.dumps({"type": "error", "message": (
|
|
921
|
+
"failed while reading " + sys.argv[1] + "/private/thing"
|
|
922
|
+
" sk-" + sys.argv[2] + " token=supersecretvalue"
|
|
923
|
+
" access_token=accesssecretvalue client_secret=clientsecretvalue"
|
|
924
|
+
' {"refresh_token": "refreshsecretvalue"}'
|
|
925
|
+
" session=sessionsecretvalue cookie=cookiesecretvalue auth=authsecretvalue"
|
|
926
|
+
" code=codesecretvalue bearer=bearersecretvalue sid=sidsecretvalue"
|
|
927
|
+
" " + credentialed_url +
|
|
928
|
+
' quoted="quotedsecretvalue more of it"'
|
|
929
|
+
)}))
|
|
930
|
+
PY_SENSITIVE
|
|
931
|
+
printf 'STDERRONLYMARKER5z should never reach the receipt\n' >&2
|
|
932
|
+
exit 1
|
|
933
|
+
fi
|
|
934
|
+
if [ "$behavior" = "silent_failure" ]; then
|
|
935
|
+
exit 1
|
|
936
|
+
fi
|
|
937
|
+
if [ "$behavior" = "stderr_only" ]; then
|
|
938
|
+
python3 -c 'print("startup noise line. " * 60)' >&2
|
|
939
|
+
printf 'the real failure is here STDERRTAILMARKER9x\n' >&2
|
|
940
|
+
exit 1
|
|
941
|
+
fi
|
|
942
|
+
if [ "$behavior" = "long_error_event" ]; then
|
|
943
|
+
python3 -c 'import json;print(json.dumps({"type":"error","message":"E"*4000}))'
|
|
944
|
+
exit 1
|
|
945
|
+
fi
|
|
876
946
|
case "$behavior" in
|
|
877
947
|
packet_read|packet_search|packet_tampered) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
|
|
878
948
|
pass|shell_disable_required|missing_shell_feature|removed_shell_feature|ignored_shell_disable|mcp_capability_missing|inherited_mcp|inherited_mcp_dot|inherited_mcp_space|inherited_mcp_quote|skills_budget_warning|skills_budget_warning_after_concern|hook_trust_warning|hook_trust_warning_started|hook_trust_warning_repeated_after_concern|unknown_error_valid_result) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
|
|
@@ -2186,6 +2256,98 @@ out="$(run_codex signal_exit)"; rc=$?
|
|
|
2186
2256
|
check "Codex process signals are terminal operator interrupts" \
|
|
2187
2257
|
'[ "$rc" = 2 ] && [ "$(field reason_code "$out")" = operator_interrupt ] && [ "$(field cascade_eligible "$out")" = False ]'
|
|
2188
2258
|
|
|
2259
|
+
# The CLI reports supply and credential failures as structured events on stdout,
|
|
2260
|
+
# not on stderr. Classifying from stderr alone turns a cascade-eligible quota
|
|
2261
|
+
# into a terminal unknown failure and stops the whole reviewer lane.
|
|
2262
|
+
out="$(run_codex usage_limit_event)"; rc=$?
|
|
2263
|
+
check "Codex reads a usage limit reported only on the event stream" \
|
|
2264
|
+
'[ "$rc" = 2 ] && [ "$(field reason_code "$out")" = quota ] && [ "$(field cascade_eligible "$out")" = True ]'
|
|
2265
|
+
|
|
2266
|
+
out="$(run_codex usage_limit_turn_failed)"; rc=$?
|
|
2267
|
+
check "Codex reads a rate limit carried by a failed turn" \
|
|
2268
|
+
'[ "$rc" = 2 ] && [ "$(field reason_code "$out")" = quota ] && [ "$(field cascade_eligible "$out")" = True ]'
|
|
2269
|
+
|
|
2270
|
+
out="$(run_codex auth_event)"; rc=$?
|
|
2271
|
+
check "Codex reads a credential rejection reported only on the event stream" \
|
|
2272
|
+
'[ "$rc" = 2 ] && [ "$(field reason_code "$out")" = provider_unavailable ] && [ "$(field cascade_eligible "$out")" = True ]'
|
|
2273
|
+
|
|
2274
|
+
# Only the transport's own top-level errors classify. Model-authored content
|
|
2275
|
+
# arrives nested under `item`, and the packet it quotes is untrusted.
|
|
2276
|
+
out="$(run_codex quota_in_model_output)"; rc=$?
|
|
2277
|
+
check "Codex refuses to classify from packet-derived model output" \
|
|
2278
|
+
'[ "$rc" = 2 ] && [ "$(field reason_code "$out")" = unknown_client_failure ] && [ "$(field cascade_eligible "$out")" = False ]'
|
|
2279
|
+
|
|
2280
|
+
|
|
2281
|
+
|
|
2282
|
+
|
|
2283
|
+
# The one invariant that replaces every redaction row: no input can influence
|
|
2284
|
+
# the receipt's text, because there is none derived from the run. The fixture
|
|
2285
|
+
# below carries home paths, several credential-assignment shapes, URL userinfo
|
|
2286
|
+
# with a separator in the password, and a stderr-only marker; none of it can
|
|
2287
|
+
# reach the receipt, and the check is equality with a constant rather than the
|
|
2288
|
+
# absence of a list of shapes.
|
|
2289
|
+
DIAG_CONSTANT="the transport output for this failure is in transport_run_dir"
|
|
2290
|
+
out="$(run_codex sensitive_streams)"; rc=$?
|
|
2291
|
+
diag="$(field transport_diagnostic "$out")"
|
|
2292
|
+
check "Codex records a receipt text no input can influence" \
|
|
2293
|
+
'[ "$rc" = 2 ] && [ "$diag" = "$DIAG_CONSTANT" ]'
|
|
2294
|
+
|
|
2295
|
+
out="$(run_codex silent_failure)"; rc=$?
|
|
2296
|
+
check "Codex records the same text when the transport says nothing" \
|
|
2297
|
+
'[ "$rc" = 2 ] && [ "$(field transport_diagnostic "$out")" = "$DIAG_CONSTANT" ]'
|
|
2298
|
+
|
|
2299
|
+
out="$(run_codex stderr_only)"; rc=$?
|
|
2300
|
+
diag="$(field transport_diagnostic "$out")"
|
|
2301
|
+
check "Codex does not quote stderr into the receipt" \
|
|
2302
|
+
'[ "$diag" = "$DIAG_CONSTANT" ]'
|
|
2303
|
+
|
|
2304
|
+
# The streams have to survive the failure. Deleting them with the run directory
|
|
2305
|
+
# is the defect this whole round started from.
|
|
2306
|
+
run_dir="$(field transport_run_dir "$out")"
|
|
2307
|
+
# The wrapper records a path under $HOME with $HOME replaced, so a committed
|
|
2308
|
+
# receipt carries no username. Expand it before testing the directory, or this
|
|
2309
|
+
# row false-REDs on any host whose TMPDIR sits under $HOME.
|
|
2310
|
+
case "$run_dir" in "~/"*) run_dir="$HOME/${run_dir#\~/}" ;; esac
|
|
2311
|
+
check "Codex preserves the run directory on a transport failure and names it" \
|
|
2312
|
+
'[ -n "$run_dir" ] && [ -d "$run_dir" ] && [ -s "$run_dir/stderr.log" ] && grep -q STDERRTAILMARKER9x "$run_dir/stderr.log"'
|
|
2313
|
+
|
|
2314
|
+
check "Codex keeps the preserved run directory private" \
|
|
2315
|
+
'[ "$(dir_mode "$run_dir")" = 700 ]'
|
|
2316
|
+
|
|
2317
|
+
# The eliding compares physical paths, so a home spelled with a trailing slash
|
|
2318
|
+
# -- or through a symlink -- still elides. Comparing the literal `$HOME` string
|
|
2319
|
+
# would put the username in a committed receipt on exactly those hosts.
|
|
2320
|
+
# Invoked directly rather than through run_codex: that helper pins TMPDIR, and
|
|
2321
|
+
# this row needs a TMPDIR that sits under the home it is testing.
|
|
2322
|
+
mkdir -p "$WORK/fakehome/tmp"
|
|
2323
|
+
out="$(STUB_BEHAVIOR=stderr_only REVIEW_WRAPPER_TEST_STATE="$WORK/state" \
|
|
2324
|
+
PATH="$WORK/bin:$PATH" TMPDIR="$WORK/fakehome/tmp" HOME="$WORK/fakehome/" \
|
|
2325
|
+
CODEX_HOME="$WORK/codex-source" TEST_DIFF_PATH="$WORK/diff.patch" \
|
|
2326
|
+
TEST_PROFILE_PATH="$WORK/review-profile.json" \
|
|
2327
|
+
bash "$DIR/codex_review.sh" --implementer-family claude \
|
|
2328
|
+
--diff-file "$WORK/diff.patch" --review-profile-file "$WORK/review-profile.json" \
|
|
2329
|
+
--mode review --timeout 30)"; rc=$?
|
|
2330
|
+
run_dir="$(field transport_run_dir "$out")"
|
|
2331
|
+
# HOME=/ is real in root and arbitrary-uid containers. Treated as a home
|
|
2332
|
+
# spelling it would rewrite every separator in the excerpt.
|
|
2333
|
+
out2="$(STUB_BEHAVIOR=sensitive_streams REVIEW_WRAPPER_TEST_STATE="$WORK/state" \
|
|
2334
|
+
PATH="$WORK/bin:$PATH" TMPDIR="$WORK/tmp" HOME=/ \
|
|
2335
|
+
CODEX_HOME="$WORK/codex-source" TEST_DIFF_PATH="$WORK/diff.patch" \
|
|
2336
|
+
TEST_PROFILE_PATH="$WORK/review-profile.json" \
|
|
2337
|
+
bash "$DIR/codex_review.sh" --implementer-family claude \
|
|
2338
|
+
--diff-file "$WORK/diff.patch" --review-profile-file "$WORK/review-profile.json" \
|
|
2339
|
+
--mode review --timeout 30)"
|
|
2340
|
+
dir2="$(field transport_run_dir "$out2")"
|
|
2341
|
+
check "Codex does not treat a separator-only home as a path to elide" \
|
|
2342
|
+
'case "$dir2" in "~"*) false ;; /*) [ -d "$dir2" ] ;; *) false ;; esac'
|
|
2343
|
+
|
|
2344
|
+
check "Codex elides a home path spelled with a trailing slash" \
|
|
2345
|
+
'case "$run_dir" in "~/"*) case "$run_dir" in *fakehome*) false ;; *) true ;; esac ;; *) false ;; esac'
|
|
2346
|
+
|
|
2347
|
+
out="$(run_codex pass)"; rc=$?
|
|
2348
|
+
check "Codex adds no diagnostic field to a successful review" \
|
|
2349
|
+
'[ "$rc" = 0 ] && json_lacks_key "$out" transport_diagnostic && json_lacks_key "$out" transport_run_dir'
|
|
2350
|
+
|
|
2189
2351
|
echo '----'
|
|
2190
2352
|
if [ "$fails" -eq 0 ]; then
|
|
2191
2353
|
echo cli_review_wrapper_tests_ok
|
|
@@ -20,6 +20,22 @@ from unittest import mock
|
|
|
20
20
|
|
|
21
21
|
|
|
22
22
|
SCRIPT_DIR = Path(__file__).resolve().parent
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def required_concerns(stage: str, *risk_tags: str) -> list[str]:
|
|
26
|
+
"""The concern ids a plan owes, asked of the controller that enforces them.
|
|
27
|
+
|
|
28
|
+
A fixture holding its own copy of this list silently stops satisfying the gate
|
|
29
|
+
when the set changes, and the suite runner aborts at its first failing target,
|
|
30
|
+
so the drift surfaces rounds later. There is one owner; ask it.
|
|
31
|
+
"""
|
|
32
|
+
argv = [sys.executable, str(SCRIPT_DIR / "review_gate.py"),
|
|
33
|
+
"--print-required-concerns", "--stage", stage]
|
|
34
|
+
for tag in risk_tags:
|
|
35
|
+
argv += ["--risk-tag", tag]
|
|
36
|
+
printed = subprocess.run(argv, capture_output=True, text=True, check=True).stdout.split()
|
|
37
|
+
assert printed, "the controller printed no required concerns"
|
|
38
|
+
return printed
|
|
23
39
|
if str(SCRIPT_DIR) not in sys.path:
|
|
24
40
|
sys.path.insert(0, str(SCRIPT_DIR))
|
|
25
41
|
import review_gate
|
|
@@ -736,7 +752,7 @@ class CompletionFindingDispositionTest(unittest.TestCase):
|
|
|
736
752
|
{"concern": concern,
|
|
737
753
|
"conclusion": f"The synthetic completion fixture preserves {concern} boundaries.",
|
|
738
754
|
"evidence_refs": ["fixture"]}
|
|
739
|
-
for concern in ("
|
|
755
|
+
for concern in required_concerns("build")
|
|
740
756
|
],
|
|
741
757
|
"evidence": [{"id": "fixture", "result": "Synthetic exact-candidate completion and history fixture."}],
|
|
742
758
|
})
|
|
@@ -47,7 +47,7 @@ case "$client" in
|
|
|
47
47
|
codex) family=openai; provider=openai; model=codex-local-default ;;
|
|
48
48
|
*) exit 2 ;;
|
|
49
49
|
esac
|
|
50
|
-
concern_results='[{"concern":"correctness","conclusion":"Checked routing correctness."},{"concern":"safety","conclusion":"Checked fail-closed safety."},{"concern":"failure_paths","conclusion":"Checked fallback failure paths."},{"concern":"tests_evidence","conclusion":"Checked deterministic test evidence."},{"concern":"compatibility","conclusion":"Checked client-order compatibility."}]'
|
|
50
|
+
concern_results='[{"concern":"correctness","conclusion":"Checked routing correctness."},{"concern":"safety","conclusion":"Checked fail-closed safety."},{"concern":"failure_paths","conclusion":"Checked fallback failure paths."},{"concern":"tests_evidence","conclusion":"Checked deterministic test evidence."},{"concern":"compatibility","conclusion":"Checked client-order compatibility."},{"concern":"claim_strength","conclusion":"Checked that no claim reaches past the client-order fixture."}]'
|
|
51
51
|
|
|
52
52
|
if [ "$client" = "claude" ]; then
|
|
53
53
|
case "$behavior" in
|
|
@@ -105,20 +105,35 @@ done
|
|
|
105
105
|
|
|
106
106
|
printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n' >"$WORK/diff.patch"
|
|
107
107
|
printf 'diff --git a/c b/c\n--- a/c\n+++ b/c\n@@ -1 +1 @@\n-x\n+aws_key = "AKIAIOSFODNN7EXAMPLE"\n' >"$WORK/secret-diff.patch"
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
108
|
+
# The plan's required concern set has ONE owner. A fixture keeping its own copy
|
|
109
|
+
# stops satisfying the gate the moment that set changes, and the suite runner
|
|
110
|
+
# aborts at its first failing target, so the drift surfaces rounds later -- five
|
|
111
|
+
# fixtures drifted that way at once. Ask the controller instead.
|
|
112
|
+
python3 - "$WORK/review-plan.json" "$DIR" <<'PY'
|
|
113
|
+
import json
|
|
114
|
+
import subprocess
|
|
115
|
+
import sys
|
|
116
|
+
from pathlib import Path
|
|
117
|
+
|
|
118
|
+
plan_path, script_dir = Path(sys.argv[1]), Path(sys.argv[2])
|
|
119
|
+
required = subprocess.run(
|
|
120
|
+
[sys.executable, str(script_dir / "review_gate.py"),
|
|
121
|
+
"--print-required-concerns", "--stage", "build"],
|
|
122
|
+
capture_output=True, text=True, check=True,
|
|
123
|
+
).stdout.split()
|
|
124
|
+
assert required, "the controller printed no required concerns"
|
|
125
|
+
plan_path.write_text(json.dumps({
|
|
126
|
+
"intent": "Preserve independent reviewer routing while adding staged review.",
|
|
127
|
+
"acceptance": ["Client ordering and model-family exclusion remain deterministic."],
|
|
128
|
+
"self_review": [
|
|
129
|
+
{"concern": concern,
|
|
130
|
+
"conclusion": f"The deterministic client-routing stubs cover {concern}.",
|
|
131
|
+
"evidence_refs": ["e1"]}
|
|
132
|
+
for concern in required
|
|
133
|
+
],
|
|
134
|
+
"evidence": [{"id": "e1", "result": "Deterministic client routing contract fixture."}],
|
|
135
|
+
}, indent=2) + "\n", encoding="utf-8")
|
|
136
|
+
PY
|
|
122
137
|
|
|
123
138
|
reset_case() {
|
|
124
139
|
rm -f "$WORK/state"/*
|