@ccoalm/ccl-skills 0.1.1 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/README.md +12 -0
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +93 -16
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +10 -0
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +249 -5
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate_abort_leak.sh +394 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +7 -0
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md +50 -1
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +1 -0
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +1 -1
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +59 -1
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +3 -1
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +34 -0
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +16 -4
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +391 -14
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +74 -0
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +74 -33
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh +11 -4
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +421 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_round_attribution.sh +576 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_source_refuted.sh +176 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_regression_runner_lanes.sh +101 -0
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_root_depth.sh +6 -2
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +1 -1
  24. package/dist/assets/release.json +66 -21
  25. package/dist/cli.js +23 -2
  26. package/dist/update-notice.d.ts +71 -0
  27. package/dist/update-notice.js +173 -0
  28. package/dist/version-check.d.ts +2 -0
  29. package/dist/version-check.js +7 -0
  30. package/package.json +1 -1
@@ -1,11 +1,13 @@
1
1
  #!/usr/bin/env bash
2
2
  # Aggregate deterministic shell regressions for check-ccl-skills wrappers.
3
3
  #
4
- # Default / local layer: --fast. CI has a changes-gated job that runs --full so
5
- # Makefile and CI both call this single entrypoint instead of copying test lists.
4
+ # Default / local layer: --fast. CI runs the two layers as parallel jobs
5
+ # (regression-fast --fast, regression-heavy --heavy-only); --full remains the
6
+ # local aggregate (`make test-check-ccl-regressions`). Makefile and CI both call
7
+ # this single entrypoint instead of copying test lists.
6
8
  #
7
9
  # Usage:
8
- # bash skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh [--fast|--full]
10
+ # bash skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh [--fast|--full|--heavy-only]
9
11
  #
10
12
  # --fast runs the quick/mid wrapper regressions:
11
13
  # - test_ai_coding_implementation_gates.sh
@@ -31,19 +33,24 @@ set -euo pipefail
31
33
 
32
34
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
33
35
  REPO_ROOT="$(git -C "$SCRIPT_DIR" rev-parse --show-toplevel 2>/dev/null || { cd "$SCRIPT_DIR/../../.." && pwd -P; })"
34
- SCRIPTS_DIR="$REPO_ROOT/skills/skill-extraction-workflow/scripts"
36
+ # REGRESSION_SCRIPTS_DIR redirects the whole runner (execution and audit) at a
37
+ # fixture tree, so the lane-semantics regression can prove mode dispatch with
38
+ # stub suites instead of the real multi-minute ones.
39
+ SCRIPTS_DIR="${REGRESSION_SCRIPTS_DIR:-$REPO_ROOT/skills/skill-extraction-workflow/scripts}"
35
40
 
36
41
  usage() {
37
42
  cat <<'EOF'
38
- Usage: test_check_ccl_regressions.sh [--fast|--full]
43
+ Usage: test_check_ccl_regressions.sh [--fast|--full|--heavy-only]
39
44
 
40
45
  Runs deterministic shell regressions for check-ccl-skills wrapper behavior.
41
- Default is --fast to keep local `make test` light. GitLab CI runs --full from a
42
- changes-gated job for public deterministic regressions.
46
+ Default is --fast to keep local `make test` light. CI runs --fast and
47
+ --heavy-only as parallel jobs; --full is the local aggregate.
43
48
 
44
49
  Modes:
45
50
  --fast the quick/mid wrapper regressions (exact set = the fast_tests array below / the list above)
46
51
  --full fast layer plus R0 status and source-register lifecycle regressions
52
+ --heavy-only only the heavy full-checker regressions (the heavy_tests array);
53
+ gives CI a heavy execution surface that does not repeat the fast layer
47
54
  --list-unregistered print sibling test_*.sh NOT in fast_tests/heavy_tests (one per
48
55
  line) and exit 0; used by the registration guard test so a new test cannot
49
56
  silently skip CI. REGRESSION_SCRIPTS_DIR overrides the scanned directory.
@@ -54,6 +61,7 @@ mode="fast"
54
61
  case "${1:-}" in
55
62
  ""|--fast) mode="fast" ;;
56
63
  --full) mode="full" ;;
64
+ --heavy-only) mode="heavy-only" ;;
57
65
  --list-unregistered) mode="list-unregistered" ;;
58
66
  -h|--help) usage; exit 0 ;;
59
67
  *) usage >&2; exit 2 ;;
@@ -64,20 +72,25 @@ if [ "$#" -gt 1 ]; then
64
72
  exit 2
65
73
  fi
66
74
 
67
- run_test() {
68
- local test_name="$1"
69
- local test_path="$SCRIPTS_DIR/$test_name"
70
- local started_at ended_at rc
71
- [ -f "$test_path" ] || { echo "FAIL: missing regression test: $test_path" >&2; exit 1; }
72
- started_at="$(date +%s)"
73
- if bash "$test_path"; then
74
- rc=0
75
- else
76
- rc=$?
77
- fi
78
- ended_at="$(date +%s)"
79
- echo "regression_test_timing: test=$test_name seconds=$((ended_at - started_at)) status=$rc"
80
- return "$rc"
75
+ # Lanes execute through the shared bounded-concurrency runner: these suites are
76
+ # dominated by waiting (process timeouts, stub sleeps) and by independent
77
+ # subprocesses, so serial execution burned wall-clock the runner host was not
78
+ # using. The runner owns the missing-file precondition, the per-suite timing
79
+ # line, in-order output replay, and failure propagation; assertions inside the
80
+ # suites are untouched. `SUITE_JOBS=1` restores serial execution for debugging a
81
+ # suspected concurrency interaction. See specs/037-ci-intra-job-parallel/plan.md
82
+ # for the per-lane shared-state audit that makes this safe.
83
+ PARALLEL_RUNNER="$REPO_ROOT/scripts/run-parallel-suites.sh"
84
+
85
+ run_lane() {
86
+ local label="$1"; shift
87
+ local paths=() entry
88
+ for entry in "$@"; do paths+=("$SCRIPTS_DIR/$entry"); done
89
+ [ -f "$PARALLEL_RUNNER" ] || {
90
+ echo "FAIL: missing parallel suite runner: $PARALLEL_RUNNER" >&2
91
+ exit 1
92
+ }
93
+ bash "$PARALLEL_RUNNER" --label "$label" "${paths[@]}"
81
94
  }
82
95
 
83
96
  fast_tests=(
@@ -98,6 +111,13 @@ fast_tests=(
98
111
  # differential stays in the heavy suite, but a require-date regression must
99
112
  # go RED in every `make test`, not only under --full.
100
113
  test_impact_chain_gate_dateless_host.sh
114
+ # Row-ownership attribution probes: a control leg plus the refusals the gate
115
+ # owes on a surviving row that vouches for an unchanged owner, a row citing a
116
+ # package path other than SKILL.md, a row resolving to two selected owners,
117
+ # and the two shapes that must keep their prior silent skip (an unrelated
118
+ # five-column register table, a non-curated owner). Synthetic repos only, no
119
+ # clone, so it belongs in the lane every run exercises.
120
+ test_impact_chain_round_attribution.sh
101
121
  test_eval_routing_bank_grader_diagnostics.sh
102
122
  test_eval_routing_bank_surface_binding.sh
103
123
  test_eval_routing_prose_target.sh
@@ -106,6 +126,10 @@ fast_tests=(
106
126
  test_validate_skill_cross_refs.sh
107
127
  test_git_identity_predicate_gate.sh
108
128
  test_regression_runner_registration.sh
129
+ # Lane-semantics guard for this runner itself (036 challenge P2): proves
130
+ # --heavy-only / --fast / --full each run exactly their lane against a stub
131
+ # fixture via REGRESSION_SCRIPTS_DIR, and that a red heavy stub propagates.
132
+ test_regression_runner_lanes.sh
109
133
  test_routing_pointer_integrity.sh
110
134
  test_routing_bank_integrity.sh
111
135
  test_governing_chain_diff.sh
@@ -129,6 +153,17 @@ heavy_tests=(
129
153
  # here costs no enforcement — it only stops charging every pre-commit run for a
130
154
  # repo clone, and stops a slow entry competing for the runner host.
131
155
  test_register_firing_path_wiring.sh
156
+ # No-verdict-regression differential: materializes twelve pinned integration
157
+ # points as detached worktrees and runs both the baseline and candidate gate
158
+ # against each. Proves the other half of a gate change — that it did not start
159
+ # refusing what it used to accept — which is the half that blocks every author
160
+ # when it goes wrong. Worktree-per-point makes it far too slow for pre-commit.
161
+ # source-refuted 证据类的滥用面:16 条合成用例,每条一条分支,且整仓 clone 一次
162
+ # 作 fixture 底座。它测的正是「这个类会不会变成通用豁免」——七轮独立评审各击穿过
163
+ # 一版门槛,所以负向用例(真行为变更披标签、抵消式新增、路径穿越、子串锚、改权限、
164
+ # 对照表不交代被删内容)必须每轮都跑。clone + 16 分支对 pre-commit 太慢,进 heavy。
165
+ test_impact_chain_source_refuted.sh
166
+ test_impact_chain_gate_verdict_differential.sh
132
167
  )
133
168
 
134
169
  # Registration self-audit: every sibling test_*.sh must appear in fast_tests or
@@ -155,19 +190,25 @@ if [ "$mode" = "list-unregistered" ]; then
155
190
  exit 0
156
191
  fi
157
192
 
158
- for test_name in "${fast_tests[@]}"; do
159
- run_test "$test_name"
160
- done
193
+ # Unregistered siblings hard-fail every execution mode (036 challenge P1): the
194
+ # old advisory-only tail made enforcement depend on the registration guard test
195
+ # itself staying registered — removing it from fast_tests silenced the only red
196
+ # path. Failing here keeps enforcement independent of any one array entry.
197
+ if [ -n "$unregistered" ]; then
198
+ echo "FAIL: test_*.sh not registered in fast_tests/heavy_tests (CI would silently skip them):$unregistered" >&2
199
+ exit 1
200
+ fi
161
201
 
162
- if [ "$mode" = "full" ]; then
163
- for test_name in "${heavy_tests[@]}"; do
164
- run_test "$test_name"
165
- done
166
- echo "test_check_ccl_regressions_full_ok"
167
- else
168
- echo "test_check_ccl_regressions_fast_ok"
202
+ if [ "$mode" != "heavy-only" ]; then
203
+ run_lane regression_fast_lane "${fast_tests[@]}"
169
204
  fi
170
205
 
171
- if [ -n "$unregistered" ]; then
172
- echo "regression_runner_unregistered_tests_advisory: not in fast_tests/heavy_tests, CI will NOT run them:$unregistered" >&2
206
+ if [ "$mode" = "full" ] || [ "$mode" = "heavy-only" ]; then
207
+ run_lane regression_heavy_lane "${heavy_tests[@]}"
173
208
  fi
209
+
210
+ case "$mode" in
211
+ full) echo "test_check_ccl_regressions_full_ok" ;;
212
+ heavy-only) echo "test_check_ccl_regressions_heavy_only_ok" ;;
213
+ *) echo "test_check_ccl_regressions_fast_ok" ;;
214
+ esac
@@ -205,7 +205,10 @@ expect_block() {
205
205
  printf '%s\n' "$CASE_OUTPUT" >&2
206
206
  exit 1
207
207
  }
208
- printf '%s\n' "$CASE_OUTPUT" | grep -q "$token" || {
208
+ # SIGPIPE-safe match; see the note at the first such comparison below.
209
+ # `=~`, not a glob: callers pass REGEX tokens (c1 uses `.*`), so a literal
210
+ # substring test would reject a correct gate message.
211
+ [[ "$CASE_OUTPUT" =~ $token ]] || {
209
212
  echo "FAIL[$label]: gate blocked without naming $token" >&2
210
213
  printf '%s\n' "$CASE_OUTPUT" >&2
211
214
  exit 1
@@ -258,7 +261,11 @@ run_case
258
261
  printf '%s\n' "$CASE_OUTPUT" >&2
259
262
  exit 1
260
263
  }
261
- printf '%s\n' "$CASE_OUTPUT" | grep -q 'skill_catalog_contract_ok' || {
264
+ # Pure-bash substring match, not `printf | grep -q`: grep -q exits on its first
265
+ # hit, the producer takes SIGPIPE, and under `set -o pipefail` the SUCCESS path
266
+ # then reports failure. Timing-dependent, so it stayed latent until this lane
267
+ # ran concurrently (specs/037-ci-intra-job-parallel/plan.md).
268
+ [[ "$CASE_OUTPUT" == *skill_catalog_contract_ok* ]] || {
262
269
  echo "FAIL[c6]: pristine tree passed without the contract marker" >&2
263
270
  printf '%s\n' "$CASE_OUTPUT" >&2
264
271
  exit 1
@@ -313,12 +320,12 @@ run_case
313
320
  new_case
314
321
  rm -f "$CASE_DIR/agent-context/session-start.md"
315
322
  run_case
316
- printf '%s\n' "$CASE_OUTPUT" | grep -q 'skill_catalog_entry_coverage_unevaluated' || {
323
+ [[ "$CASE_OUTPUT" == *skill_catalog_entry_coverage_unevaluated* ]] || {
317
324
  echo "FAIL[c9]: missing routing layer was not reported as unevaluated" >&2
318
325
  printf '%s\n' "$CASE_OUTPUT" >&2
319
326
  exit 1
320
327
  }
321
- printf '%s\n' "$CASE_OUTPUT" | grep -q 'skill_catalog_contract_ok' && {
328
+ [[ "$CASE_OUTPUT" == *skill_catalog_contract_ok* ]] && {
322
329
  echo "FAIL[c9]: contract_ok claimed while entry coverage was never evaluated" >&2
323
330
  printf '%s\n' "$CASE_OUTPUT" >&2
324
331
  exit 1
@@ -0,0 +1,421 @@
1
+ #!/usr/bin/env bash
2
+ # No-verdict-regression differential for the impact-chain gate.
3
+ #
4
+ # The probe suite proves the gate refuses what it should. This proves it did not
5
+ # start refusing anything it used to accept — the other half, and the half that
6
+ # blocks every author when it goes wrong. A shared merge gate cannot ship a
7
+ # tightening on an asserted "I checked": the check has to be re-runnable by
8
+ # whoever reviews or reverts it.
9
+ #
10
+ # Method: for each pinned integration point M on the integration branch, the
11
+ # gate's real input was (base = M's first parent, HEAD = M) — that is the merge
12
+ # result CI evaluates. Replay exactly that for the BASELINE gate blob and for the
13
+ # working-tree gate, and compare exit verdicts. A range that the baseline accepted
14
+ # and the candidate refuses is a regression and fails this test.
15
+ #
16
+ # Refs are pinned by SHA, not by branch name, so this keeps testing the same
17
+ # history after the branch moves. Missing refs FAIL rather than skip: a
18
+ # differential that silently tests nothing is worse than no differential.
19
+ #
20
+ # TWO INPUT SETS, because the pins alone can only prove one direction. Every
21
+ # pinned integration point is baseline-GREEN — asserted per point, not assumed,
22
+ # since a baseline that failed for an EXECUTION reason (missing runtime,
23
+ # incompatible checkout, moved path) is otherwise indistinguishable from a
24
+ # candidate refusing for a POLICY reason: both nonzero, no mismatch recorded, a
25
+ # green run certifying nothing. But an all-green input set makes the
26
+ # newly-accepted arm unreachable, so a loosened candidate would clear all twelve
27
+ # and the run would still claim "no change in either direction".
28
+ #
29
+ # The synthetic replay cases at the bottom carry the other half: one is
30
+ # baseline-RED by construction (an owner changed with no ledger row), so a
31
+ # candidate that stopped refusing it is caught there and nowhere else. Each case
32
+ # declares the verdict the baseline must produce, and a refusal must also carry
33
+ # its expected diagnostic token — that token is what separates a policy refusal
34
+ # from a broken fixture, which would otherwise be banked as evidence.
35
+ #
36
+ # Oracle: an always-refuse candidate fails on the pins; an always-accept
37
+ # candidate fails on the baseline-red case. Both arms are exercised.
38
+ set -euo pipefail
39
+
40
+ SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
41
+ CANDIDATE_GATE="$SCRIPT_DIR/impact-chain-gate.rb"
42
+ [ -f "$CANDIDATE_GATE" ] || { echo "FAIL: gate not found: $CANDIDATE_GATE" >&2; exit 1; }
43
+
44
+ REPO_ROOT="$(git -C "$SCRIPT_DIR" rev-parse --show-toplevel)" || exit 1
45
+
46
+ # The gate as it stood before the row-ownership change, pinned by the commit it
47
+ # last shipped in. Update this together with the integration points below only
48
+ # when re-baselining deliberately, and say so in the commit that does it.
49
+ BASELINE_GATE_COMMIT="5146b3b315c487f93ba352584a0b27af9414676b"
50
+ GATE_PATH="skills/skill-extraction-workflow/scripts/impact-chain-gate.rb"
51
+
52
+ # Integration points, newest first: EVERY merge commit reachable from
53
+ # BASELINE_GATE_COMMIT — the whole integration history this gate has judged, not
54
+ # a window into it. The list is asserted against git's own answer below, so a pin
55
+ # quietly dropped or left stale after a history rewrite fails the test instead of
56
+ # shrinking its coverage silently.
57
+ INTEGRATION_POINTS="
58
+ 5146b3b315c487f93ba352584a0b27af9414676b
59
+ b9de138694624df13cba05ec9bdb4c99c3cc8ef5
60
+ 17db43bcb4083423ba071de26071041d3b3c1c67
61
+ 1b21d2efda0e64da47ab468d338a671be483bc0b
62
+ 73a62ca2da1f41bfe7bb9fb728315bdc0b9d0f27
63
+ f7b11522e7e46fce2dee5ed2d6fc2b73b04eb6ac
64
+ e645ea975e922bdb0a7036f853a2ecfdd19bb184
65
+ d767d2ff6ed80ff91b05a5dcd1fe946eea0e785c
66
+ af3f5e7d1410f62b7eaafe88bdff915af5d41831
67
+ 2b11657630e1a3c4cb8650b4d1f5f8282bcf0a16
68
+ ff31d0ca0bd79405ac2dded54647c3a743511425
69
+ 4b7afd22a1001769ce970de4dfb41be28da578fc
70
+ d4c9b092655e284e511855c4306ed7a9d1b0aa10
71
+ f03b1140fc4c2b304411eb67d7771479c82874bf
72
+ d2d6337e26493e762a3af3db624894e300008081
73
+ a04f9a0bea97a1c53b3cab95ce7337cccec2583c
74
+ 76e8b8ccbd3178fc2341b674c8486356e63081d9
75
+ c5d2d35e2e7c9b6725c2b533dd480bda3a9bacbf
76
+ 1b0c6fb0683d86fb9c6458a61beb32516c63f3cc
77
+ 408e11104c2cf5e47dcfe170243a260ffb5e8165
78
+ 93d09c563ff7bfe7ccc6c20b8b0d2f9c1758c031
79
+ 4516e30952cc429e2d0e5dbb7971a4e64796feae
80
+ 1ab0af0966044b1a1875a8737a448e394566db15
81
+ 6437abee61f9c41f3f890d75165aee701dfbbc65
82
+ bd9d01b3b909f85529b5ef87539c6e73a7fa0be9
83
+ e8c128d178ae2df94454859081ab0cad5ef52206
84
+ b05b74d6796a511f88208673872990aa20027675
85
+ 61f5b2e1937c87ed4eeb763c1adf883fdbc97d79
86
+ 98735e027f10dcda57a6649534b9f6af3a451a71
87
+ 989cf5f9a0e4f33c4cf8ccb839b01da78b94af18
88
+ c84beaf47805c4187714dbbf9010b04c08b0a585
89
+ 138ce664f1e40ccd563b3be896652b8335988003
90
+ 06163ecb1a2448deef50ff91527849b22a89fbfd
91
+ 9f233728a3b9deac9ba9a8c12d1d4fd3693bf2e6
92
+ 8cea35e6dba34a8af6eba3746d14b89efacc147a
93
+ 31d3b3f1ba130622fd888c2a6e35ba1b6eaafd1e
94
+ 34d504bec2dc5ec9935a9896fff71872359042fa
95
+ 759dc6030e9ae6ba8dca06d77a4afdb93231b12a
96
+ 2205292a900ad609f92d2d89d4ed9fbdc5f9456d
97
+ 673fece3cf3d893ed6707119f60cd656f3ae6331
98
+ 95f06b2e6057c4f070ab072f42ef959042a44036
99
+ 0fd26e7aeed9832307b7a6724855f31269c71b5f
100
+ 19a04b6a3e4c3025185690470fd501ba82f126d8
101
+ 8b0f44be666b894af319cecf823fca64ba448631
102
+ fb551574b0d8cfac7022c406daa847717f46cc46
103
+ 37faff720a4786eaeeda328605a046cc1d66189d
104
+ c0561c74e0f7249b8041f7c5800a8d6deadf496f
105
+ 7f114e56d9893f78834dc4f3bca2e2cfc6dfb320
106
+ d249ab85d17705e4a30935759a15ff613a38a05c
107
+ d63ac44bd37b9375309515c8ff62bc91d3f78072
108
+ 2aa8cd3d92efe18cb7a18990488481664f6b1286
109
+ 48e9eb627be2ee9f484f7b11a966ce8d4b146958
110
+ 1ecc8558da485eb8784030d629b0bb21547c07cb
111
+ 2a0ec222a847fb1a71396d0dcbd439004dfc1c0c
112
+ 6a795af5690a30bcef598e37886dd0cf06d9dc15
113
+ 87d42560005030dca3bef7a7aa0ef21e340059ae
114
+ b332b01059eb7322a32d3ef186fd8bbb25e0037e
115
+ a63246578f53d849e0a5cb41cb65763a5cdf4fcc
116
+ fad480296eb0334291d804cdc3f1f7f0928d802d
117
+ 90ec533e172acdda8c6b42a20b572488bfe29b59
118
+ ba0a1cc44d28f17ae1aafed390518655eaf7598c
119
+ 046612652f4613ae8f569a1f3287afcbd7509de6
120
+ 4c0904bf7c7d342bac0deafced16a8ba157d40d5
121
+ 15a89e95b7fc6ba0112fdb777545b55cc683f90b
122
+ "
123
+
124
+ TMP="$(mktemp -d "${TMPDIR:-/tmp}/icverdict.XXXXXX")"
125
+ cleanup() {
126
+ rc=$?
127
+ git -C "$REPO_ROOT" worktree remove --force "$TMP/probe" >/dev/null 2>&1 || true
128
+ rm -rf "$TMP"
129
+ exit $rc
130
+ }
131
+ trap cleanup EXIT
132
+ trap 'exit 130' INT
133
+ trap 'exit 143' TERM
134
+
135
+ BASELINE_GATE="$TMP/impact-chain-gate-baseline.rb"
136
+ git -C "$REPO_ROOT" show "$BASELINE_GATE_COMMIT:$GATE_PATH" > "$BASELINE_GATE" 2>/dev/null || {
137
+ echo "FAIL: baseline gate blob is unreachable at $BASELINE_GATE_COMMIT:$GATE_PATH" >&2
138
+ echo " fetch the integration branch's history, or re-baseline deliberately" >&2
139
+ exit 1
140
+ }
141
+
142
+ # EXPECTED DIVERGENCES. Four landings back-filled a ledger row by corrective
143
+ # rewrite — the repair for a round that merged with this gate red — so the owner
144
+ # change sits below their base and the row cites an owner the range does not
145
+ # touch. The candidate refuses that shape by design and deliberately offers no
146
+ # author-declared escape, so these four historical ranges diverge. The gate is
147
+ # diff-scoped and never re-judges landed history, so nothing operational depends
148
+ # on them; this differential is the only thing that replays them.
149
+ #
150
+ # Each exemption is constrained to ONE direction and ONE diagnostic. A blanket
151
+ # "any mismatch at this SHA is fine" would also swallow the opposite direction —
152
+ # a loosening — which is the failure this whole suite exists to catch. Entries are
153
+ # named individually, never matched by pattern, and an entry that stops diverging
154
+ # is reported as stale rather than tolerated.
155
+ EXPECTED_DIVERGENCE_SHAS="f03b1140f 93d09c563 9f233728a 046612652"
156
+ EXPECTED_DIVERGENCE_DIRECTION="newly refused"
157
+ EXPECTED_DIVERGENCE_TOKEN="impact_chain_row_vouches_for_unchanged_owner"
158
+ expected_divergence() { # <full sha> <direction> <candidate output>
159
+ local short="${1:0:9}" direction="$2" out="$3" known
160
+ [ "$direction" = "$EXPECTED_DIVERGENCE_DIRECTION" ] || return 1
161
+ case "$out" in *"$EXPECTED_DIVERGENCE_TOKEN"*) : ;; *) return 1 ;; esac
162
+ for known in $EXPECTED_DIVERGENCE_SHAS; do
163
+ [ "$known" = "$short" ] && return 0
164
+ done
165
+ return 1
166
+ }
167
+
168
+ # A no-op differential would pass vacuously and certify nothing. With expected
169
+ # divergences configured it is worse than vacuous: identical gates cannot produce
170
+ # them, so the exemptions are stale and the run would pass while silently failing
171
+ # the staleness check it never reaches.
172
+ if cmp -s "$BASELINE_GATE" "$CANDIDATE_GATE"; then
173
+ if [ -n "$(printf '%s' "$EXPECTED_DIVERGENCE_SHAS" | tr -d '[:space:]')" ]; then
174
+ echo "FAIL: baseline and candidate are byte-identical, yet expected divergences are configured" >&2
175
+ echo " identical gates cannot diverge — the exemptions are stale and must be removed" >&2
176
+ exit 1
177
+ fi
178
+ echo "impact_chain_gate_verdict_differential: baseline and candidate are byte-identical — nothing to compare"
179
+ exit 0
180
+ fi
181
+
182
+ # Completeness assertion: the pinned list must be exactly every merge commit
183
+ # reachable from the baseline, which is the whole integration history this gate
184
+ # has ever judged. Without it the list is a hand sample that can lose points
185
+ # silently while still reporting a clean run.
186
+ expected_pins="$(git -C "$REPO_ROOT" rev-list --merges "$BASELINE_GATE_COMMIT" | tr -d ' ')"
187
+ actual_pins="$(printf '%s\n' $INTEGRATION_POINTS)"
188
+ if [ "$expected_pins" != "$actual_pins" ]; then
189
+ echo "FAIL: the pinned set is not every merge commit reachable from $BASELINE_GATE_COMMIT" >&2
190
+ echo " re-baselining is a deliberate edit: regenerate the list and say so in the commit" >&2
191
+ diff <(printf '%s\n' "$expected_pins") <(printf '%s\n' "$actual_pins") | head -20 | sed 's/^/ | /' >&2
192
+ exit 1
193
+ fi
194
+
195
+ regressions=0
196
+ expected_seen=0
197
+ compared=0
198
+ for point in $INTEGRATION_POINTS; do
199
+ git -C "$REPO_ROOT" rev-parse -q --verify "$point^{commit}" >/dev/null || {
200
+ echo "FAIL: pinned integration point is not present: $point" >&2
201
+ exit 1
202
+ }
203
+ parent="$(git -C "$REPO_ROOT" rev-parse "$point^1")"
204
+ subject="$(git -C "$REPO_ROOT" log --format=%s -n 1 "$point" | cut -c1-46)"
205
+ git -C "$REPO_ROOT" worktree remove --force "$TMP/probe" >/dev/null 2>&1 || true
206
+ git -C "$REPO_ROOT" worktree add --detach -q "$TMP/probe" "$point" >/dev/null 2>&1 || {
207
+ echo "FAIL: could not materialize $point" >&2
208
+ exit 1
209
+ }
210
+ set +e
211
+ baseline_out="$(env -u ALIAS_AUDIT_CMD CCL_SKILL_BASE_REF="$parent" ruby "$BASELINE_GATE" "$TMP/probe" 2>&1)"
212
+ baseline_rc=$?
213
+ set -e
214
+ # A nonzero baseline is only usable as a verdict when it is a POLICY refusal.
215
+ # An execution failure — missing runtime, incompatible checkout, moved path —
216
+ # also exits nonzero, and treating it as "the baseline refused" would let it
217
+ # mask a real candidate regression on the same point. The gate names every
218
+ # refusal it makes, so requiring one of those names is what separates the two.
219
+ if [ "$baseline_rc" != 0 ]; then
220
+ case "$baseline_out" in
221
+ *impact_chain_*) : ;;
222
+ *)
223
+ echo "FAIL: baseline exited $baseline_rc on $point without naming a gate refusal" >&2
224
+ echo " that is an execution failure, not a verdict; fix the environment or re-baseline" >&2
225
+ printf '%s\n' "$baseline_out" | head -10 | sed 's/^/ | /' >&2
226
+ exit 1
227
+ ;;
228
+ esac
229
+ fi
230
+ set +e
231
+ candidate_out="$(env -u ALIAS_AUDIT_CMD CCL_SKILL_BASE_REF="$parent" ruby "$CANDIDATE_GATE" "$TMP/probe" 2>&1)"
232
+ candidate_rc=$?
233
+ set -e
234
+ compared=$((compared + 1))
235
+ # The same rule on the candidate side. Comparing exit codes alone counts
236
+ # "both nonzero" as agreement, so a candidate that CRASHES on a point the
237
+ # baseline legitimately refuses reads as unchanged — the one place a real
238
+ # regression could hide behind a matching number.
239
+ if [ "$candidate_rc" != 0 ]; then
240
+ case "$candidate_out" in
241
+ *impact_chain_*) : ;;
242
+ *)
243
+ echo "FAIL: candidate exited $candidate_rc on $point without naming a gate refusal" >&2
244
+ echo " that is an execution failure, not a verdict" >&2
245
+ printf '%s\n' "$candidate_out" | head -10 | sed 's/^/ | /' >&2
246
+ exit 1
247
+ ;;
248
+ esac
249
+ fi
250
+ # EITHER direction is a mismatch. Reporting only the tighten-side drift would
251
+ # let a loosening pass the very evidence a tightening-only change offers: a
252
+ # point the baseline refused and the candidate now accepts is a silently
253
+ # widened gate, which on a shared merge gate is the worse of the two failures.
254
+ # An intended verdict change re-baselines the pins deliberately and says so;
255
+ # it does not slip through as an informational line.
256
+ flag=""
257
+ direction=""
258
+ if [ "$baseline_rc" = 0 ] && [ "$candidate_rc" != 0 ]; then
259
+ direction="newly refused"
260
+ elif [ "$baseline_rc" != 0 ] && [ "$candidate_rc" = 0 ]; then
261
+ direction="newly accepted"
262
+ fi
263
+ if [ -n "$direction" ]; then
264
+ if expected_divergence "$point" "$direction" "$candidate_out"; then
265
+ flag=" (expected divergence: corrective-rewrite back-fill, $direction)"
266
+ expected_seen=$((expected_seen + 1))
267
+ else
268
+ flag=" <== VERDICT MISMATCH: $direction"
269
+ regressions=$((regressions + 1))
270
+ fi
271
+ fi
272
+ # Only mismatches are printed: 64 agreeing lines is noise that hides the one
273
+ # line that matters.
274
+ if [ -n "$flag" ]; then
275
+ printf '%-10s %-10s rc=%-4s rc=%-4s %s%s\n' \
276
+ "${point:0:9}" "${parent:0:9}" "$baseline_rc" "$candidate_rc" "$subject" "$flag"
277
+ fi
278
+ if [ -n "$flag" ] && [ "$candidate_rc" != 0 ]; then
279
+ printf '%s\n' "$candidate_out" | head -8 | sed 's/^/ | /'
280
+ fi
281
+ done
282
+
283
+
284
+
285
+ # ---------------------------------------------------------------------------
286
+ # BASELINE-RED REPLAY CASES.
287
+ #
288
+ # Every pinned integration point is baseline-green, and the assertion above
289
+ # exits on any that is not. That makes the newly-accepted arm unreachable over
290
+ # the pins alone: an always-accept candidate would clear all twelve and the run
291
+ # would still report "no change in either direction" — a claim wider than the
292
+ # evidence. The refusals the replaced gate made are the other half of its
293
+ # behavior and need their own cases.
294
+ #
295
+ # Each case declares the verdict the BASELINE must produce and, for a refusal,
296
+ # the diagnostic token that refusal must carry. Asserting the token is what
297
+ # keeps a policy refusal distinguishable from an execution failure: a ruby
298
+ # crash, a missing path, or a broken fixture also exits nonzero, and counting
299
+ # that as "the baseline refused" would bank a broken setup as evidence.
300
+ # ---------------------------------------------------------------------------
301
+ LEDGER_REL="skills/skill-extraction-workflow/references/source-register.md"
302
+
303
+ seed_case() { # <dir>
304
+ local repo="$1"
305
+ mkdir -p "$repo/skills/product-rd-workflow" "$repo/skills/skill-extraction-workflow/references"
306
+ git -C "$repo" init -q -b trunk
307
+ git -C "$repo" config user.email test@example.invalid
308
+ git -C "$repo" config user.name "Test User"
309
+ git -C "$repo" config commit.gpgsign false
310
+ cat > "$repo/skills/product-rd-workflow/SKILL.md" <<'MD'
311
+ ---
312
+ name: product-rd-workflow
313
+ description: Fixture owner for the verdict differential's baseline-red cases.
314
+ ---
315
+
316
+ # Fixture Owner
317
+
318
+ ## Rules
319
+
320
+ - Baseline rule line present before the case commit.
321
+ MD
322
+ printf '| Lesson | Downstream owner | Behavior | Status | Evidence |\n| --- | --- | --- | --- | --- |\n' \
323
+ > "$repo/$LEDGER_REL"
324
+ git -C "$repo" add -A
325
+ git -C "$repo" commit -qm "seed"
326
+ git -C "$repo" branch fixture-base HEAD
327
+ git -C "$repo" switch -q -c case fixture-base
328
+ printf -- '- Rule DIFFERENTIAL-CASE must be recorded before the round lands.\n' \
329
+ >> "$repo/skills/product-rd-workflow/SKILL.md"
330
+ }
331
+
332
+ case_failures=0
333
+ run_case() { # <name> <expected-baseline-rc> <expected-token-or-empty> <declare:yes|no>
334
+ local name="$1" want_rc="$2" want_token="$3" declare_row="$4"
335
+ local repo="$TMP/case-$name"
336
+ seed_case "$repo"
337
+ if [ "$declare_row" = yes ]; then
338
+ printf '| Fixture differential row | `downstream-executor` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#DIFFERENTIAL-CASE | `updated` | `product-rd-workflow/SKILL.md` fixture change |\n' \
339
+ >> "$repo/$LEDGER_REL"
340
+ fi
341
+ git -C "$repo" add -A
342
+ git -C "$repo" commit -qm "case commit"
343
+
344
+ set +e
345
+ local b_out c_out b_rc c_rc
346
+ b_out="$(env -u ALIAS_AUDIT_CMD CCL_SKILL_BASE_REF=fixture-base ruby "$BASELINE_GATE" "$repo" 2>&1)"
347
+ b_rc=$?
348
+ c_out="$(env -u ALIAS_AUDIT_CMD CCL_SKILL_BASE_REF=fixture-base ruby "$CANDIDATE_GATE" "$repo" 2>&1)"
349
+ c_rc=$?
350
+ set -e
351
+
352
+ # The baseline must produce the verdict this case exists to pin. Anything else
353
+ # is a broken fixture or environment, and continuing would compare against it.
354
+ if [ "$b_rc" != "$want_rc" ]; then
355
+ echo "FAIL: case '$name' expected baseline rc=$want_rc, got rc=$b_rc" >&2
356
+ printf '%s\n' "$b_out" | head -6 | sed 's/^/ | /' >&2
357
+ exit 1
358
+ fi
359
+ if [ -n "$want_token" ]; then
360
+ case "$b_out" in
361
+ *"$want_token"*) : ;;
362
+ *)
363
+ echo "FAIL: case '$name' baseline refused, but not for '$want_token' — treat as setup failure, not policy" >&2
364
+ printf '%s\n' "$b_out" | head -6 | sed 's/^/ | /' >&2
365
+ exit 1
366
+ ;;
367
+ esac
368
+ fi
369
+
370
+ # Same rule on the candidate as on the baseline: a nonzero exit only counts as
371
+ # a verdict when it names a refusal. Comparing exit codes alone lets a candidate
372
+ # that CRASHES on the baseline-red case match rc=1 and read as agreement — the
373
+ # one place this case could certify nothing while looking green.
374
+ if [ "$c_rc" != 0 ]; then
375
+ case "$c_out" in
376
+ *impact_chain_*) : ;;
377
+ *)
378
+ echo "FAIL: case '$name' candidate exited $c_rc without naming a gate refusal" >&2
379
+ echo " that is an execution failure, not a verdict" >&2
380
+ printf '%s\n' "$c_out" | head -6 | sed 's/^/ | /' >&2
381
+ exit 1
382
+ ;;
383
+ esac
384
+ fi
385
+ local flag=""
386
+ if [ "$b_rc" = 0 ] && [ "$c_rc" != 0 ]; then
387
+ flag=" <== VERDICT MISMATCH: newly refused"; case_failures=$((case_failures + 1))
388
+ elif [ "$b_rc" != 0 ] && [ "$c_rc" = 0 ]; then
389
+ flag=" <== VERDICT MISMATCH: newly accepted"; case_failures=$((case_failures + 1))
390
+ fi
391
+ printf '%-28s rc=%-4s rc=%-4s %s%s\n' "$name" "$b_rc" "$c_rc" "(expected baseline rc=$want_rc)" "$flag"
392
+ if [ -n "$flag" ]; then
393
+ printf '%s\n' "$c_out" | head -6 | sed 's/^/ | /'
394
+ fi
395
+ }
396
+
397
+ echo
398
+ printf '%-28s %-7s %-7s %s\n' CASE BASELINE CANDIDATE EXPECTATION
399
+ # Baseline-RED: an owner changed with no ledger row. This is the case that makes
400
+ # the newly-accepted arm reachable — a candidate that stopped refusing it would
401
+ # be caught here and nowhere else.
402
+ run_case "undeclared-owner-change" 1 "impact_chain_gate_missing" no
403
+ # Baseline-GREEN control on the same fixture shape, so a case that fails does so
404
+ # because of the missing declaration and not because the shape itself is refused.
405
+ run_case "declared-owner-change" 0 "" yes
406
+
407
+ echo
408
+ total_failures=$((regressions + case_failures))
409
+ if [ "$total_failures" -gt 0 ]; then
410
+ echo "impact_chain_gate_verdict_differential: $regressions/$compared integration points and $case_failures replay cases changed verdict unexpectedly" >&2
411
+ exit 1
412
+ fi
413
+ # An expected divergence that stops diverging means the exemption is stale and
414
+ # should be removed, so it is reported rather than silently tolerated.
415
+ expected_total="$(for e in $EXPECTED_DIVERGENCE_SHAS; do echo "$e"; done | wc -l | tr -d ' ')"
416
+ if [ "$expected_seen" != "$expected_total" ]; then
417
+ echo "FAIL: $expected_seen of $expected_total expected divergences actually diverged" >&2
418
+ echo " an exemption that no longer fires is stale — remove it" >&2
419
+ exit 1
420
+ fi
421
+ echo "impact_chain_gate_verdict_differential: ok ($compared integration points, $expected_seen named expected divergences, 2 replay cases incl. one baseline-red; no unexpected verdict change in either direction)"