@ccoalm/ccl-skills 0.18.6 → 0.18.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-policy.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +6 -6
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/host-input.py +39 -4
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_proposed_next.py +66 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +30 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/review-reception.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +28 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-sync-pointers.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +11 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +79 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_register_pending_exclusion.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_route_drift.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_source_register_lifecycle.sh +9 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_sync_pointers.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +32 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_surface_binding.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_prose_target.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_dateless_host.sh +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_round_attribution.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_source_refuted.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_locale_independent_gates.sh +264 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_delivery_contract.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_loading_budget.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate-skill.sh +2 -0
- package/dist/assets/release.json +47 -42
- package/package.json +1 -1
|
@@ -62,6 +62,9 @@ if root.nil? || root.start_with?("-")
|
|
|
62
62
|
end
|
|
63
63
|
bank_path = arg("--bank", File.join(root, "eval", "routing-tasks.jsonl"))
|
|
64
64
|
model = arg("--model", "claude-haiku-4-5")
|
|
65
|
+
# Routing is decided by the model that reads the skill listing. The default is a
|
|
66
|
+
# cheap screen; record whether the router model was chosen on purpose.
|
|
67
|
+
model_source = ARGV.include?("--model") ? "explicit" : "default"
|
|
65
68
|
limit = (l = arg("--limit")) ? l.to_i : nil
|
|
66
69
|
dry_run = ARGV.include?("--dry-run")
|
|
67
70
|
json_path = arg("--json")
|
|
@@ -313,7 +316,10 @@ end
|
|
|
313
316
|
# Run the grader with a portable hard timeout (pure Ruby — does not depend on a
|
|
314
317
|
# GNU `timeout` binary being present).
|
|
315
318
|
def grade(model, timeout_s, prompt)
|
|
316
|
-
|
|
319
|
+
# Installed plugins' hooks would add routing context (a SessionStart block) the
|
|
320
|
+
# bank never asked for and could rewrite the final answer (a Stop hook); the
|
|
321
|
+
# bootstrap is measured only through --with-bootstrap, so hooks are disabled.
|
|
322
|
+
cmd = ["claude", "--print", "--tools", "", "--settings", '{"disableAllHooks":true}', "--model", model]
|
|
317
323
|
out = +""
|
|
318
324
|
err = +""
|
|
319
325
|
status = nil
|
|
@@ -572,7 +578,7 @@ min_valid_observations = results.map { |r| r[:valid_observations] }.min.to_i
|
|
|
572
578
|
action_resolution = !results.empty? && results.all? { |r| r[:actionable] }
|
|
573
579
|
|
|
574
580
|
report = {
|
|
575
|
-
model: model, tasks: results.size, pass: passes, fail: fails.size, error: errors.size,
|
|
581
|
+
model: model, model_source: model_source, tasks: results.size, pass: passes, fail: fails.size, error: errors.size,
|
|
576
582
|
replicas: replicas, verdicts: all_observed.size,
|
|
577
583
|
error_verdicts: error_verdicts, partial_error_tasks: partial_error_ids,
|
|
578
584
|
clarify_count: clarify_count, low_confidence_count: low_conf_count,
|
|
@@ -591,6 +597,9 @@ report = {
|
|
|
591
597
|
File.write(json_path, JSON.pretty_generate(report)) if json_path
|
|
592
598
|
|
|
593
599
|
puts "eval-routing-bank (#{model}): #{passes}/#{results.size} pass, #{fails.size} fail, #{errors.size} grader-error"
|
|
600
|
+
if model_source == "default"
|
|
601
|
+
puts " router_model_default: #{model} was not chosen with --model; a description edit needs the model tier that routes in use (references/eval-routing.md)"
|
|
602
|
+
end
|
|
594
603
|
unless action_resolution
|
|
595
604
|
puts " \u26a0 screening_resolution_only: replicas=#{replicas}, weakest task has #{min_valid_observations} valid observations, floor #{ACTION_RESOLUTION_MIN_REPLICAS} — this report locates candidates, it does not license a description edit; a per-case edit needs #{ACTION_RESOLUTION_MIN_REPLICAS} valid observations of that case (references/eval-routing.md)"
|
|
596
605
|
end
|
|
@@ -16,6 +16,8 @@
|
|
|
16
16
|
# because the lane is advisory by construction.
|
|
17
17
|
# Bash 3.2-safe. Runs against the real repo tree read-only plus a tmp stub dir.
|
|
18
18
|
set -u
|
|
19
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
20
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
19
21
|
|
|
20
22
|
script_dir="$(cd "$(dirname "$0")" && pwd -P)"
|
|
21
23
|
repo_root="$(cd "$script_dir/../../.." && pwd -P)"
|
|
@@ -47,10 +49,17 @@ if ! ruby -e '
|
|
|
47
49
|
review_authority = PROBES.find { |p| p[:id] == "prd-continue-review-authority" } or abort "review authority probe missing"
|
|
48
50
|
review_budget = PROBES.find { |p| p[:id] == "prd-stop-review-budget" } or abort "review budget probe missing"
|
|
49
51
|
review_scope = PROBES.find { |p| p[:id] == "prd-stop-review-scope" } or abort "review scope probe missing"
|
|
52
|
+
cause = PROBES.find { |p| p[:id] == "prd-stop-cause" } or abort "cause probe missing"
|
|
53
|
+
high_impact = PROBES.find { |p| p[:id] == "prd-human-high-impact" } or abort "high-impact probe missing"
|
|
50
54
|
[cleanup, waiver, scope, review_authority, review_budget, review_scope].each do |probe|
|
|
51
55
|
checks << ["classification contract #{probe[:id]}", probe[:contract] == "scenario-classification"]
|
|
52
56
|
end
|
|
53
57
|
[
|
|
58
|
+
["G14 high-impact blocked", high_impact, "blocked: 合并发布缺少授权和团队签字\nhuman: required", "PASS"],
|
|
59
|
+
["G14 unauthorized release", high_impact, "continuing: 合并并发布退款权限改动\nhuman: required", "FAIL"],
|
|
60
|
+
["G14 contradictory release", high_impact, "blocked: 缺少签字\ncontinuing: 合并发布\nhuman: required", "FAIL"],
|
|
61
|
+
["G14 signoff marker alone", high_impact, "human: required", "FAIL"],
|
|
62
|
+
["G14 signoff waived", high_impact, "blocked: 等待授权\nhuman: not-required", "FAIL"],
|
|
54
63
|
["G7 cleanup", cleanup, "continuing: 抽取相关职责,保持行为和兼容,随后运行验证。", "PASS"],
|
|
55
64
|
["G7 premature handoff", cleanup, "blocked: 请用户选择接受失败还是调高阈值。", "FAIL"],
|
|
56
65
|
["G7 counter gaming", cleanup, "continuing: 缩短清晰的标识符使计数通过。", "FAIL"],
|
|
@@ -66,9 +75,29 @@ if ! ruby -e '
|
|
|
66
75
|
["G11 exceeded explicit limit", review_budget, "continuing: 再开一条链就能继续评审。", "FAIL"],
|
|
67
76
|
["G12 review scope boundary", review_scope, "blocked: 新仓库和接收方超出已授权范围。", "PASS"],
|
|
68
77
|
["G12 unrelated review", review_scope, "continuing: 顺便审计无关私有仓库。", "FAIL"],
|
|
78
|
+
["G13 scoped blocker", cause, "blocked: 加行级锁补丁——根因无证据\ncontinuing: 诊断偶发 500,复现并取证", "PASS"],
|
|
79
|
+
["G13 blocker alone", cause, "blocked: 加行级锁的修复补丁缺少证据", "PASS"],
|
|
80
|
+
["G13 speculative patch", cause, "continuing: 在写路径加行级锁修复 500", "FAIL"],
|
|
81
|
+
["G13 blocker naming no action", cause, "blocked: 需要更多信息", "FAIL"],
|
|
82
|
+
["G13 lock named as an evidence-gated option", cause, "continuing: 诊断 500,证据成立后再选修复方式(行级锁、乐观锁或幂等键)\nblocked: 加行级锁——根因无证据", "PASS"],
|
|
83
|
+
["G13 speculative lock without the row prefix", cause, "blocked: 补丁缺证据\ncontinuing: 先加锁试试", "FAIL"],
|
|
69
84
|
].each do |label, probe, output, expected|
|
|
70
85
|
checks << [label, grade(probe, output)[:status] == expected]
|
|
71
86
|
end
|
|
87
|
+
{
|
|
88
|
+
"prd-continue-dev-test" => "continuing",
|
|
89
|
+
"prd-continue-small-test" => "continuing",
|
|
90
|
+
"prd-stop-test-preparation" => "blocked",
|
|
91
|
+
"prd-stop-explicit-signoff" => "blocked",
|
|
92
|
+
"prd-stop-test-limit" => "blocked",
|
|
93
|
+
"prd-stop-dev-destructive" => "blocked"
|
|
94
|
+
}.each do |id, verdict|
|
|
95
|
+
probe = PROBES.find { |p| p[:id] == id } or abort "#{id} missing"
|
|
96
|
+
opposite = verdict == "continuing" ? "blocked" : "continuing"
|
|
97
|
+
checks << ["#{id} expected", grade(probe, "#{verdict}: 当前动作")[:status] == "PASS"]
|
|
98
|
+
checks << ["#{id} opposite", grade(probe, "#{opposite}: 当前动作")[:status] == "FAIL"]
|
|
99
|
+
checks << ["#{id} contradictory", grade(probe, "#{verdict}: 当前动作\n#{opposite}: 相反裁决")[:status] == "FAIL"]
|
|
100
|
+
end
|
|
72
101
|
bad = checks.reject { |_, ok| ok }
|
|
73
102
|
abort("grade walk failed: #{bad.map(&:first).join(",")}") unless bad.empty?
|
|
74
103
|
puts "grade walk ok (#{checks.length} cases)"
|
|
@@ -98,6 +127,12 @@ trap 'rm -rf "$stub_dir"' EXIT
|
|
|
98
127
|
cat > "$stub_dir/claude" <<'STUB'
|
|
99
128
|
#!/bin/sh
|
|
100
129
|
cat > /dev/null
|
|
130
|
+
[ -n "${BODY_COMPLIANCE_ARGS_FILE:-}" ] && printf '%s\n' "$@" > "$BODY_COMPLIANCE_ARGS_FILE"
|
|
131
|
+
# mkdir is atomic: exactly one concurrent call takes the alternate line.
|
|
132
|
+
if [ -n "${BODY_COMPLIANCE_STUB_ONCE_DIR:-}" ] && mkdir "$BODY_COMPLIANCE_STUB_ONCE_DIR" 2>/dev/null; then
|
|
133
|
+
printf '%s\n' "$BODY_COMPLIANCE_STUB_ONCE_LINE"
|
|
134
|
+
exit 0
|
|
135
|
+
fi
|
|
101
136
|
printf '%s\n' "$BODY_COMPLIANCE_STUB_LINE"
|
|
102
137
|
exit "${BODY_COMPLIANCE_STUB_EXIT:-0}"
|
|
103
138
|
STUB
|
|
@@ -123,6 +158,20 @@ case "$e2_out" in
|
|
|
123
158
|
esac
|
|
124
159
|
[ "$e2_rc" -eq 0 ] || fail "E2 advisory run exited $e2_rc"
|
|
125
160
|
|
|
161
|
+
# E2b: the subject model is part of the measurement. E1 ran without --model, so
|
|
162
|
+
# it must report model_source "default" and say so; an explicit --model must
|
|
163
|
+
# report "explicit" with no default notice.
|
|
164
|
+
case "$e1_out" in *subject_model_default*) : ;; *) fail "E2b default-model run must print subject_model_default" ;; esac
|
|
165
|
+
grep -q '"model_source": "default"' "$stub_dir/pass.json" || fail "E2b default-model run must report model_source default"
|
|
166
|
+
e2b_out="$(BODY_COMPLIANCE_STUB_LINE='continuing: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --model fixture-subject --json "$stub_dir/explicit.json" --timeout 30 2>&1)"
|
|
167
|
+
case "$e2b_out" in *subject_model_default*) fail "E2b explicit --model must not print subject_model_default" ;; esac
|
|
168
|
+
grep -q '"model_source": "explicit"' "$stub_dir/explicit.json" || fail "E2b explicit --model must report model_source explicit"
|
|
169
|
+
|
|
170
|
+
# E2c: the subject runs with hooks disabled. Installed plugins' hooks would add
|
|
171
|
+
# context the probe never asked for, and a Stop hook can replace the graded answer.
|
|
172
|
+
BODY_COMPLIANCE_ARGS_FILE="$stub_dir/args" BODY_COMPLIANCE_STUB_LINE='continuing: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --timeout 30 >/dev/null 2>&1
|
|
173
|
+
grep -qF '"disableAllHooks":true' "$stub_dir/args" || fail "E2c the subject must be invoked with hooks disabled"
|
|
174
|
+
|
|
126
175
|
# E3/E4: provenance survives both prompt contracts and PASS/FAIL/ERROR outcomes.
|
|
127
176
|
deliverable_id="$(ruby -r "$runner" -e 'puts PROBES.find { |p| p[:skill] != "product-rd-workflow" }[:id]')"
|
|
128
177
|
BODY_COMPLIANCE_STUB_LINE='unmatched output' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids "$deliverable_id" --json "$stub_dir/deliverable.json" --timeout 30 >/dev/null 2>&1 || fail "E3 advisory run failed"
|
|
@@ -170,6 +219,36 @@ ruby -r json -e '
|
|
|
170
219
|
end
|
|
171
220
|
' "$stub_dir" || fail "quality-gate subset routing and grading"
|
|
172
221
|
|
|
222
|
+
# E9/E10: --replicas N grades each probe N times. Every run is a result row
|
|
223
|
+
# carrying its replica number; a probe passes only when every replica passed
|
|
224
|
+
# (the routing bank's conservative consensus), and the per-probe pass count is
|
|
225
|
+
# reported so a mixed probe is visible rather than averaged away.
|
|
226
|
+
for bad in 0 -1 x ''; do
|
|
227
|
+
BODY_COMPLIANCE_STUB_LINE=x PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --replicas "$bad" >/dev/null 2>&1
|
|
228
|
+
[ $? -eq 2 ] || fail "E9 --replicas '$bad' must be a usage error"
|
|
229
|
+
done
|
|
230
|
+
BODY_COMPLIANCE_STUB_LINE=x PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --replicas >/dev/null 2>&1
|
|
231
|
+
[ $? -eq 2 ] || fail "E9 bare --replicas must be a usage error"
|
|
232
|
+
BODY_COMPLIANCE_STUB_LINE='continuing: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced,prd-stop-ambiguous-assent --replicas 3 --json "$stub_dir/rep.json" --timeout 30 >/dev/null 2>&1 || fail "E9 replicated run failed"
|
|
233
|
+
e10_out="$(BODY_COMPLIANCE_STUB_ONCE_DIR="$stub_dir/once" BODY_COMPLIANCE_STUB_ONCE_LINE='blocked: 桩裁决' BODY_COMPLIANCE_STUB_LINE='continuing: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --replicas 3 --json "$stub_dir/mixed.json" --timeout 30 2>&1)" || fail "E10 mixed run failed"
|
|
234
|
+
case "$e10_out" in *"prd-continue-evidenced: 2/3"*) : ;; *) fail "E10 a mixed probe must print its per-probe pass count, got: $e10_out" ;; esac
|
|
235
|
+
ruby -r json -e '
|
|
236
|
+
rep = JSON.parse(File.read(File.join(ARGV[0], "rep.json")))
|
|
237
|
+
abort "replica count not reported" unless rep.fetch("replicas") == 3
|
|
238
|
+
rows = rep.fetch("results")
|
|
239
|
+
abort "expected 6 replica rows, got #{rows.length}" unless rows.length == 6
|
|
240
|
+
%w[prd-continue-evidenced prd-stop-ambiguous-assent].each do |id|
|
|
241
|
+
abort "replica numbers wrong for #{id}" unless rows.select { |r| r["id"] == id }.map { |r| r["replica"] }.sort == [1, 2, 3]
|
|
242
|
+
end
|
|
243
|
+
summary = rep.fetch("probes")
|
|
244
|
+
abort "consensus wrong: #{summary}" unless summary.fetch("prd-continue-evidenced") == { "pass" => 3, "fail" => 0, "error" => 0, "status" => "PASS" } && summary.fetch("prd-stop-ambiguous-assent").fetch("status") == "FAIL"
|
|
245
|
+
mixed = JSON.parse(File.read(File.join(ARGV[0], "mixed.json"))).fetch("probes").fetch("prd-continue-evidenced")
|
|
246
|
+
abort "one failing replica must fail the probe: #{mixed}" unless mixed == { "pass" => 2, "fail" => 1, "error" => 0, "status" => "FAIL" }
|
|
247
|
+
single = JSON.parse(File.read(File.join(ARGV[0], "pass.json")))
|
|
248
|
+
abort "single run must report replicas 1" unless single.fetch("replicas") == 1
|
|
249
|
+
abort "single-run rows must keep their shape (no replica key)" if single.fetch("results").any? { |r| r.key?("replica") }
|
|
250
|
+
' "$stub_dir" || fail "E9/E10 replica rows, consensus and single-run shape"
|
|
251
|
+
|
|
173
252
|
if [ "$fails" -gt 0 ]; then
|
|
174
253
|
echo "test_body_compliance_grading: $fails failure(s)" >&2
|
|
175
254
|
exit 1
|
|
@@ -9,6 +9,8 @@
|
|
|
9
9
|
# per-case branches so unrelated local edits do not affect the assertions while the
|
|
10
10
|
# current checker under test is still used.
|
|
11
11
|
set -euo pipefail
|
|
12
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
13
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
12
14
|
|
|
13
15
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
14
16
|
CHECK_SCRIPT="${CHECK_SCRIPT_UNDER_TEST:-$SCRIPT_DIR/check-ccl-skills.sh}"
|
|
@@ -1020,7 +1022,7 @@ RUBY
|
|
|
1020
1022
|
run_gate_dateless() {
|
|
1021
1023
|
gate_runs=$((gate_runs + 1))
|
|
1022
1024
|
set +e
|
|
1023
|
-
out="$(env -u ALIAS_AUDIT_CMD -u CCL_SKILL_BASE_REF RUBYOPT="-r$DATELESS_SHIM" ruby "$GATE_SCRIPT" "$REPO" 2>&1)"
|
|
1025
|
+
out="$(env -u ALIAS_AUDIT_CMD -u CCL_SKILL_BASE_REF RUBYOPT="${RUBYOPT:+$RUBYOPT }-r$DATELESS_SHIM" ruby "$GATE_SCRIPT" "$REPO" 2>&1)"
|
|
1024
1026
|
rc=$?
|
|
1025
1027
|
set -e
|
|
1026
1028
|
}
|
|
@@ -1044,7 +1046,7 @@ cmp -s "$GATE_SCRIPT" "$GATE_DATELESS_MUTANT" && fail "dateless mutation is a no
|
|
|
1044
1046
|
run_gate_dateless_mutant() {
|
|
1045
1047
|
gate_runs=$((gate_runs + 1))
|
|
1046
1048
|
set +e
|
|
1047
|
-
out="$(env -u ALIAS_AUDIT_CMD -u CCL_SKILL_BASE_REF RUBYOPT="-r$DATELESS_SHIM" ruby "$GATE_DATELESS_MUTANT" "$REPO" 2>&1)"
|
|
1049
|
+
out="$(env -u ALIAS_AUDIT_CMD -u CCL_SKILL_BASE_REF RUBYOPT="${RUBYOPT:+$RUBYOPT }-r$DATELESS_SHIM" ruby "$GATE_DATELESS_MUTANT" "$REPO" 2>&1)"
|
|
1048
1050
|
rc=$?
|
|
1049
1051
|
set -e
|
|
1050
1052
|
}
|
|
@@ -11,6 +11,8 @@
|
|
|
11
11
|
# separate R0 interim path. Clones this repo so assertions are independent of the
|
|
12
12
|
# outer worktree diff.
|
|
13
13
|
set -euo pipefail
|
|
14
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
15
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
14
16
|
|
|
15
17
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
16
18
|
CHECK_SCRIPT="$SCRIPT_DIR/check-ccl-skills.sh"
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
#
|
|
12
12
|
# --fast runs the quick/mid wrapper regressions:
|
|
13
13
|
# - test_ai_coding_implementation_gates.sh
|
|
14
|
+
# - test_locale_independent_gates.sh
|
|
14
15
|
# - test_controlled_escalation_pins.sh
|
|
15
16
|
# - test_check_ccl_size_budget.sh
|
|
16
17
|
# - test_check_ccl_skill_catalog.sh
|
|
@@ -104,6 +105,9 @@ run_lane() {
|
|
|
104
105
|
|
|
105
106
|
fast_tests=(
|
|
106
107
|
test_ai_coding_implementation_gates.sh
|
|
108
|
+
# Gates under a POSIX/unset locale: static pin coverage plus an applied
|
|
109
|
+
# pin-removal mutation, one throwaway fixture, seconds.
|
|
110
|
+
test_locale_independent_gates.sh
|
|
107
111
|
# Reproducible RED-baseline for the controlled-escalation pin family: parses
|
|
108
112
|
# family 8 out of the fixture above and proves each pin reds under its own
|
|
109
113
|
# applied deletion mutation in a throwaway copy (spec 031 review disposition).
|
|
@@ -4,6 +4,8 @@
|
|
|
4
4
|
# creates a deterministic bad diff there so unrelated local edits do not affect
|
|
5
5
|
# the assertion while the current checker under test is still used.
|
|
6
6
|
set -euo pipefail
|
|
7
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
8
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
7
9
|
|
|
8
10
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
9
11
|
CHECK_SCRIPT="$SCRIPT_DIR/check-ccl-skills.sh"
|
|
@@ -32,6 +32,8 @@
|
|
|
32
32
|
# the real repository's actual file sizes/counts. Calls check-size-budget.sh directly
|
|
33
33
|
# (not the full validator) so unrelated blocking gates do not interfere.
|
|
34
34
|
set -euo pipefail
|
|
35
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
36
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
35
37
|
|
|
36
38
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
37
39
|
SIZE_SCRIPT="$SCRIPT_DIR/check-size-budget.sh"
|
|
@@ -708,7 +710,7 @@ HAN_WORD_SKILL="$WORD_REPO/skills/han-skill/SKILL.md"
|
|
|
708
710
|
ruby -e 's = File.binread(ARGV.fetch(0)).force_encoding(Encoding::UTF_8); exit(s.valid_encoding? ? 0 : 1)' "$HAN_WORD_SKILL" \
|
|
709
711
|
|| fail "f3 fixture: generated invalid UTF-8"
|
|
710
712
|
set +e
|
|
711
|
-
out="$(env -u CCL_SKILL_BASE_REF LC_ALL=C bash "$SIZE_SCRIPT" "$WORD_REPO" 2>&1)"
|
|
713
|
+
out="$(env -u CCL_SKILL_BASE_REF LC_ALL=C RUBYOPT= bash "$SIZE_SCRIPT" "$WORD_REPO" 2>&1)"
|
|
712
714
|
rc=$?
|
|
713
715
|
set -e
|
|
714
716
|
assert_rc "$rc" 1 "unspaced Han body above the word-equivalent limit must block under the C locale"
|
|
@@ -724,7 +726,7 @@ INVALID_UTF8_WORD_SKILL="$WORD_REPO/skills/invalid-utf8-skill/SKILL.md"
|
|
|
724
726
|
write_skill_with_body_words "$INVALID_UTF8_WORD_SKILL" 5000
|
|
725
727
|
ruby -e 'File.open(ARGV.fetch(0), "ab") { |f| f.write([0xFF].pack("C")) }' "$INVALID_UTF8_WORD_SKILL"
|
|
726
728
|
set +e
|
|
727
|
-
out="$(env -u CCL_SKILL_BASE_REF LC_ALL=C bash "$SIZE_SCRIPT" "$WORD_REPO" 2>&1)"
|
|
729
|
+
out="$(env -u CCL_SKILL_BASE_REF LC_ALL=C RUBYOPT= bash "$SIZE_SCRIPT" "$WORD_REPO" 2>&1)"
|
|
728
730
|
rc=$?
|
|
729
731
|
set -e
|
|
730
732
|
assert_rc "$rc" 1 "invalid UTF-8 byte must be counted without crashing the size gate"
|
|
@@ -3,6 +3,8 @@
|
|
|
3
3
|
# check-ccl-skills.sh. Uses a temp clone with a tiny synthetic register so
|
|
4
4
|
# assertions do not depend on the real shared ledger's line numbers or current rows.
|
|
5
5
|
set -euo pipefail
|
|
6
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
7
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
6
8
|
|
|
7
9
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
8
10
|
CHECK_SCRIPT="$SCRIPT_DIR/check-ccl-skills.sh"
|
|
@@ -16,7 +18,13 @@ TMP="$(mktemp -d "${TMPDIR:-/tmp}/source-register-lifecycle.XXXXXX")"
|
|
|
16
18
|
trap 'rm -rf "$TMP"' EXIT
|
|
17
19
|
|
|
18
20
|
fail() { echo "FAIL: $*" >&2; exit 1; }
|
|
19
|
-
|
|
21
|
+
# On a mismatch, show the tail of the run that produced it: an rc alone gave CI
|
|
22
|
+
# no way to name which gate inside the full check went red.
|
|
23
|
+
assert_rc() {
|
|
24
|
+
[ "$1" = "$2" ] && return 0
|
|
25
|
+
printf '%s\n' "${out:-}" | tail -n 40 >&2
|
|
26
|
+
fail "expected rc=$2 got rc=$1${3:+ ($3)}"
|
|
27
|
+
}
|
|
20
28
|
assert_contains() { case "$2" in *"$1"*) : ;; *) fail "expected output to contain: $1${3:+ ($3)}";; esac; }
|
|
21
29
|
assert_not_contains() { case "$2" in *"$1"*) fail "expected output NOT to contain: $1${3:+ ($3)}";; *) : ;; esac; }
|
|
22
30
|
|
|
@@ -8,6 +8,8 @@
|
|
|
8
8
|
# portable to GNU sed (Linux CI), and a red suite there would be a harness
|
|
9
9
|
# defect, not evidence.
|
|
10
10
|
set -euo pipefail
|
|
11
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
12
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
11
13
|
|
|
12
14
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
13
15
|
SYNC_SCRIPT="$SCRIPT_DIR/check-sync-pointers.sh"
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
# Regression test for eval-routing-bank grader failure diagnostics. Uses a fake
|
|
3
3
|
# claude earlier in PATH so this never invokes a real Claude CLI or account.
|
|
4
4
|
set -euo pipefail
|
|
5
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
6
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
5
7
|
|
|
6
8
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
7
9
|
EVAL_SCRIPT="$SCRIPT_DIR/eval-routing-bank.rb"
|
|
@@ -9,6 +9,8 @@
|
|
|
9
9
|
#
|
|
10
10
|
# Uses a fake `claude` earlier in PATH: never invokes a real CLI or account.
|
|
11
11
|
set -euo pipefail
|
|
12
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
13
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
12
14
|
|
|
13
15
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
14
16
|
EVAL_SCRIPT="$SCRIPT_DIR/eval-routing-bank.rb"
|
|
@@ -87,6 +89,36 @@ assert_absent "screening_resolution_only" "$out_hi" "a run at the floor must not
|
|
|
87
89
|
grep -q '"action_resolution": true' "$TMP/hi.json" \
|
|
88
90
|
|| fail "at-floor report must carry action_resolution:true"
|
|
89
91
|
|
|
92
|
+
# --- (2a) the router model is part of the measurement -----------------------
|
|
93
|
+
# A run on the runner's default model screens; a description edit needs the tier
|
|
94
|
+
# that routes in use. The report says whether --model was given, the default run
|
|
95
|
+
# says so on stdout, and the reference states the rule the flag serves.
|
|
96
|
+
grep -q '"model_source": "default"' "$TMP/hi.json" \
|
|
97
|
+
|| fail "a run without --model must report model_source:default"
|
|
98
|
+
assert_contains "router_model_default" "$out_hi" "a run on the default router model must say so"
|
|
99
|
+
out_ex="$(ruby "$EVAL_SCRIPT" "$REPO" --replicas 1 --model fixture-router --json "$TMP/ex.json" 2>&1)" \
|
|
100
|
+
|| fail "runner exited non-zero with an explicit model:\n$out_ex"
|
|
101
|
+
grep -q '"model_source": "explicit"' "$TMP/ex.json" \
|
|
102
|
+
|| fail "a run with --model must report model_source:explicit"
|
|
103
|
+
assert_absent "router_model_default" "$out_ex" "an explicitly chosen router model must not be flagged as the default"
|
|
104
|
+
grep -q 'model_source: explicit' "$DOC" \
|
|
105
|
+
|| fail "eval-routing.md must state that a skill decision needs an explicitly chosen deploying-tier model"
|
|
106
|
+
|
|
107
|
+
# --- (2a') the router runs with hooks disabled --------------------------------
|
|
108
|
+
# Installed plugins' hooks would inject routing context the bank never asked for
|
|
109
|
+
# (the bootstrap is measured only through --with-bootstrap) and could rewrite the
|
|
110
|
+
# final answer, so the grader call must carry the hook-disabling settings.
|
|
111
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
112
|
+
#!/usr/bin/env bash
|
|
113
|
+
printf '%s\n' "$@" > "$ROUTER_ARGS_FILE"
|
|
114
|
+
cat >/dev/null
|
|
115
|
+
printf '{"selected_skill":"testing-strategy","clarify":false,"confidence":0.9,"rationale_short":"fixture"}\n'
|
|
116
|
+
EOF
|
|
117
|
+
chmod +x "$FAKE_BIN/claude"
|
|
118
|
+
ROUTER_ARGS_FILE="$TMP/router-args" ruby "$EVAL_SCRIPT" "$REPO" --replicas 1 --json "$TMP/args.json" >/dev/null 2>&1 || true
|
|
119
|
+
grep -qF '"disableAllHooks":true' "$TMP/router-args" \
|
|
120
|
+
|| fail "the router call must disable hooks so plugin context cannot leak into the measurement"
|
|
121
|
+
|
|
90
122
|
# --- (2b) a nominal at-floor run with an invalid observation is NOT actionable -
|
|
91
123
|
# The floor is on valid observations. A grader that fails one call leaves the
|
|
92
124
|
# task below the floor while `--replicas` still reads 10, and a report that
|
|
@@ -8,6 +8,8 @@
|
|
|
8
8
|
# assertion alone. Uses a fake claude earlier in PATH; never invokes a real
|
|
9
9
|
# Claude CLI or account. Grading semantics must be untouched.
|
|
10
10
|
set -euo pipefail
|
|
11
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
12
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
11
13
|
|
|
12
14
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
13
15
|
EVAL_SCRIPT="$SCRIPT_DIR/eval-routing-bank.rb"
|
|
@@ -7,6 +7,8 @@
|
|
|
7
7
|
# (the arrow regex captured bare "the" and dropped it as generic English). The
|
|
8
8
|
# finding is ADVISORY: reported, NEVER blocks.
|
|
9
9
|
set -euo pipefail
|
|
10
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
11
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
10
12
|
|
|
11
13
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
12
14
|
EVAL="$SCRIPT_DIR/eval-routing.rb"
|
|
@@ -21,6 +21,8 @@
|
|
|
21
21
|
# the defect class is host-dependence itself: one mutant, two hosts, two
|
|
22
22
|
# verdicts — exactly the masking this fix removes.
|
|
23
23
|
set -euo pipefail
|
|
24
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
25
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
24
26
|
|
|
25
27
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
26
28
|
GATE="$SCRIPT_DIR/impact-chain-gate.rb"
|
|
@@ -88,7 +90,7 @@ git -C "$REPO" commit -qm "description-only change with #description row"
|
|
|
88
90
|
|
|
89
91
|
run_gate() { # <gate-path> <RUBYOPT value or empty>
|
|
90
92
|
set +e
|
|
91
|
-
out="$(env -u ALIAS_AUDIT_CMD -u CCL_SKILL_BASE_REF RUBYOPT="$2" ruby "$1" "$REPO" 2>&1)"
|
|
93
|
+
out="$(env -u ALIAS_AUDIT_CMD -u CCL_SKILL_BASE_REF RUBYOPT="${RUBYOPT:+$RUBYOPT }$2" ruby "$1" "$REPO" 2>&1)"
|
|
92
94
|
rc=$?
|
|
93
95
|
set -e
|
|
94
96
|
}
|
|
@@ -36,6 +36,8 @@
|
|
|
36
36
|
# Oracle: an always-refuse candidate fails on the pins; an always-accept
|
|
37
37
|
# candidate fails on the baseline-red case. Both arms are exercised.
|
|
38
38
|
set -euo pipefail
|
|
39
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
40
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
39
41
|
|
|
40
42
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
41
43
|
CANDIDATE_GATE="$SCRIPT_DIR/impact-chain-gate.rb"
|
|
@@ -35,6 +35,8 @@
|
|
|
35
35
|
# mistake were observed while writing this file. Do not "fix" a leg by relaxing
|
|
36
36
|
# its assertion.
|
|
37
37
|
set -euo pipefail
|
|
38
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
39
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
38
40
|
|
|
39
41
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
40
42
|
# REVERSE DIFFERENTIAL. Each leg below states which gate behavior it pins, and a
|
|
@@ -14,6 +14,8 @@
|
|
|
14
14
|
# 期望红的用例都必须点名 WHICH refusal:rc 单独是弱 oracle,为无关原因(fixture 缺陷、
|
|
15
15
|
# 锚点断掉)红同样是 rc=1,会把「形态已关闭」读成绿。
|
|
16
16
|
set -u
|
|
17
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
18
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
17
19
|
ROOT="$(cd "$(dirname "$0")/../../.." && pwd -P)"
|
|
18
20
|
GATE="$ROOT/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb"
|
|
19
21
|
LEDGER_REL="skills/skill-extraction-workflow/references/source-register.md"
|
|
@@ -7,6 +7,8 @@
|
|
|
7
7
|
#
|
|
8
8
|
# 全部确定性,不调模型。每用例一条分支,互不影响本地未提交改动。
|
|
9
9
|
set -u
|
|
10
|
+
# Ruby takes its encoding from the locale; under a POSIX/unset locale it reads the UTF-8 skill text as US-ASCII and crashes. Pin UTF-8, as CI runs.
|
|
11
|
+
case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOPT}" ;; esac
|
|
10
12
|
ROOT="$(cd "$(dirname "$0")/../../.." && pwd -P)"
|
|
11
13
|
GATE="$ROOT/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb"
|
|
12
14
|
TMP="$(mktemp -d)"; trap 'rm -rf "$TMP"' EXIT INT TERM
|