@ccoalm/ccl-skills 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/feature-risk-router/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/architecture-playbook.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/multi-tenant-isolation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/state-machine-task-patterns.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/inference-capacity-operations.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/llm-client-gateway.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +4 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/contracts-and-state.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/metrics-conventions.md +8 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/sli-slo-design.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/canary-and-rollout-strategy.md +16 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/promotion-gate-and-review.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +12 -12
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/code-review-checklist.md +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/multi-tenant-isolation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/state-machine-task-patterns.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +12 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +37 -30
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +81 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/classical-test-design-techniques.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc-review-and-prioritization.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/update-lifecycle.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +9 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +5 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/e2e-real-flow-testing.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/integration-contract-testing.md +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-code-authoring-patterns.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-topology-and-commands.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +2 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/annotation-driven-revision.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/figure-and-table-craft.md +8 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/react-architecture.md +3 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-quality-release.md +37 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-ui-quality.md +10 -1
- package/dist/assets/release.json +127 -67
- package/package.json +1 -1
|
@@ -35,6 +35,16 @@ description: Use when choosing test layers and regression evidence.
|
|
|
35
35
|
# Testing Strategy
|
|
36
36
|
EOF
|
|
37
37
|
|
|
38
|
+
# A second skill keeps the outcome space larger than {expected, none}, so
|
|
39
|
+
# acceptable[] fixtures below stay non-vacuous under the anti-gaming rule.
|
|
40
|
+
mkdir -p "$REPO/skills/tighten-doc"
|
|
41
|
+
cat > "$REPO/skills/tighten-doc/SKILL.md" <<'EOF'
|
|
42
|
+
---
|
|
43
|
+
description: Use when polishing document wording.
|
|
44
|
+
---
|
|
45
|
+
# Tighten Doc
|
|
46
|
+
EOF
|
|
47
|
+
|
|
38
48
|
cat > "$REPO/eval/routing-tasks.jsonl" <<'EOF'
|
|
39
49
|
{"id":"fake-claude-auth","utterance":"补一个回归测试","expected_skill":"testing-strategy","frozen_at_sha":""}
|
|
40
50
|
EOF
|
|
@@ -187,4 +197,216 @@ set -e
|
|
|
187
197
|
[ "$twice_rc" = 3 ] || fail "a grader that hangs on both attempts must stay unmeasured; rc=$twice_rc"
|
|
188
198
|
assert_contains "grader_timeout_1s" "$twice_out" "a doubly-timed-out task should still report the timeout"
|
|
189
199
|
|
|
200
|
+
# --- negative-control sentinel + failure-mode labels -------------------------
|
|
201
|
+
# expected_skill "none" is an outcome, not a catalog entry: a grader that picks a
|
|
202
|
+
# real skill for such a row must FAIL the task and label it "absorbed", and its
|
|
203
|
+
# clarify flag must be counted as a first-class metric.
|
|
204
|
+
none_bank="$TMP/bank-none.jsonl"
|
|
205
|
+
cat > "$none_bank" <<'EOF'
|
|
206
|
+
{"id":"neg-probe","utterance":"帮我写封邮件","expected_skill":"none","frozen_at_sha":""}
|
|
207
|
+
EOF
|
|
208
|
+
none_json="$TMP/report-none.json"
|
|
209
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
210
|
+
#!/usr/bin/env bash
|
|
211
|
+
printf '%s\n' '{"selected_skill": "testing-strategy", "clarify": true, "confidence": 0.4, "rationale_short": "nearest neighbor"}'
|
|
212
|
+
EOF
|
|
213
|
+
chmod +x "$FAKE_BIN/claude"
|
|
214
|
+
set +e
|
|
215
|
+
none_out="$(PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$none_bank" --json "$none_json" --timeout 60 2>&1)"
|
|
216
|
+
none_rc=$?
|
|
217
|
+
set -e
|
|
218
|
+
[ "$none_rc" = 0 ] || fail "sentinel run should complete advisory rc=0, got rc=$none_rc; out=$none_out"
|
|
219
|
+
none_status="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "status")' "$none_json")"
|
|
220
|
+
[ "$none_status" = "FAIL" ] || fail "a skill claiming an expected-none row must FAIL, got: $none_status"
|
|
221
|
+
none_labels="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "labels").join(",")' "$none_json")"
|
|
222
|
+
assert_contains "absorbed" "$none_labels" "expected-none row claimed by a skill must carry the absorbed label"
|
|
223
|
+
none_clarify="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0]))["clarify_count"]' "$none_json")"
|
|
224
|
+
[ "$none_clarify" = 1 ] || fail "clarify flag must be counted (expected clarify_count=1, got $none_clarify)"
|
|
225
|
+
none_lowconf="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0]))["low_confidence_count"]' "$none_json")"
|
|
226
|
+
[ "$none_lowconf" = 1 ] || fail "confidence 0.4 must count as low-confidence (<0.5), got $none_lowconf"
|
|
227
|
+
# And the honest direction: a grader answering "none" on the same row must PASS.
|
|
228
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
229
|
+
#!/usr/bin/env bash
|
|
230
|
+
printf '%s\n' '{"selected_skill": "none", "clarify": false, "confidence": 0.9, "rationale_short": "out of catalog"}'
|
|
231
|
+
EOF
|
|
232
|
+
chmod +x "$FAKE_BIN/claude"
|
|
233
|
+
set +e
|
|
234
|
+
none2_out="$(PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$none_bank" --json "$none_json" --timeout 60 2>&1)"
|
|
235
|
+
set -e
|
|
236
|
+
none2_status="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "status")' "$none_json")"
|
|
237
|
+
[ "$none2_status" = "PASS" ] || fail "a none verdict on an expected-none row must PASS, got: $none2_status; out=$none2_out"
|
|
238
|
+
|
|
239
|
+
# --- replicas: conservative consensus, ownership_split, agreement metric ------
|
|
240
|
+
split_bank="$TMP/bank-split.jsonl"
|
|
241
|
+
cat > "$split_bank" <<'EOF'
|
|
242
|
+
{"id":"split-probe","utterance":"补一个回归测试","expected_skill":"testing-strategy","frozen_at_sha":""}
|
|
243
|
+
EOF
|
|
244
|
+
split_json="$TMP/report-split.json"
|
|
245
|
+
rm -f "$GRADER_PID_FILE.calls"
|
|
246
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
247
|
+
#!/usr/bin/env bash
|
|
248
|
+
if [ -f "$GRADER_PID_FILE.calls" ]; then
|
|
249
|
+
printf '%s\n' '{"selected_skill": "none", "clarify": false, "confidence": 0.6, "rationale_short": "second replica disagrees"}'
|
|
250
|
+
else
|
|
251
|
+
: > "$GRADER_PID_FILE.calls"
|
|
252
|
+
printf '%s\n' '{"selected_skill": "testing-strategy", "clarify": false, "confidence": 0.9, "rationale_short": "first replica"}'
|
|
253
|
+
fi
|
|
254
|
+
EOF
|
|
255
|
+
chmod +x "$FAKE_BIN/claude"
|
|
256
|
+
set +e
|
|
257
|
+
split_out="$(PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$split_bank" --replicas 2 --json "$split_json" --timeout 60 2>&1)"
|
|
258
|
+
split_rc=$?
|
|
259
|
+
set -e
|
|
260
|
+
[ "$split_rc" = 0 ] || fail "replica run should complete advisory rc=0, got rc=$split_rc; out=$split_out"
|
|
261
|
+
split_status="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "status")' "$split_json")"
|
|
262
|
+
[ "$split_status" = "FAIL" ] || fail "one failing replica must fail the task (conservative consensus), got: $split_status"
|
|
263
|
+
split_labels="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "labels").join(",")' "$split_json")"
|
|
264
|
+
assert_contains "ownership_split" "$split_labels" "disagreeing replicas must carry the ownership_split label"
|
|
265
|
+
split_agree="$(ruby -rjson -e 'r=JSON.parse(File.read(ARGV[0]))["replica_agreement"]; print "#{r["agree"]}/#{r["measured"]}"' "$split_json")"
|
|
266
|
+
[ "$split_agree" = "0/1" ] || fail "expected replica_agreement 0/1, got: $split_agree"
|
|
267
|
+
split_verdicts="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "verdicts").length' "$split_json")"
|
|
268
|
+
[ "$split_verdicts" = 2 ] || fail "expected 2 recorded verdicts, got: $split_verdicts"
|
|
269
|
+
|
|
270
|
+
# --- replicas: a PASS+ERROR pair is PARTIALLY measured, never a clean PASS ----
|
|
271
|
+
# Task-level consensus sees only observed verdicts, so without replica-level
|
|
272
|
+
# error accounting a failed replica would vanish (zero grader-errors reported).
|
|
273
|
+
partial_bank="$TMP/bank-partial.jsonl"
|
|
274
|
+
cat > "$partial_bank" <<'EOF'
|
|
275
|
+
{"id":"partial-probe","utterance":"补一个回归测试","expected_skill":"testing-strategy","frozen_at_sha":""}
|
|
276
|
+
EOF
|
|
277
|
+
partial_json="$TMP/report-partial.json"
|
|
278
|
+
rm -f "$GRADER_PID_FILE.partial"
|
|
279
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
280
|
+
#!/usr/bin/env bash
|
|
281
|
+
if [ -f "$GRADER_PID_FILE.partial" ]; then
|
|
282
|
+
printf 'grader exploded\n' >&2
|
|
283
|
+
exit 1
|
|
284
|
+
fi
|
|
285
|
+
: > "$GRADER_PID_FILE.partial"
|
|
286
|
+
printf '%s\n' '{"selected_skill": "testing-strategy", "clarify": false, "confidence": 0.9, "rationale_short": "first replica ok"}'
|
|
287
|
+
EOF
|
|
288
|
+
chmod +x "$FAKE_BIN/claude"
|
|
289
|
+
set +e
|
|
290
|
+
partial_out="$(PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$partial_bank" --replicas 2 --json "$partial_json" --timeout 60 2>&1)"
|
|
291
|
+
partial_rc=$?
|
|
292
|
+
set -e
|
|
293
|
+
[ "$partial_rc" = 0 ] || fail "partial-error run should complete advisory rc=0, got rc=$partial_rc; out=$partial_out"
|
|
294
|
+
partial_status="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "status")' "$partial_json")"
|
|
295
|
+
[ "$partial_status" = "PASS" ] || fail "consensus over observed verdicts should stay PASS, got: $partial_status"
|
|
296
|
+
partial_labels="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "labels").join(",")' "$partial_json")"
|
|
297
|
+
assert_contains "partial_error" "$partial_labels" "a PASS+ERROR pair must carry the partial_error label"
|
|
298
|
+
partial_ev="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0]))["error_verdicts"]' "$partial_json")"
|
|
299
|
+
[ "$partial_ev" = 1 ] || fail "replica-level errors must be counted (expected error_verdicts=1, got $partial_ev)"
|
|
300
|
+
assert_contains "grader-error verdicts: 1" "$partial_out" "partial errors must be surfaced in human output"
|
|
301
|
+
|
|
302
|
+
# --- acceptable[]: a defensible alternate outcome passes, and is marked -------
|
|
303
|
+
acc_bank="$TMP/bank-acc.jsonl"
|
|
304
|
+
cat > "$acc_bank" <<'EOF'
|
|
305
|
+
{"id":"acc-probe","utterance":"用 COBOL 写个批处理","expected_skill":"testing-strategy","acceptable":["none"],"frozen_at_sha":""}
|
|
306
|
+
EOF
|
|
307
|
+
acc_json="$TMP/report-acc.json"
|
|
308
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
309
|
+
#!/usr/bin/env bash
|
|
310
|
+
printf '%s\n' '{"selected_skill": "none", "clarify": false, "confidence": 0.8, "rationale_short": "alternate outcome"}'
|
|
311
|
+
EOF
|
|
312
|
+
chmod +x "$FAKE_BIN/claude"
|
|
313
|
+
set +e
|
|
314
|
+
acc_out="$(PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$acc_bank" --json "$acc_json" --timeout 60 2>&1)"
|
|
315
|
+
acc_rc=$?
|
|
316
|
+
set -e
|
|
317
|
+
[ "$acc_rc" = 0 ] || fail "acceptable run should complete advisory rc=0, got rc=$acc_rc; out=$acc_out"
|
|
318
|
+
acc_status="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "status")' "$acc_json")"
|
|
319
|
+
[ "$acc_status" = "PASS" ] || fail "a selection inside acceptable[] must PASS, got: $acc_status"
|
|
320
|
+
acc_hit="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0])).dig("results", 0, "acceptable_hit")' "$acc_json")"
|
|
321
|
+
[ "$acc_hit" = "true" ] || fail "acceptable_hit must be recorded, got: $acc_hit"
|
|
322
|
+
|
|
323
|
+
# --- baseline ruler guard: mismatched bank or replicas suppresses the diff ----
|
|
324
|
+
# The docs declare reports from different (bank, replicas) configurations
|
|
325
|
+
# non-comparable; the runner must enforce that, or a stale baseline emits false
|
|
326
|
+
# newly_failed/newly_passed regressions.
|
|
327
|
+
ruler_bank="$TMP/bank-ruler.jsonl"
|
|
328
|
+
cat > "$ruler_bank" <<'EOF'
|
|
329
|
+
{"id":"ruler-probe","utterance":"补一个回归测试","expected_skill":"testing-strategy","frozen_at_sha":""}
|
|
330
|
+
EOF
|
|
331
|
+
cat > "$FAKE_BIN/claude" <<'EOF'
|
|
332
|
+
#!/usr/bin/env bash
|
|
333
|
+
printf '%s\n' '{"selected_skill": "none", "clarify": false, "confidence": 0.9, "rationale_short": "fails vs expected"}'
|
|
334
|
+
EOF
|
|
335
|
+
chmod +x "$FAKE_BIN/claude"
|
|
336
|
+
ruler_now="$TMP/report-ruler-now.json"
|
|
337
|
+
set +e
|
|
338
|
+
PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$ruler_bank" --json "$ruler_now" --timeout 60 >/dev/null 2>&1
|
|
339
|
+
set -e
|
|
340
|
+
[ -f "$ruler_now" ] || fail "expected current-ruler report to be written"
|
|
341
|
+
# Baseline A: same shape but a different bank fingerprint, task previously PASS.
|
|
342
|
+
ruler_base_bank="$TMP/report-ruler-base-bank.json"
|
|
343
|
+
ruby -rjson -e '
|
|
344
|
+
r = JSON.parse(File.read(ARGV[0]))
|
|
345
|
+
r["routing_surface"]["bank_sha256"] = "0" * 64
|
|
346
|
+
r["results"][0]["status"] = "PASS"
|
|
347
|
+
File.write(ARGV[1], JSON.generate(r))
|
|
348
|
+
' "$ruler_now" "$ruler_base_bank"
|
|
349
|
+
mismatch_json="$TMP/report-ruler-mismatch.json"
|
|
350
|
+
set +e
|
|
351
|
+
mismatch_out="$(PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$ruler_bank" --baseline "$ruler_base_bank" --json "$mismatch_json" --timeout 60 2>&1)"
|
|
352
|
+
set -e
|
|
353
|
+
assert_contains "baseline not compared" "$mismatch_out" "a bank-fingerprint mismatch must suppress the baseline diff"
|
|
354
|
+
mm_nf="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0]))["newly_failed"].length' "$mismatch_json")"
|
|
355
|
+
[ "$mm_nf" = 0 ] || fail "newly_failed must be empty under a mismatched-bank baseline, got $mm_nf entries"
|
|
356
|
+
mm_cmp="$(ruby -rjson -e 'print JSON.parse(File.read(ARGV[0]))["baseline_comparable"].inspect' "$mismatch_json")"
|
|
357
|
+
[ "$mm_cmp" = "false" ] || fail "baseline_comparable must be false for a bank mismatch, got: $mm_cmp"
|
|
358
|
+
# Baseline B: same bank fingerprint but a different replica count.
|
|
359
|
+
ruler_base_rep="$TMP/report-ruler-base-rep.json"
|
|
360
|
+
ruby -rjson -e '
|
|
361
|
+
r = JSON.parse(File.read(ARGV[0]))
|
|
362
|
+
r["replicas"] = 2
|
|
363
|
+
r["results"][0]["status"] = "PASS"
|
|
364
|
+
File.write(ARGV[1], JSON.generate(r))
|
|
365
|
+
' "$ruler_now" "$ruler_base_rep"
|
|
366
|
+
set +e
|
|
367
|
+
repmm_out="$(PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$ruler_bank" --baseline "$ruler_base_rep" --json "$mismatch_json" --timeout 60 2>&1)"
|
|
368
|
+
set -e
|
|
369
|
+
assert_contains "replica count differs" "$repmm_out" "a replica-count mismatch must suppress the baseline diff"
|
|
370
|
+
# Control: an identical-ruler baseline still produces the diff.
|
|
371
|
+
ruler_base_ok="$TMP/report-ruler-base-ok.json"
|
|
372
|
+
ruby -rjson -e '
|
|
373
|
+
r = JSON.parse(File.read(ARGV[0]))
|
|
374
|
+
r["results"][0]["status"] = "PASS"
|
|
375
|
+
File.write(ARGV[1], JSON.generate(r))
|
|
376
|
+
' "$ruler_now" "$ruler_base_ok"
|
|
377
|
+
set +e
|
|
378
|
+
okdiff_out="$(PATH="$FAKE_BIN:$PATH" ruby "$EVAL_SCRIPT" "$REPO" --bank "$ruler_bank" --baseline "$ruler_base_ok" --json "$mismatch_json" --timeout 60 2>&1)"
|
|
379
|
+
set -e
|
|
380
|
+
assert_contains "newly_failed=ruler-probe" "$okdiff_out" "an identical-ruler baseline must still yield the regression diff"
|
|
381
|
+
|
|
382
|
+
# --- schema strictness: vacuous rows and non-list fields are usage errors ----
|
|
383
|
+
vac_bank="$TMP/bank-vacuous.jsonl"
|
|
384
|
+
cat > "$vac_bank" <<'EOF'
|
|
385
|
+
{"id":"vac-probe","utterance":"x","expected_skill":"testing-strategy","acceptable":["tighten-doc","none"],"frozen_at_sha":""}
|
|
386
|
+
EOF
|
|
387
|
+
set +e
|
|
388
|
+
vac_out="$(ruby "$EVAL_SCRIPT" "$REPO" --bank "$vac_bank" --dry-run 2>&1)"
|
|
389
|
+
vac_rc=$?
|
|
390
|
+
set -e
|
|
391
|
+
[ "$vac_rc" = 2 ] || fail "a row covering the whole outcome space must be a usage error (rc=2), got rc=$vac_rc; out=$vac_out"
|
|
392
|
+
assert_contains "vacuous row" "$vac_out" "vacuous-row rejection should name the failure"
|
|
393
|
+
|
|
394
|
+
badtype_bank="$TMP/bank-badtype.jsonl"
|
|
395
|
+
cat > "$badtype_bank" <<'EOF'
|
|
396
|
+
{"id":"badtype-probe","utterance":"x","expected_skill":"testing-strategy","must_not_route_to":1,"frozen_at_sha":""}
|
|
397
|
+
EOF
|
|
398
|
+
set +e
|
|
399
|
+
badtype_out="$(ruby "$EVAL_SCRIPT" "$REPO" --bank "$badtype_bank" --dry-run 2>&1)"
|
|
400
|
+
badtype_rc=$?
|
|
401
|
+
set -e
|
|
402
|
+
[ "$badtype_rc" = 2 ] || fail "a non-list must_not_route_to must be a usage error (rc=2), got rc=$badtype_rc; out=$badtype_out"
|
|
403
|
+
assert_contains "must_not_route_to must be a list" "$badtype_out" "non-list field must produce the documented diagnostic, not a stack trace"
|
|
404
|
+
|
|
405
|
+
# --- --replicas argument strictness ------------------------------------------
|
|
406
|
+
set +e
|
|
407
|
+
rep_out="$(ruby "$EVAL_SCRIPT" "$REPO" --replicas 0 2>&1)"
|
|
408
|
+
rep_rc=$?
|
|
409
|
+
set -e
|
|
410
|
+
[ "$rep_rc" = 2 ] || fail "--replicas 0 must be a usage error (rc=2), got rc=$rep_rc; out=$rep_out"
|
|
411
|
+
|
|
190
412
|
echo "test_eval_routing_bank_grader_diagnostics: ok"
|
|
@@ -33,7 +33,7 @@ import sys
|
|
|
33
33
|
from pathlib import Path
|
|
34
34
|
|
|
35
35
|
args = [item.decode() for item in Path(sys.argv[1]).read_bytes().split(b"\0") if item]
|
|
36
|
-
assert args[:2] == ["--challenge-budget", "
|
|
36
|
+
assert args[:2] == ["--challenge-budget", "1"], args
|
|
37
37
|
assert args.count("--challenge-budget") == 1, args
|
|
38
38
|
PY
|
|
39
39
|
|
|
@@ -47,14 +47,14 @@ import sys
|
|
|
47
47
|
from pathlib import Path
|
|
48
48
|
|
|
49
49
|
args = [item.decode() for item in Path(sys.argv[1]).read_bytes().split(b"\0") if item]
|
|
50
|
-
assert args[:2] == ["--challenge-budget", "
|
|
50
|
+
assert args[:2] == ["--challenge-budget", "1"], args
|
|
51
51
|
assert args[-1] == "--base", args
|
|
52
52
|
assert args.count("--challenge-budget") == 1, args
|
|
53
53
|
PY
|
|
54
54
|
|
|
55
55
|
# Exercise the installed wrapper/controller pair without invoking a model. The
|
|
56
56
|
# same real controller defaults to budget 0 when called directly, while the
|
|
57
|
-
# extraction wrapper must make the emitted receipt report budget
|
|
57
|
+
# extraction wrapper must make the emitted receipt report budget 1.
|
|
58
58
|
python3 - "$TMP/real-controller.diff" "$TMP/real-controller-plan.json" <<'PY'
|
|
59
59
|
import json
|
|
60
60
|
import sys
|
|
@@ -74,7 +74,7 @@ conclusions = {
|
|
|
74
74
|
"safety": "The probe selects only the implementer family and invokes no external reviewer.",
|
|
75
75
|
"failure_paths": "The no-independent-reviewer boundary remains structured and fail closed.",
|
|
76
76
|
"tests_evidence": "Direct and wrapped calls provide a differential budget assertion.",
|
|
77
|
-
"compatibility": "The generic controller default remains zero while extraction fixes
|
|
77
|
+
"compatibility": "The generic controller default remains zero while extraction fixes one.",
|
|
78
78
|
}
|
|
79
79
|
skills = {
|
|
80
80
|
"correctness": "skill-extraction-workflow",
|
|
@@ -84,8 +84,8 @@ skills = {
|
|
|
84
84
|
"compatibility": "terminal-cli-dev",
|
|
85
85
|
}
|
|
86
86
|
plan = {
|
|
87
|
-
"intent": "Prove the extraction wrapper and real review controller agree on budget
|
|
88
|
-
"acceptance": ["The wrapped real-controller receipt reports challenge_budget
|
|
87
|
+
"intent": "Prove the extraction wrapper and real review controller agree on budget one.",
|
|
88
|
+
"acceptance": ["The wrapped real-controller receipt reports challenge_budget one."],
|
|
89
89
|
"self_review": [
|
|
90
90
|
{
|
|
91
91
|
"concern": concern,
|
|
@@ -134,7 +134,7 @@ assert_rc "$wrapped_rc" 2 "wrapped real controller must stop before model infere
|
|
|
134
134
|
assert_contains '"reason_code":"no_independent_reviewer_available"' "$direct_out" "direct real-controller boundary"
|
|
135
135
|
assert_contains '"reason_code":"no_independent_reviewer_available"' "$wrapped_out" "wrapped real-controller boundary"
|
|
136
136
|
assert_contains '"challenge_budget":0' "$direct_out" "generic controller default budget"
|
|
137
|
-
assert_contains '"challenge_budget":
|
|
137
|
+
assert_contains '"challenge_budget":1' "$wrapped_out" "wrapper-enforced real-controller budget"
|
|
138
138
|
[ ! -e "$TMP/codex-invoked" ] || fail "same-family Codex executable was invoked"
|
|
139
139
|
|
|
140
140
|
# Join the producer and consumer contracts. First feed the exact real receipt
|
|
@@ -231,7 +231,7 @@ for spelling in --challenge-budget --challenge-budget=4 --challenge-b=4; do
|
|
|
231
231
|
rc=$?
|
|
232
232
|
set -e
|
|
233
233
|
assert_rc "$rc" 2 "caller budget override must be rejected"
|
|
234
|
-
assert_contains "challenge budget is fixed at
|
|
234
|
+
assert_contains "challenge budget is fixed at 1" "$out" "override reason"
|
|
235
235
|
[ ! -s "$TMP/args" ] || fail "controller ran after budget override"
|
|
236
236
|
done
|
|
237
237
|
|
|
@@ -321,16 +321,22 @@ assert "--challenge-budget" not in quickstart, (
|
|
|
321
321
|
"quickstart must not let callers override the extraction review budget"
|
|
322
322
|
)
|
|
323
323
|
assert re.search(
|
|
324
|
-
r"[Rr]ound 2.{0,500}ready_for_human_decision",
|
|
324
|
+
r"[Rr]ound 2 challenge.{0,500}ready_for_human_decision",
|
|
325
325
|
quickstart,
|
|
326
326
|
re.DOTALL,
|
|
327
327
|
), "quickstart does not validate an early-clean round-2 terminal checkpoint"
|
|
328
328
|
assert re.search(
|
|
329
|
-
r"[Rr]ound
|
|
329
|
+
r"[Rr]ound 2 findings.{0,200}continuation_authorization_required",
|
|
330
330
|
quickstart,
|
|
331
331
|
re.DOTALL,
|
|
332
332
|
), "quickstart does not validate the exhausted-budget terminal checkpoint"
|
|
333
333
|
assert "baseline_race" in quickstart
|
|
334
|
+
assert "at most two challenges" not in quickstart, (
|
|
335
|
+
"quickstart still advertises the retired two-challenge budget"
|
|
336
|
+
)
|
|
337
|
+
assert "At the third Agent-autonomous round" not in dual, (
|
|
338
|
+
"dual-track still gates the lane at the retired third round"
|
|
339
|
+
)
|
|
334
340
|
PY
|
|
335
341
|
|
|
336
342
|
echo "test_extraction_review_gate: ok"
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Frozen-case sanctity gate (regressions-are-sacred, deterministic half).
|
|
3
|
+
#
|
|
4
|
+
# SCOPE — read this before citing the lane as evidence. A frozen eval case
|
|
5
|
+
# (routing task-bank row, golden trace) is a pinned judgment: deleting it or
|
|
6
|
+
# rewriting its judgment fields makes a red disappear without anyone ruling on
|
|
7
|
+
# it. This lane makes that trade VISIBLE: any base-relative deletion or
|
|
8
|
+
# judgment-field change of a frozen case must be named by an ADDED
|
|
9
|
+
# source-register line carrying `case-retired: <id>` or `case-rescoped: <id>`.
|
|
10
|
+
# It does NOT judge whether the retirement/rescope is justified — that ruling
|
|
11
|
+
# belongs to the round's independent review (the register row is what puts it
|
|
12
|
+
# in front of the reviewer). Like the impact-chain gate, it trusts the author
|
|
13
|
+
# to write the declaration; it defends against SILENT trades, not forged ones.
|
|
14
|
+
# Without CCL_SKILL_BASE_REF (or an unresolvable base) it degrades to an
|
|
15
|
+
# explicit skip token — a skipped run is not a passed run.
|
|
16
|
+
set -u
|
|
17
|
+
|
|
18
|
+
ROOT="${SANCTITY_ROOT:-$(cd "$(dirname "$0")/../../.." && pwd)}"
|
|
19
|
+
[ -n "${SANCTITY_ROOT:-}" ] && echo "NOTICE: SANCTITY_ROOT set — testing tree: $ROOT" >&2
|
|
20
|
+
REGISTER="skills/skill-extraction-workflow/references/source-register.md"
|
|
21
|
+
BANK="eval/routing-tasks.jsonl"
|
|
22
|
+
TRACES_DIR="eval/golden-traces"
|
|
23
|
+
|
|
24
|
+
BASE="${CCL_SKILL_BASE_REF:-}"
|
|
25
|
+
if [ -z "$BASE" ]; then
|
|
26
|
+
echo "frozen_case_sanctity_skipped no-base-ref (set CCL_SKILL_BASE_REF to enable; skipped is not passed)"
|
|
27
|
+
exit 0
|
|
28
|
+
fi
|
|
29
|
+
if ! git -C "$ROOT" rev-parse --verify --quiet "${BASE}^{commit}" >/dev/null; then
|
|
30
|
+
echo "frozen_case_sanctity_skipped base-unresolvable ($BASE; skipped is not passed)"
|
|
31
|
+
exit 0
|
|
32
|
+
fi
|
|
33
|
+
|
|
34
|
+
TMPDIR_SANCTITY="$(mktemp -d)"
|
|
35
|
+
trap 'rm -rf "$TMPDIR_SANCTITY"' EXIT
|
|
36
|
+
|
|
37
|
+
# Base snapshots; a path absent at base means every current case is new (no
|
|
38
|
+
# violation possible on that surface).
|
|
39
|
+
git -C "$ROOT" show "$BASE:$BANK" > "$TMPDIR_SANCTITY/base-bank.jsonl" 2>/dev/null || : > "$TMPDIR_SANCTITY/base-bank.jsonl"
|
|
40
|
+
git -C "$ROOT" show "$BASE:$REGISTER" > "$TMPDIR_SANCTITY/base-register.md" 2>/dev/null || : > "$TMPDIR_SANCTITY/base-register.md"
|
|
41
|
+
mkdir -p "$TMPDIR_SANCTITY/base-traces"
|
|
42
|
+
# -r: enumerate blobs recursively so a trace in a subdirectory is still guarded
|
|
43
|
+
# (the head-side walk below recurses symmetrically).
|
|
44
|
+
n=0
|
|
45
|
+
git -C "$ROOT" ls-tree -r --name-only "$BASE" -- "$TRACES_DIR/" 2>/dev/null | while IFS= read -r tp; do
|
|
46
|
+
case "$tp" in
|
|
47
|
+
*.json)
|
|
48
|
+
n=$((n+1))
|
|
49
|
+
git -C "$ROOT" show "$BASE:$tp" > "$TMPDIR_SANCTITY/base-traces/trace-$n-$(basename "$tp")" 2>/dev/null || true ;;
|
|
50
|
+
esac
|
|
51
|
+
done
|
|
52
|
+
|
|
53
|
+
python3 - "$ROOT" "$TMPDIR_SANCTITY" "$REGISTER" "$BANK" "$TRACES_DIR" <<'PY'
|
|
54
|
+
import json, os, sys
|
|
55
|
+
|
|
56
|
+
root, tmp, register_rel, bank_rel, traces_rel = sys.argv[1:6]
|
|
57
|
+
fail = 0
|
|
58
|
+
|
|
59
|
+
def bad(msg):
|
|
60
|
+
global fail
|
|
61
|
+
print(f"FAIL: {msg}", file=sys.stderr)
|
|
62
|
+
fail = 1
|
|
63
|
+
|
|
64
|
+
def load_bank(path):
|
|
65
|
+
rows = {}
|
|
66
|
+
if not os.path.exists(path):
|
|
67
|
+
return rows
|
|
68
|
+
with open(path, encoding="utf-8") as fh:
|
|
69
|
+
for raw in fh:
|
|
70
|
+
raw = raw.strip()
|
|
71
|
+
if not raw:
|
|
72
|
+
continue
|
|
73
|
+
try:
|
|
74
|
+
row = json.loads(raw)
|
|
75
|
+
except json.JSONDecodeError:
|
|
76
|
+
continue # malformed rows are the integrity lane's finding, not ours
|
|
77
|
+
rid = row.get("id")
|
|
78
|
+
if rid:
|
|
79
|
+
rows[rid] = row
|
|
80
|
+
return rows
|
|
81
|
+
|
|
82
|
+
def judgment(row):
|
|
83
|
+
# The judgment surface of a bank case: what outcome passes or fails it.
|
|
84
|
+
# Provenance/commentary fields (source, why_expected, frozen_at_sha) may
|
|
85
|
+
# change without re-scoping the case.
|
|
86
|
+
return (
|
|
87
|
+
row.get("expected_skill"),
|
|
88
|
+
sorted(row.get("acceptable") or []) if isinstance(row.get("acceptable"), list) else row.get("acceptable"),
|
|
89
|
+
sorted(row.get("must_not_route_to") or []) if isinstance(row.get("must_not_route_to"), list) else row.get("must_not_route_to"),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
base_bank = load_bank(os.path.join(tmp, "base-bank.jsonl"))
|
|
93
|
+
head_bank = load_bank(os.path.join(root, bank_rel))
|
|
94
|
+
|
|
95
|
+
violations = [] # (kind, id)
|
|
96
|
+
for rid, row in base_bank.items():
|
|
97
|
+
if rid not in head_bank:
|
|
98
|
+
violations.append(("deleted bank case", rid))
|
|
99
|
+
elif judgment(row) != judgment(head_bank[rid]):
|
|
100
|
+
violations.append(("re-scoped bank case (judgment fields changed)", rid))
|
|
101
|
+
|
|
102
|
+
def load_trace(path):
|
|
103
|
+
try:
|
|
104
|
+
with open(path, encoding="utf-8") as fh:
|
|
105
|
+
return json.load(fh)
|
|
106
|
+
except (OSError, json.JSONDecodeError):
|
|
107
|
+
return None
|
|
108
|
+
|
|
109
|
+
base_traces = {}
|
|
110
|
+
base_dir = os.path.join(tmp, "base-traces")
|
|
111
|
+
for name in sorted(os.listdir(base_dir)):
|
|
112
|
+
data = load_trace(os.path.join(base_dir, name))
|
|
113
|
+
if isinstance(data, dict) and data.get("id"):
|
|
114
|
+
base_traces[data["id"]] = data
|
|
115
|
+
|
|
116
|
+
head_traces = {}
|
|
117
|
+
head_dir = os.path.join(root, traces_rel)
|
|
118
|
+
if os.path.isdir(head_dir):
|
|
119
|
+
for dirpath, _dirnames, filenames in os.walk(head_dir):
|
|
120
|
+
for name in sorted(filenames):
|
|
121
|
+
if not name.endswith(".json"):
|
|
122
|
+
continue
|
|
123
|
+
data = load_trace(os.path.join(dirpath, name))
|
|
124
|
+
if isinstance(data, dict) and data.get("id"):
|
|
125
|
+
head_traces[data["id"]] = data
|
|
126
|
+
|
|
127
|
+
for tid, data in base_traces.items():
|
|
128
|
+
if tid not in head_traces:
|
|
129
|
+
violations.append(("deleted golden trace", tid))
|
|
130
|
+
elif data.get("assert") != head_traces[tid].get("assert"):
|
|
131
|
+
violations.append(("re-scoped golden trace (assert block changed)", tid))
|
|
132
|
+
|
|
133
|
+
# Added register lines = head lines not present at base (the register is
|
|
134
|
+
# append-only, so a set diff is the added-row surface). Only added Markdown
|
|
135
|
+
# TABLE ROWS may carry adjudication credit — an adjudication is a register row
|
|
136
|
+
# a reviewer rules on, so an HTML comment or loose prose that merely mentions
|
|
137
|
+
# the token mints nothing.
|
|
138
|
+
with open(os.path.join(tmp, "base-register.md"), encoding="utf-8") as fh:
|
|
139
|
+
base_lines = set(fh.read().splitlines())
|
|
140
|
+
register_path = os.path.join(root, register_rel)
|
|
141
|
+
added_rows = []
|
|
142
|
+
if os.path.exists(register_path):
|
|
143
|
+
with open(register_path, encoding="utf-8") as fh:
|
|
144
|
+
added_rows = [
|
|
145
|
+
ln for ln in fh.read().splitlines()
|
|
146
|
+
if ln not in base_lines and ln.lstrip().startswith("|")
|
|
147
|
+
]
|
|
148
|
+
added_blob = "\n".join(added_rows)
|
|
149
|
+
|
|
150
|
+
import re
|
|
151
|
+
def adjudicated(case_id):
|
|
152
|
+
# Token-bounded: `case-retired: foo` must not be credited by
|
|
153
|
+
# `case-retired: foobar` (ids use [A-Za-z0-9_-]).
|
|
154
|
+
pattern = re.compile(
|
|
155
|
+
r"case-(?:retired|rescoped):\s*" + re.escape(case_id) + r"(?![A-Za-z0-9_-])"
|
|
156
|
+
)
|
|
157
|
+
return bool(pattern.search(added_blob))
|
|
158
|
+
|
|
159
|
+
for kind, cid in violations:
|
|
160
|
+
if adjudicated(cid):
|
|
161
|
+
continue
|
|
162
|
+
bad(
|
|
163
|
+
f"{kind} '{cid}' with no adjudication row — a frozen case may not be "
|
|
164
|
+
f"silently traded away; add a source-register row this round containing "
|
|
165
|
+
f"`case-retired: {cid}` or `case-rescoped: {cid}` (with the why), so the "
|
|
166
|
+
f"independent review rules on it"
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
if fail:
|
|
170
|
+
sys.exit(1)
|
|
171
|
+
print(f"frozen_case_sanctity_ok bank_base={len(base_bank)} traces_base={len(base_traces)} violations=0")
|
|
172
|
+
PY
|
|
173
|
+
status=$?
|
|
174
|
+
if [ $status -ne 0 ]; then
|
|
175
|
+
echo "frozen_case_sanctity_failed" >&2
|
|
176
|
+
exit 1
|
|
177
|
+
fi
|
|
178
|
+
exit 0
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Self-proof for test_frozen_case_sanctity.sh: builds a synthetic repo and
|
|
3
|
+
# proves the oracle can fail for the right reason on every guarded surface —
|
|
4
|
+
# deletion, judgment re-scope, subdirectory golden-trace deletion — and that
|
|
5
|
+
# adjudication credit cannot be minted by a prefix-colliding id or an HTML
|
|
6
|
+
# comment (only an added register TABLE ROW with a token-bounded id counts).
|
|
7
|
+
# The unmutated control and the properly-adjudicated leg must stay green, and
|
|
8
|
+
# the no-base leg must print the explicit skip token. Runs in mktemp only.
|
|
9
|
+
set -u
|
|
10
|
+
|
|
11
|
+
SCRIPTS_DIR="$(cd "$(dirname "$0")" && pwd)"
|
|
12
|
+
GATE="$SCRIPTS_DIR/test_frozen_case_sanctity.sh"
|
|
13
|
+
[ -f "$GATE" ] || { echo "FAIL: gate script missing ($GATE)" >&2; exit 1; }
|
|
14
|
+
|
|
15
|
+
WORK="$(mktemp -d)"
|
|
16
|
+
trap 'rm -rf "$WORK"' EXIT
|
|
17
|
+
REPO="$WORK/repo"
|
|
18
|
+
mkdir -p "$REPO/eval/golden-traces/nested" "$REPO/skills/skill-extraction-workflow/references"
|
|
19
|
+
|
|
20
|
+
BANK="$REPO/eval/routing-tasks.jsonl"
|
|
21
|
+
REG="$REPO/skills/skill-extraction-workflow/references/source-register.md"
|
|
22
|
+
cat > "$BANK" <<'EOF'
|
|
23
|
+
{"id": "a1", "utterance": "u1", "expected_skill": "s-one", "why_expected": "w", "frozen_at_sha": "root"}
|
|
24
|
+
{"id": "a1x", "utterance": "u2", "expected_skill": "s-two", "why_expected": "w", "frozen_at_sha": "root"}
|
|
25
|
+
EOF
|
|
26
|
+
cat > "$REPO/eval/golden-traces/nested/t1.json" <<'EOF'
|
|
27
|
+
{"id": "trace-nested-one", "assert": {"must_invoke_skill": "s-one"}}
|
|
28
|
+
EOF
|
|
29
|
+
printf '| base row |\n' > "$REG"
|
|
30
|
+
|
|
31
|
+
git -C "$REPO" init -q
|
|
32
|
+
git -C "$REPO" -c user.email=t@t -c user.name=t add -A
|
|
33
|
+
git -C "$REPO" -c user.email=t@t -c user.name=t commit -qm base
|
|
34
|
+
BASE_SHA="$(git -C "$REPO" rev-parse HEAD)"
|
|
35
|
+
|
|
36
|
+
pass=0
|
|
37
|
+
fail=0
|
|
38
|
+
ok() { pass=$((pass+1)); }
|
|
39
|
+
bad() { fail=$((fail+1)); echo "FAIL: $1" >&2; }
|
|
40
|
+
|
|
41
|
+
run_gate() {
|
|
42
|
+
SANCTITY_ROOT="$REPO" CCL_SKILL_BASE_REF="$BASE_SHA" bash "$GATE" > "$WORK/out.log" 2> "$WORK/err.log"
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
reset_tree() { git -C "$REPO" checkout -q -- .; }
|
|
46
|
+
|
|
47
|
+
# Leg 1: unmutated control is green with the success token.
|
|
48
|
+
reset_tree
|
|
49
|
+
if run_gate && grep -q "frozen_case_sanctity_ok" "$WORK/out.log"; then ok; else bad "control leg not green"; fi
|
|
50
|
+
|
|
51
|
+
# Leg 2: deleting bank case a1 reds, naming a1.
|
|
52
|
+
reset_tree
|
|
53
|
+
grep -v '"id": "a1",' "$BANK" > "$BANK.t" && mv "$BANK.t" "$BANK"
|
|
54
|
+
if run_gate; then bad "bank deletion passed"; else
|
|
55
|
+
grep -q "deleted bank case 'a1'" "$WORK/err.log" && ok || bad "bank deletion red but not attributed to a1"
|
|
56
|
+
fi
|
|
57
|
+
|
|
58
|
+
# Leg 3: re-scoping a1's expected_skill reds.
|
|
59
|
+
reset_tree
|
|
60
|
+
python3 - "$BANK" <<'PY'
|
|
61
|
+
import json, sys
|
|
62
|
+
path = sys.argv[1]
|
|
63
|
+
rows = [json.loads(l) for l in open(path) if l.strip()]
|
|
64
|
+
rows[0]["expected_skill"] = "s-two"
|
|
65
|
+
open(path, "w").write("\n".join(json.dumps(r) for r in rows) + "\n")
|
|
66
|
+
PY
|
|
67
|
+
if run_gate; then bad "re-scope passed"; else
|
|
68
|
+
grep -q "re-scoped bank case (judgment fields changed) 'a1'" "$WORK/err.log" && ok || bad "re-scope red but not attributed"
|
|
69
|
+
fi
|
|
70
|
+
|
|
71
|
+
# Leg 4: deleting the SUBDIRECTORY golden trace reds (recursive base walk).
|
|
72
|
+
reset_tree
|
|
73
|
+
rm "$REPO/eval/golden-traces/nested/t1.json"
|
|
74
|
+
if run_gate; then bad "nested trace deletion passed"; else
|
|
75
|
+
grep -q "deleted golden trace 'trace-nested-one'" "$WORK/err.log" && ok || bad "trace deletion red but not attributed"
|
|
76
|
+
fi
|
|
77
|
+
|
|
78
|
+
# Leg 5: a prefix-colliding id mints no credit (case-retired: a1x != a1).
|
|
79
|
+
reset_tree
|
|
80
|
+
grep -v '"id": "a1",' "$BANK" > "$BANK.t" && mv "$BANK.t" "$BANK"
|
|
81
|
+
printf '| retired | case-retired: a1x | reason |\n' >> "$REG"
|
|
82
|
+
if run_gate; then bad "prefix-colliding adjudication credited"; else
|
|
83
|
+
grep -q "deleted bank case 'a1'" "$WORK/err.log" && ok || bad "prefix leg red but not attributed"
|
|
84
|
+
fi
|
|
85
|
+
|
|
86
|
+
# Leg 6: an HTML comment mints no credit (row-only surface).
|
|
87
|
+
reset_tree
|
|
88
|
+
grep -v '"id": "a1",' "$BANK" > "$BANK.t" && mv "$BANK.t" "$BANK"
|
|
89
|
+
printf '<!-- case-retired: a1 -->\n' >> "$REG"
|
|
90
|
+
if run_gate; then bad "HTML-comment adjudication credited"; else
|
|
91
|
+
grep -q "deleted bank case 'a1'" "$WORK/err.log" && ok || bad "comment leg red but not attributed"
|
|
92
|
+
fi
|
|
93
|
+
|
|
94
|
+
# Leg 7: a real added table row with the exact id is credited — green.
|
|
95
|
+
reset_tree
|
|
96
|
+
grep -v '"id": "a1",' "$BANK" > "$BANK.t" && mv "$BANK.t" "$BANK"
|
|
97
|
+
printf '| retired | case-retired: a1 | superseded by a2 probes |\n' >> "$REG"
|
|
98
|
+
if run_gate && grep -q "frozen_case_sanctity_ok" "$WORK/out.log"; then ok; else bad "adjudicated deletion not green"; fi
|
|
99
|
+
|
|
100
|
+
# Leg 8: no base ref prints the explicit skip token and exits 0. env -u so an
|
|
101
|
+
# outer CI-supplied CCL_SKILL_BASE_REF cannot leak into this leg (the known
|
|
102
|
+
# nested-suite leak class).
|
|
103
|
+
reset_tree
|
|
104
|
+
if env -u CCL_SKILL_BASE_REF SANCTITY_ROOT="$REPO" bash "$GATE" > "$WORK/out.log" 2>&1 && grep -q "frozen_case_sanctity_skipped no-base-ref" "$WORK/out.log"; then ok; else bad "no-base leg missing skip token"; fi
|
|
105
|
+
|
|
106
|
+
echo "frozen_case_sanctity_selfproof: pass=$pass fail=$fail"
|
|
107
|
+
[ "$fail" -eq 0 ] && [ "$pass" -eq 8 ] || exit 1
|
|
108
|
+
echo "frozen_case_sanctity_selfproof_ok"
|