@ccoalm/ccl-skills 0.15.1 → 0.15.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +3 -1
  2. package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +1 -1
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +2 -2
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -5
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +26 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +37 -13
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +5 -2
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +77 -5
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_packet_mcp.py +98 -4
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_cli_review.py +48 -1
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +230 -16
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +165 -11
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_kimi_packet_mcp.py +143 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +572 -0
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +3 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +1 -1
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -0
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +1 -1
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +2 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +3 -1
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +2 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +1 -1
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +20 -11
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/refactoring-discipline.md +7 -1
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +1 -1
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +2 -2
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +17 -17
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +4 -4
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/resume-paused-delivery.md +3 -3
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +11 -0
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +83 -48
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +80 -2
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +3 -2
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +190 -2
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +106 -4
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +1 -1
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +3 -1
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -1
  42. package/dist/assets/release.json +54 -49
  43. package/dist/codex-host.d.ts +1 -1
  44. package/dist/codex-host.js +39 -15
  45. package/dist/host-probe.d.ts +16 -0
  46. package/dist/host-probe.js +29 -5
  47. package/dist/operations.js +34 -8
  48. package/dist/unified.d.ts +1 -1
  49. package/dist/unified.js +11 -4
  50. package/package.json +1 -1
@@ -238,8 +238,21 @@ assert_same_paragraph "$REPO_ROOT/skills/skill-extraction-workflow/references/du
238
238
  "product-rd-workflow/references/design-review-gate-mechanics.md" \
239
239
  "claim liveness (shared-skill instantiation pointer)"
240
240
 
241
- # 2. Affirmative-assent binding
242
- assert_contains "$PRE_FINAL_REF" 'the required `proposed-next:` marker makes that binding observable' "assent binding (destination)"
241
+ # 2. Text-consistency pins for intent recovery and retained boundaries.
242
+ assert_contains "$PRE_FINAL_REF" 'even without a `proposed-next:` marker' "assent binding (unmarked recovery)"
243
+ assert_contains "$PRODUCT_SKILL" 'A missing, repeated, or conflicting marker triggers intent recovery, not a stop.' "assent binding (format recovery)"
244
+ assert_contains "$PRE_FINAL_REF" 'A marker alone never supplies missing authority' "assent binding (authority boundary)"
245
+ assert_contains "$PRODUCT_SKILL" 'Scope each blocker to its dependent action or claim.' "continuation (dependent blocker scope)"
246
+ assert_contains "$PRODUCT_SKILL" 'An unproven cause blocks the speculative patch, not available diagnosis' "continuation (diagnosis recovery)"
247
+ assert_contains "$PRODUCT_SKILL" 'Every user reply immediately following an assistant message that states or implies a next action requires a visible `continuing:` or `blocked:` outcome before finalizing' "continuation (mandatory reply outcome)"
248
+ assert_contains "$PRE_FINAL_REF" 'Independent work must neither depend on the pending verdict nor modify the candidate being evaluated.' "continuation (independence boundary)"
249
+ assert_contains "$PRODUCT_SKILL" 'Never bypass the blocked gate, invent a pass, widen scope' "continuation (no gate bypass)"
250
+ assert_same_bullet "$PRODUCT_SKILL" 'Quality-gate failures require diagnosis and available related behavior-preserving cleanup before escalation' \
251
+ 'references/refactoring-discipline.md' "quality gate (entry signal+pointer)"
252
+ assert_contains "$PRE_FINAL_REF" 'inspect and perform a safe structural cleanup related to the current change when available, then rerun the gate and affected tests' "quality gate (remediation before escalation)"
253
+ assert_contains "$REPO_ROOT/skills/product-rd-workflow/references/refactoring-discipline.md" 'do not ask again merely because it involves refactoring' "quality gate (authorized cleanup)"
254
+ assert_contains "$REPO_ROOT/skills/product-rd-workflow/references/refactoring-discipline.md" 'Do not abbreviate meaningful names, remove necessary explanations, pack statements, fragment responsibilities arbitrarily, or change the threshold/history just to satisfy a counter.' "quality gate (readability and metric integrity)"
255
+ assert_contains "$REPO_ROOT/skills/product-rd-workflow/references/refactoring-discipline.md" 'Broader redesign and breaking changes retain their scope and approval checks.' "quality gate (scope and compatibility boundary)"
243
256
  assert_same_bullet "$PRODUCT_SKILL" 'Affirmative-assent binding rule' \
244
257
  "references/pre-final-continuation-gate.md" "assent binding (entry signal+pointer)"
245
258
  assert_contains "$PRODUCT_SKILL" 'self-classifying the reply or marker away is never an exit' "assent binding (entry no-exit clause)"
@@ -746,52 +759,35 @@ assert_same_line "$DUAL_TRACK_REF" 'cannot be checked false and is inconclusive'
746
759
  assert_same_line "$DUAL_TRACK_REF" 'any "full X" adjective is scoped to the named axes, never wider' \
747
760
  'must be written falsifiably' \
748
761
  "process controls (full-adjective scoped to named axes)"
749
- # Continuation authorization: third human state; binding dead-end is by design; checkpoint names the remainder.
750
- assert_same_line "$DUAL_TRACK_REF" 'it waives nothing and decides no merge' \
751
- '`continuation_authorization`' \
752
- "process controls (continuation authorization waives nothing)"
753
- assert_same_line "$DUAL_TRACK_REF" 'both lanes stay intact and blocking' \
754
- '`continuation_authorization`' \
755
- "process controls (both lanes stay intact and blocking)"
756
- assert_same_line "$DUAL_TRACK_REF" 'never counted as Agent-autonomous' \
757
- '`continuation_authorization`' \
758
- "process controls (human-authorized rounds not Agent-autonomous)"
759
- assert_same_line "$DUAL_TRACK_REF" 'run as a fresh chain bound to the current candidate' \
760
- '`continuation_authorization`' \
761
- "process controls (continuation rounds take a fresh current-candidate chain)"
762
- assert_same_line "$DUAL_TRACK_REF" "carries forward the complete review ledger and every prior round's focuses and dispositions" \
763
- '`continuation_authorization`' \
764
- "process controls (fresh chain restarts binding, never history)"
765
- assert_same_line "$DUAL_TRACK_REF" 'The grant itself is scope-bound, not reusable' \
766
- '`continuation_authorization`' \
767
- "process controls (continuation grant is scope-bound)"
768
- assert_same_line "$DUAL_TRACK_REF" "it names the granting session and either one exact candidate or, explicitly, this program's rounds to convergence in that session" \
769
- '`continuation_authorization`' \
770
- "process controls (the grant scope is defined: session plus exact candidate or explicit to-convergence)"
771
- assert_same_line "$DUAL_TRACK_REF" 'a candidate or session outside the named scope requires a fresh authorization' \
772
- '`continuation_authorization`' \
773
- "process controls (out-of-scope continuation needs a fresh grant)"
774
- assert_same_line "$DUAL_TRACK_REF" "a finding's fix that edits the owner package's own files breaks the review chain's content binding" \
775
- '`continuation_authorization`' \
776
- "process controls (binding dead-end is by design)"
777
- assert_same_line "$DUAL_TRACK_REF" 'refuses both another autonomous round and a challenge bound to the stale prior result' \
778
- '`continuation_authorization`' \
779
- "process controls (tracker refusal on stale binding)"
780
- assert_same_line "$DUAL_TRACK_REF" "names each lane's terminal state and the exact un-run remainder" \
781
- '`continuation_authorization`' \
782
- "process controls (interim checkpoint names the un-run remainder)"
783
- assert_same_line "$DUAL_TRACK_REF" 'explicit risk acceptance with the record as the disposition trail' \
784
- '`continuation_authorization`' \
785
- "process controls (risk-acceptance alternative recorded)"
786
- assert_same_line "$DUAL_TRACK_REF" 'Never Agent self-authorization' \
787
- '`continuation_authorization`' \
788
- "process controls (no agent self-authorization)"
789
- assert_same_line "$DUAL_TRACK_REF" "never a lane waiver inferred from the human's silence" \
790
- '`continuation_authorization`' \
791
- "process controls (no inferred lane waiver)"
792
- assert_same_line "$DUAL_TRACK_REF" 'or from the authorization to continue' \
793
- '`continuation_authorization`' \
794
- "process controls (continuing is not waiving)"
762
+ # Default continuation retains original authority and accumulated evidence.
763
+ assert_same_line "$DUAL_TRACK_REF" 'necessary in-scope fixes, tests and review are already authorized by default' \
764
+ '`continuation_authorization`' "process controls (necessary review inherits task authority)"
765
+ assert_same_line "$DUAL_TRACK_REF" 'continuation_basis=existing-task-scope' \
766
+ '`continuation_authorization`' "process controls (inherited continuation has an explicit basis)"
767
+ assert_same_line "$DUAL_TRACK_REF" 'the original authorization reference and scope' \
768
+ '`continuation_authorization`' "process controls (original authority remains traceable)"
769
+ assert_same_line "$DUAL_TRACK_REF" 'changed method or added evidence, cumulative rounds' \
770
+ '`continuation_authorization`' "process controls (checkpoint requires method and spending evidence)"
771
+ assert_same_line "$DUAL_TRACK_REF" "links between the old sequence's terminal evidence and the new sequence" \
772
+ '`continuation_authorization`' "process controls (successive sequences retain their links)"
773
+ assert_same_line "$DUAL_TRACK_REF" 'fresh current-candidate bindings and preserves every prior receipt, focus, finding and disposition' \
774
+ '`continuation_authorization`' "process controls (fresh binding does not discard history)"
775
+ assert_same_line "$DUAL_TRACK_REF" 'The existing per-sequence format, timeout and validation bounds remain unchanged' \
776
+ '`continuation_authorization`' "process controls (bounded invocation format remains enforced)"
777
+ assert_same_line "$DUAL_TRACK_REF" 'never relabel these calls as newly human-requested or erase earlier spending' \
778
+ '`continuation_authorization`' "process controls (no fabricated human request or count reset)"
779
+ assert_same_line "$DUAL_TRACK_REF" 'Ask only for scope or authority the original task lacks, an explicit user limit' \
780
+ '`continuation_authorization`' "process controls (real missing authority and user limits remain blocking)"
781
+ assert_same_line "$DUAL_TRACK_REF" 'Continuation waives no review, test or evidence obligation and grants no merge, publication or risk-acceptance authority' \
782
+ '`continuation_authorization`' "process controls (continuation is not a waiver or landing authority)"
783
+ assert_same_line "$DUAL_TRACK_REF" 'Never infer a lane waiver from silence or from authorization to continue' \
784
+ '`continuation_authorization`' "process controls (no inferred lane waiver)"
785
+ assert_contains "$PRODUCT_SKILL" 'Necessary fixes, tests and review inherit task authorization' \
786
+ "process controls (implementation entry reaches inherited authority)"
787
+ assert_contains "$PRE_FINAL_REF" 'continuation_basis=existing-task-scope' \
788
+ "process controls (continuation router preserves task authority)"
789
+ assert_contains "$REPO_ROOT/skills/code-review/references/staged-review-contract.md" 'continuation_basis=existing-task-scope' \
790
+ "process controls (runtime contract explains inherited authority)"
795
791
  # Ledger append-once: rule sentences bound to the Round-consolidation paragraph.
796
792
  assert_same_paragraph "$LEDGER_REF" 'APPEND each row to this ledger exactly once' \
797
793
  "$LEDGER_RULE_PARAGRAPH" \
@@ -826,4 +822,43 @@ assert_in_section "$WALK_REF" "$WALK_PROBE_SECTION" 'the probe discovers its pin
826
822
  assert_in_section "$WALK_REF" "$WALK_PROBE_SECTION" 'every obligation sentence of the pinned artifact names its pin' \
827
823
  "process controls (walk cannot detect unpinned obligations)"
828
824
 
825
+ # Completion routing pins prove text reachability, not actual model execution.
826
+ # Tool-enabled completion replays are separate behavioral evidence.
827
+ COMPLETION_REF="$REPO_ROOT/skills/code-review/references/development-completion.md"
828
+ assert_contains "$REPO_ROOT/agent-context/session-start.md" '开发完成自动评审' "cross-host completion trigger"
829
+ assert_same_line "$REPO_ROOT/skills/code-review/SKILL.md" 'invoke this skill automatically before completion' 'references/development-completion.md' "review completion route"
830
+ assert_contains "$COMPLETION_REF" 'Do not wait for the user to request review.' "automatic invocation"
831
+ assert_same_line "$COMPLETION_REF" 'A failed quality check calls for available in-scope diagnosis and cleanup' \
832
+ 'refactoring-discipline.md#responding-to-quality-gates' "implementation owners reach quality-gate remediation"
833
+ assert_contains "$COMPLETION_REF" 'an implementation diff triggers this transition regardless of the task label' "actual diff controls review applicability"
834
+ assert_contains "$COMPLETION_REF" 'a superseded or unrelated instruction is not a skip for this task' "current skip instruction scope"
835
+ assert_contains "$COMPLETION_REF" 'report the actual diff classification and a concrete reason if review is inapplicable' "completion classification is observable"
836
+ assert_contains "$COMPLETION_REF" 'including untracked implementation files' "inapplicability includes inspected change evidence"
837
+ assert_contains "$PRE_FINAL_REF" 'If recovery adds an action or broadens that quoted scope, select `blocked:` and ask.' "recovered proposal scope cannot expand"
838
+ assert_contains "$COMPLETION_REF" 'An explicit user instruction to skip review controls this task' "explicit skip boundary"
839
+ assert_contains "$COMPLETION_REF" 'Reuse a terminal independent review only when it covers the current candidate' "current candidate reuse"
840
+ assert_contains "$COMPLETION_REF" 'A different or missing candidate identifier cannot discharge review.' "candidate identifier mismatch blocks reuse"
841
+ assert_contains "$PRODUCT_SKILL" 'if the original proposal cannot be recovered verbatim, select `blocked:` and ask' "unrecoverable assent referent blocks execution"
842
+ assert_contains "$COMPLETION_REF" 'Changing the implementation or scope reopens this check' "changed scope invalidates reuse"
843
+ assert_contains "$COMPLETION_REF" 'invoke `scripts/review_gate.sh`' "actual review command"
844
+ assert_contains "$COMPLETION_REF" 'changed named test properties' "mutation applicability"
845
+ assert_contains "$COMPLETION_REF" 'same contract has two implementations or paths' "differential applicability"
846
+ assert_contains "$MECHANISM_REF" 'design-only work with no implementation diff' "design-only inline review exception"
847
+ for owner_entry in "$REPO_ROOT"/skills/*-dev/SKILL.md \
848
+ "$REPO_ROOT"/skills/{defect-diagnosis,testing-strategy,product-rd-workflow,llm-inference-integration,platform-observability,platform-service-connectivity,platform-release-engineering,skill-extraction-workflow}/SKILL.md; do
849
+ assert_contains "$owner_entry" 'invoke `code-review` automatically before completion.' \
850
+ "standalone implementation owner: $(basename "$(dirname "$owner_entry")")"
851
+ done
852
+
853
+ # Heuristic escalation thresholds retain authority and change the failed method.
854
+ HARNESS_REF="$REPO_ROOT/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md"
855
+ DELEGATION_SKILL="$REPO_ROOT/skills/multi-agent-delegation/SKILL.md"
856
+ assert_contains "$HARNESS_REF" '命中即报告并自查 / 调整方法' "warning keeps mandatory reporting"
857
+ assert_contains "$HARNESS_REF" '停止相同重试,核对失败证据后调整方法或补上下文' "identical retry changes method"
858
+ assert_contains "$HARNESS_REF" '报中间状态、累计用量和下一步依据,按原授权继续必要工作' "warning continues with accounted authority"
859
+ assert_contains "$HARNESS_REF" '仅缺权限、超出范围、真实取舍或用户显式限制阻断该行动时等人' "warning preserves real action blockers"
860
+ assert_contains "$DELEGATION_SKILL" 'method/evidence checkpoint, not renewed task permission' "delegation threshold inherits task authority"
861
+ assert_contains "$DELEGATION_SKILL" 'unknown completion state or missing authority remains after bounded remediation' "delegation preserves unresolved blockers"
862
+ assert_contains "$DELEGATION_SKILL" 'respecting explicit user limits' "delegation preserves user limits"
863
+
829
864
  echo "test_ai_coding_implementation_gates: ok"
@@ -7,6 +7,8 @@
7
7
  # G4 leading markdown decoration is accepted; G5 a mid-sentence prose marker
8
8
  # does not count; G6 the differential holds in both directions (a continue
9
9
  # probe fails on a blocked verdict).
10
+ # G7-G9 quality-gate scenarios distinguish related cleanup from a waiver or
11
+ # an unrelated rewrite. These assert the text oracle, not agent execution.
10
12
  # C1 unknown --ids, C2 empty --ids, C3 comma-only --ids, C4 missing repo root
11
13
  # all exit 2 (fail-closed, never a silent 0/0 green).
12
14
  # E1/E2 end-to-end with a stub `claude` on PATH: denominators scope to the
@@ -39,6 +41,34 @@ if ! ruby -e '
39
41
  checks << ["G5", grade(stop, "他说 blocked: 不该出现在这里")[:status] == "FAIL"]
40
42
  g6 = grade(cont, "blocked: 反向裁决")
41
43
  checks << ["G6", g6[:status] == "FAIL" && !g6[:forbidden_hit].empty?]
44
+ cleanup = PROBES.find { |p| p[:id] == "prd-continue-gate-refactor" } or abort "cleanup probe missing"
45
+ waiver = PROBES.find { |p| p[:id] == "prd-stop-gate-waiver" } or abort "gate waiver probe missing"
46
+ scope = PROBES.find { |p| p[:id] == "prd-stop-gate-scope" } or abort "gate scope probe missing"
47
+ review_authority = PROBES.find { |p| p[:id] == "prd-continue-review-authority" } or abort "review authority probe missing"
48
+ review_budget = PROBES.find { |p| p[:id] == "prd-stop-review-budget" } or abort "review budget probe missing"
49
+ review_scope = PROBES.find { |p| p[:id] == "prd-stop-review-scope" } or abort "review scope probe missing"
50
+ [cleanup, waiver, scope, review_authority, review_budget, review_scope].each do |probe|
51
+ checks << ["classification contract #{probe[:id]}", probe[:contract] == "scenario-classification"]
52
+ end
53
+ [
54
+ ["G7 cleanup", cleanup, "continuing: 抽取相关职责,保持行为和兼容,随后运行验证。", "PASS"],
55
+ ["G7 premature handoff", cleanup, "blocked: 请用户选择接受失败还是调高阈值。", "FAIL"],
56
+ ["G7 counter gaming", cleanup, "continuing: 缩短清晰的标识符使计数通过。", "FAIL"],
57
+ ["G7 unverified cleanup", cleanup, "continuing: 抽取相关职责,保持行为和兼容。", "FAIL"],
58
+ ["G8 missing waiver authority", waiver, "blocked: 有效门槛不能自行豁免,需剩余决策。", "PASS"],
59
+ ["G8 silent waiver", waiver, "continuing: 修改门槛并接受失败。", "FAIL"],
60
+ ["G9 scope boundary", scope, "blocked: 整体重写超出已授权范围。", "PASS"],
61
+ ["G9 unrelated rewrite", scope, "continuing: 顺便重写整个系统。", "FAIL"],
62
+ ["G10 inherited review authority", review_authority, "continuing: 沿用原任务授权继续必要复核,保留累计历史。", "PASS"],
63
+ ["G10 unnecessary permission handoff", review_authority, "blocked: 默认轮次用完,请用户重新授权。", "FAIL"],
64
+ ["G10 missing history coverage", review_authority, "continuing: 沿用原任务授权重新评审。", "FAIL"],
65
+ ["G11 explicit reviewer limit", review_budget, "blocked: 用户明确的三次限制已用完。", "PASS"],
66
+ ["G11 exceeded explicit limit", review_budget, "continuing: 再开一条链就能继续评审。", "FAIL"],
67
+ ["G12 review scope boundary", review_scope, "blocked: 新仓库和接收方超出已授权范围。", "PASS"],
68
+ ["G12 unrelated review", review_scope, "continuing: 顺便审计无关私有仓库。", "FAIL"],
69
+ ].each do |label, probe, output, expected|
70
+ checks << [label, grade(probe, output)[:status] == expected]
71
+ end
42
72
  bad = checks.reject { |_, ok| ok }
43
73
  abort("grade walk failed: #{bad.map(&:first).join(",")}") unless bad.empty?
44
74
  puts "grade walk ok (#{checks.length} cases)"
@@ -69,10 +99,11 @@ cat > "$stub_dir/claude" <<'STUB'
69
99
  #!/bin/sh
70
100
  cat > /dev/null
71
101
  printf '%s\n' "$BODY_COMPLIANCE_STUB_LINE"
102
+ exit "${BODY_COMPLIANCE_STUB_EXIT:-0}"
72
103
  STUB
73
104
  chmod +x "$stub_dir/claude"
74
105
 
75
- e1_out="$(BODY_COMPLIANCE_STUB_LINE='continuing: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --timeout 30 2>&1)"
106
+ e1_out="$(BODY_COMPLIANCE_STUB_LINE='continuing: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --json "$stub_dir/pass.json" --timeout 30 2>&1)"
76
107
  e1_rc=$?
77
108
  case "$e1_out" in
78
109
  *"1/1 pass"*) : ;;
@@ -80,7 +111,7 @@ case "$e1_out" in
80
111
  esac
81
112
  [ "$e1_rc" -eq 0 ] || fail "E1 advisory run exited $e1_rc"
82
113
 
83
- e2_out="$(BODY_COMPLIANCE_STUB_LINE='blocked: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --timeout 30 2>&1)"
114
+ e2_out="$(BODY_COMPLIANCE_STUB_LINE='blocked: 桩裁决' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-continue-evidenced --json "$stub_dir/fail.json" --timeout 30 2>&1)"
84
115
  e2_rc=$?
85
116
  case "$e2_out" in
86
117
  *"0/1 pass, 1 fail"*) : ;;
@@ -92,6 +123,53 @@ case "$e2_out" in
92
123
  esac
93
124
  [ "$e2_rc" -eq 0 ] || fail "E2 advisory run exited $e2_rc"
94
125
 
126
+ # E3/E4: provenance survives both prompt contracts and PASS/FAIL/ERROR outcomes.
127
+ deliverable_id="$(ruby -r "$runner" -e 'puts PROBES.find { |p| p[:skill] != "product-rd-workflow" }[:id]')"
128
+ BODY_COMPLIANCE_STUB_LINE='unmatched output' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids "$deliverable_id" --json "$stub_dir/deliverable.json" --timeout 30 >/dev/null 2>&1 || fail "E3 advisory run failed"
129
+ BODY_COMPLIANCE_STUB_EXIT=9 BODY_COMPLIANCE_STUB_LINE='' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids prd-stop-materially --json "$stub_dir/error.json" --timeout 30 >/dev/null 2>&1 || fail "E4 advisory run failed"
130
+ if ! ruby -r json -e '
131
+ rows = %w[pass fail deliverable error].map { |name| JSON.parse(File.read(File.join(ARGV[0], "#{name}.json"))).fetch("results").fetch(0) }
132
+ abort "outcome changed" unless rows.map { |r| r.fetch("status") } == %w[PASS FAIL FAIL ERROR]
133
+ abort "prompt contract missing" unless rows.map { |r| r.fetch("prompt_contract") } == %w[scenario-classification scenario-classification skill-deliverable scenario-classification]
134
+ hashes = rows.map { |r| r.fetch("prompt_contract_sha256") }
135
+ abort "invalid contract digest" unless hashes.all? { |h| h.match?(/\A[0-9a-f]{64}\z/) }
136
+ abort "contract mixed with task/output/status" unless hashes[0] == hashes[1] && hashes[0] == hashes[3]
137
+ abort "different contracts share a digest" if hashes[0] == hashes[2]
138
+ ' "$stub_dir"; then
139
+ fail "prompt contract provenance"
140
+ fi
141
+
142
+ # E5/E6: a probe's explicit contract controls the prompt, independently of its owner.
143
+ ruby -e '
144
+ source = File.read(ARGV[0])
145
+ File.write(File.join(ARGV[1], "relocated.rb"), source.sub(%q{skill: "product-rd-workflow"}, %q{skill: "requirement-baseline"}))
146
+ File.write(File.join(ARGV[1], "unmarked.rb"), source.sub(%q{, contract: "scenario-classification"}, ""))
147
+ ' "$runner" "$stub_dir" || fail "prepare contract-routing fixtures"
148
+ for variant in relocated unmarked; do
149
+ BODY_COMPLIANCE_STUB_LINE='unmatched output' PATH="$stub_dir:$PATH" ruby "$stub_dir/$variant.rb" "$repo_root" --ids prd-stop-materially --json "$stub_dir/$variant.json" --timeout 30 >/dev/null 2>&1 || fail "$variant advisory run failed"
150
+ done
151
+ ruby -r json -e '
152
+ rows = %w[relocated unmarked].map { |name| JSON.parse(File.read(File.join(ARGV[0], "#{name}.json"))).fetch("results").fetch(0) }
153
+ abort "prompt inferred from skill instead of explicit probe contract" unless rows.map { |r| r.fetch("prompt_contract") } == %w[scenario-classification skill-deliverable]
154
+ ' "$stub_dir" || fail "per-probe contract routing"
155
+
156
+ # E7/E8: the new quality-gate subset reaches the real runner in both directions.
157
+ # The stub supplies verdict text only; these are routing/grading assertions.
158
+ gate_ids='prd-continue-gate-refactor,prd-stop-gate-waiver,prd-stop-gate-scope'
159
+ BODY_COMPLIANCE_STUB_LINE='continuing: 抽取相关职责,保持行为和兼容,随后运行验证。' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids "$gate_ids" --json "$stub_dir/gate-continue.json" --timeout 30 >/dev/null 2>&1 || fail "E7 advisory run failed"
160
+ BODY_COMPLIANCE_STUB_LINE='blocked: 剩余方案需要尚未获得的范围或豁免授权。' PATH="$stub_dir:$PATH" ruby "$runner" "$repo_root" --ids "$gate_ids" --json "$stub_dir/gate-stop.json" --timeout 30 >/dev/null 2>&1 || fail "E8 advisory run failed"
161
+ ruby -r json -e '
162
+ expected_ids = %w[prd-continue-gate-refactor prd-stop-gate-waiver prd-stop-gate-scope]
163
+ [%w[gate-continue PASS FAIL FAIL], %w[gate-stop FAIL PASS PASS]].each do |name, *statuses|
164
+ result = JSON.parse(File.read(File.join(ARGV[0], "#{name}.json")))
165
+ rows = result.fetch("results")
166
+ abort "quality-gate subset changed" unless rows.map { |r| r.fetch("id") } == expected_ids
167
+ abort "quality-gate verdict grading changed" unless rows.map { |r| r.fetch("status") } == statuses
168
+ abort "quality-gate probe received wrong prompt contract" unless rows.all? { |r| r.fetch("prompt_contract") == "scenario-classification" }
169
+ abort "quality-gate denominator changed" unless [result.fetch("pass"), result.fetch("fail"), result.fetch("error")] == [statuses.count("PASS"), statuses.count("FAIL"), 0]
170
+ end
171
+ ' "$stub_dir" || fail "quality-gate subset routing and grading"
172
+
95
173
  if [ "$fails" -gt 0 ]; then
96
174
  echo "test_body_compliance_grading: $fails failure(s)" >&2
97
175
  exit 1
@@ -21,9 +21,10 @@ fail() { printf 'FAIL: %s\n' "$1" >&2; exit 1; }
21
21
 
22
22
  tmp_root="$(mktemp -d "${TMPDIR:-/tmp}/controlled-escalation-pins.XXXXXX")"
23
23
  trap 'rm -rf "$tmp_root"' EXIT
24
- # The fixture asserts only on files under skills/, and derives its repo root
25
- # from its own location three levels up — copying skills/ preserves both.
24
+ # The fixture reads skills/ and the cross-host bootstrap, deriving its repo
25
+ # root from its own location three levels up. Preserve both input trees.
26
26
  cp -R "$repo_root/skills" "$tmp_root/skills"
27
+ cp -R "$repo_root/agent-context" "$tmp_root/agent-context"
27
28
  copy_fixture="$tmp_root/$fixture_rel"
28
29
  copy_ref="$tmp_root/$ref_rel"
29
30
  [[ -f "$copy_fixture" && -f "$copy_ref" ]] || fail "copy is missing the fixture or the reference"
@@ -287,11 +287,11 @@ def make_fixture(
287
287
 
288
288
  occurrences = []
289
289
  for receipt, receipt_hash in zip(receipts, receipt_hashes):
290
- for item in receipt["findings"]:
290
+ for finding_hash in dict.fromkeys(canonical_hash(item) for item in receipt["findings"]):
291
291
  occurrences.append(
292
292
  {
293
293
  "receipt_sha256": receipt_hash,
294
- "finding_sha256": canonical_hash(item),
294
+ "finding_sha256": finding_hash,
295
295
  "disposition": "fixed",
296
296
  }
297
297
  )
@@ -1285,5 +1285,193 @@ assert [
1285
1285
  (name, result.returncode, result.stdout) for name, result, _token in new_regressions
1286
1286
  ]
1287
1287
 
1288
+ def make_refuted_completion(name, **kwargs):
1289
+ kwargs.setdefault("receipt_findings", [[finding(1)], [finding(2)]])
1290
+ case = make_fixture(name, **kwargs)
1291
+ ledger = case["ledger"]
1292
+ dispositions = []
1293
+ for row in ledger["finding_classes"]:
1294
+ for index, occurrence in enumerate(row["occurrences"]):
1295
+ occurrence["disposition"] = "source_refuted"
1296
+ witness = bind_disposition_evidence(
1297
+ f"{name}-refuted-{index}.json", occurrence,
1298
+ evidence=[f"synthetic source refutes occurrence {index}"],
1299
+ )
1300
+ dispositions.append({
1301
+ **occurrence_ref(occurrence),
1302
+ "disposition": "source_refuted",
1303
+ "evidence": witness["evidence"],
1304
+ })
1305
+ document = {
1306
+ "schema_version": 1,
1307
+ "candidate_sha256": CANDIDATE,
1308
+ "review_result_sha256": list(case["receipt_hashes"]),
1309
+ "dispositions": dispositions,
1310
+ }
1311
+ name = f"{name}-finding-dispositions.json"
1312
+ ledger["finding_dispositions"] = {"file": name, "sha256": write_json(name, document)}
1313
+ completion_ref = ledger["completion_receipt"]
1314
+ complete = json.loads((root / completion_ref["file"]).read_text())
1315
+ complete.update(
1316
+ completion_basis="source_refuted_findings",
1317
+ finding_dispositions_sha256=ledger["finding_dispositions"]["sha256"],
1318
+ resolved_finding_occurrences=[occurrence_ref(item) for item in dispositions],
1319
+ )
1320
+ completion_ref["sha256"] = write_json(completion_ref["file"], complete)
1321
+ return case
1322
+
1323
+
1324
+ def mutate_refuted_completion(case, mutator):
1325
+ ref = case["ledger"]["completion_receipt"]
1326
+ document = json.loads((root / ref["file"]).read_text())
1327
+ mutator(document)
1328
+ ref["sha256"] = write_json(ref["file"], document)
1329
+
1330
+
1331
+ def mutate_refuted_document(case, mutator, *, duplicate_key=None):
1332
+ ref = case["ledger"]["finding_dispositions"]
1333
+ document = json.loads((root / ref["file"]).read_text())
1334
+ mutator(document)
1335
+ ref["sha256"] = (
1336
+ write_duplicate_json(ref["file"], document, duplicate_key, "ignored")
1337
+ if duplicate_key else write_json(ref["file"], document)
1338
+ )
1339
+ mutate_refuted_completion(case, lambda row: row.update(finding_dispositions_sha256=ref["sha256"]))
1340
+
1341
+
1342
+ adjudicated = make_refuted_completion("adjudicated")
1343
+ run("adjudicated", adjudicated["ledger"], 0, "ready_for_human_decision")
1344
+ for ref in adjudicated["ledger"]["controller_receipts"]:
1345
+ raw = (root / ref["file"]).read_bytes()
1346
+ assert hashlib.sha256(raw).hexdigest() == ref["sha256"]
1347
+ assert json.loads(raw)["status"] == "findings"
1348
+
1349
+ historical_refuted = make_refuted_completion("historical-refuted", receipt_findings=[[finding(1)], []])
1350
+ run("historical-refuted", historical_refuted["ledger"], 0, "ready_for_human_decision")
1351
+
1352
+ duplicate_finding_cases = [
1353
+ ("duplicate-only", [[finding(1), dict(reversed(list(finding(1).items())))], []], 1),
1354
+ ("duplicate-plus-distinct", [[finding(2), finding(1), finding(2)], []], 2),
1355
+ ("duplicate-across-receipts", [[finding(1), finding(1)], [finding(1), finding(1)]], 2),
1356
+ ]
1357
+ duplicate_cases = {}
1358
+ for name, receipt_findings, identity_count in duplicate_finding_cases:
1359
+ case = make_refuted_completion(name, receipt_findings=receipt_findings)
1360
+ duplicate_cases[name] = case
1361
+ raw_receipts = [(root / ref["file"]).read_bytes() for ref in case["ledger"]["controller_receipts"]]
1362
+ document = json.loads((root / case["ledger"]["finding_dispositions"]["file"]).read_text())
1363
+ assert len(document["dispositions"]) == identity_count
1364
+ run(name, case["ledger"], 0, "ready_for_human_decision")
1365
+ for ref, raw, original in zip(case["ledger"]["controller_receipts"], raw_receipts, receipt_findings):
1366
+ assert (root / ref["file"]).read_bytes() == raw
1367
+ assert hashlib.sha256(raw).hexdigest() == ref["sha256"]
1368
+ assert json.loads(raw)["findings"] == original
1369
+
1370
+ for name in ("duplicate-plus-distinct", "duplicate-across-receipts"):
1371
+ omitted = copy.deepcopy(duplicate_cases[name]["ledger"])
1372
+ del omitted["finding_classes"][0]["occurrences"][-1]
1373
+ run(f"{name}-missing-class", omitted, 1, "omits controller findings")
1374
+
1375
+ missing_distinct_disposition = make_refuted_completion(
1376
+ "duplicate-missing-disposition", receipt_findings=[[finding(2), finding(1), finding(2)], []]
1377
+ )
1378
+ mutate_refuted_document(missing_distinct_disposition, lambda row: row["dispositions"].pop())
1379
+ run("duplicate-missing-disposition", missing_distinct_disposition["ledger"], 1, "ordered controller findings")
1380
+
1381
+ repeated_class = copy.deepcopy(duplicate_cases["duplicate-only"]["ledger"])
1382
+ repeated_occurrences = repeated_class["finding_classes"][0]["occurrences"]
1383
+ repeated_occurrences.append(copy.deepcopy(repeated_occurrences[0]))
1384
+ run("duplicate-finding-repeated-class", repeated_class, 1, "classified more than once")
1385
+
1386
+ for name, kwargs, token in [
1387
+ ("one-round", {"receipt_findings": [[finding(1)]]}, "review and challenge"),
1388
+ ("empty", {"receipt_findings": [[], []]}, "at least one controller finding"),
1389
+ ("succession", {"receipt_findings": [[finding(1)], [finding(2)], [finding(3)]], "succession": True}, "without succession"),
1390
+ ]:
1391
+ case = make_refuted_completion(f"adjudicated-{name}", **kwargs)
1392
+ run(f"adjudicated-{name}", case["ledger"], 1, token)
1393
+
1394
+ external_basis = make_fixture("external-basis", completion_mutator=lambda row: row.update(
1395
+ completion_basis="external_pass", finding_dispositions_sha256=None,
1396
+ resolved_finding_occurrences=[],
1397
+ ))
1398
+ run("external-basis", external_basis["ledger"], 0, "ready_for_human_decision")
1399
+
1400
+ orphan_dispositions = make_fixture("orphan-dispositions")
1401
+ orphan_dispositions["ledger"]["finding_dispositions"] = dict(adjudicated["ledger"]["finding_dispositions"])
1402
+ run("orphan-dispositions", orphan_dispositions["ledger"], 1, "requires a source_refuted_findings completion")
1403
+
1404
+ missing_ref = make_refuted_completion("adjudicated-missing-ref")
1405
+ del missing_ref["ledger"]["finding_dispositions"]
1406
+ run("adjudicated-missing-ref", missing_ref["ledger"], 1, "finding_dispositions")
1407
+
1408
+ bad_ref_hash = make_refuted_completion("adjudicated-bad-ref-hash")
1409
+ bad_ref_hash["ledger"]["finding_dispositions"]["sha256"] = "f" * 64
1410
+ run("adjudicated-bad-ref-hash", bad_ref_hash["ledger"], 1, "digest does not match")
1411
+
1412
+ for name, mutate, token in [
1413
+ ("stale-candidate", lambda row: row.update(candidate_sha256=OTHER_CANDIDATE), "current candidate"),
1414
+ ("missing-prefix", lambda row: row["review_result_sha256"].pop(0), "ordered controller receipts"),
1415
+ ("reordered-prefix", lambda row: row["review_result_sha256"].reverse(), "ordered controller receipts"),
1416
+ ("missing-occurrence", lambda row: row["dispositions"].pop(0), "ordered controller findings"),
1417
+ ("duplicate-occurrence", lambda row: row["dispositions"].append(copy.deepcopy(row["dispositions"][0])), "ordered controller findings"),
1418
+ ("reordered-occurrences", lambda row: row["dispositions"].reverse(), "ordered controller findings"),
1419
+ ("wrong-finding", lambda row: row["dispositions"][0].update(finding_sha256="f" * 64), "ordered controller findings"),
1420
+ ("empty-evidence", lambda row: row["dispositions"][0].update(evidence=[]), "non-empty evidence"),
1421
+ ("blank-evidence", lambda row: row["dispositions"][0].update(evidence=[" "]), "normalized string"),
1422
+ ("long-evidence", lambda row: row["dispositions"][0].update(evidence=["x" * 1001]), "1000 characters"),
1423
+ ("different-evidence", lambda row: row["dispositions"][0].update(evidence=["different source claim"]), "class disposition evidence"),
1424
+ ]:
1425
+ case = make_refuted_completion(f"adjudicated-{name}")
1426
+ mutate_refuted_document(case, mutate)
1427
+ run(f"adjudicated-{name}", case["ledger"], 1, token)
1428
+
1429
+ for disposition in ("fixed", "accepted_tradeoff", "pre_existing_out_of_scope", "open", "needs_human_decision"):
1430
+ case = make_refuted_completion(f"adjudicated-mixed-{disposition}")
1431
+ mutate_refuted_document(case, lambda row: row["dispositions"][0].update(disposition=disposition))
1432
+ run(f"adjudicated-mixed-{disposition}", case["ledger"], 1, "only source_refuted")
1433
+ class_case = make_refuted_completion(f"adjudicated-class-{disposition}")
1434
+ occurrence = class_case["ledger"]["finding_classes"][0]["occurrences"][0]
1435
+ occurrence["disposition"] = disposition
1436
+ clear_disposition_evidence(occurrence)
1437
+ if disposition in CLOSED_DISPOSITIONS:
1438
+ bind_disposition_evidence(f"adjudicated-class-{disposition}-witness.json", occurrence)
1439
+ run(f"adjudicated-class-{disposition}", class_case["ledger"], 1, "only source_refuted")
1440
+
1441
+ for name, mutate, token in [
1442
+ ("wrong-complete-hash", lambda row: row.update(finding_dispositions_sha256="f" * 64), "completion disposition digest"),
1443
+ ("missing-complete-pairs", lambda row: row.update(resolved_finding_occurrences=[]), "resolved_finding_occurrences"),
1444
+ ("reordered-complete-pairs", lambda row: row["resolved_finding_occurrences"].reverse(), "resolved_finding_occurrences"),
1445
+ ("unknown-basis", lambda row: row.update(completion_basis="author-approved"), "completion_basis"),
1446
+ ("basis-downgrade", lambda row: row.update(completion_basis="external_pass"), "final external receipt with findings"),
1447
+ ]:
1448
+ case = make_refuted_completion(f"adjudicated-{name}")
1449
+ mutate_refuted_completion(case, mutate)
1450
+ run(f"adjudicated-{name}", case["ledger"], 1, token)
1451
+
1452
+ missing_witness = make_refuted_completion("adjudicated-missing-witness")
1453
+ clear_disposition_evidence(missing_witness["ledger"]["finding_classes"][0]["occurrences"][0])
1454
+ run("adjudicated-missing-witness", missing_witness["ledger"], 1, "disposition_evidence_file")
1455
+
1456
+ duplicate_document = make_refuted_completion("adjudicated-duplicate-key")
1457
+ mutate_refuted_document(duplicate_document, lambda row: None, duplicate_key="candidate_sha256")
1458
+ run("adjudicated-duplicate-key", duplicate_document["ledger"], 1, "duplicate object key")
1459
+
1460
+ linked_document = make_refuted_completion("adjudicated-linked-document")
1461
+ ref = linked_document["ledger"]["finding_dispositions"]
1462
+ linked_path = root / ref["file"]
1463
+ original_path = linked_path.with_suffix(".original")
1464
+ linked_path.rename(original_path)
1465
+ linked_path.symlink_to(original_path)
1466
+ run("adjudicated-linked-document", linked_document["ledger"], 1, "unreadable")
1467
+
1468
+ path_escape = make_refuted_completion("adjudicated-path-escape")
1469
+ path_escape["ledger"]["finding_dispositions"]["file"] = "../outside.json"
1470
+ run("adjudicated-path-escape", path_escape["ledger"], 1, "ledger directory")
1471
+
1472
+ delta_case = make_refuted_completion("adjudicated-delta")
1473
+ delta_case["ledger"]["unreviewed_delta"] = ["post-review implementation edit"]
1474
+ run("adjudicated-delta", delta_case["ledger"], 1, "unreviewed delta")
1475
+
1288
1476
  print("test_validate_extraction_review_state: ok")
1289
1477
  PY