@ccoalm/ccl-skills 0.14.0 → 0.15.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +80 -4
  2. package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +16 -4
  3. package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +13 -2
  4. package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +53 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +17 -3
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +65 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +5 -5
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +8 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +14 -14
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +1 -1
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +40 -83
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +2 -2
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +44 -37
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +1 -1
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +9 -6
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +6 -0
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +4 -4
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +52 -0
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +31 -7
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +188 -13
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +36 -13
  44. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +103 -21
  45. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +261 -14
  46. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +2 -0
  47. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
  48. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_runtime.py +428 -0
  49. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
  50. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +63 -5
  51. package/dist/assets/release.json +76 -56
  52. package/dist/claude-adapter.js +14 -7
  53. package/dist/codex-host.d.ts +1 -3
  54. package/dist/codex-host.js +6 -9
  55. package/dist/host-probe.d.ts +11 -0
  56. package/dist/host-probe.js +27 -0
  57. package/dist/opencode-adapter.js +24 -19
  58. package/dist/unified.js +11 -9
  59. package/package.json +1 -1
@@ -5,7 +5,7 @@
5
5
  # upstream-owner skill must declare a behavioral-evidence status and an
6
6
  # observed-failure state, and (for non-wording changes) name an owner-scoped
7
7
  # FIRING PATH that resolves to this diff — an anchor on a changed normative
8
- # rule line, or a changed owner executable. The statuses are required author
8
+ # rule or table data line, or a changed owner executable. The statuses are required author
9
9
  # declarations; the firing path and the wording-only classification are the
10
10
  # machine-verified core. Extracted from the former inline `ruby -e` block in
11
11
  # check-ccl-skills.sh so the program gets normal Ruby tooling and no
@@ -52,19 +52,123 @@ LEDGER_PATH = "skills/skill-extraction-workflow/references/source-register.md"
52
52
  # the routing-surface class below was one such patch) only re-instantiates it on
53
53
  # the next input. The predicate now reads the round.
54
54
  #
55
- # Rounds are cut at the commits that touch the ledger itself, walked first-parent
56
- # so one merged worktree round collapses to one boundary. The partition comes from
57
- # git alone: an author cannot widen, move, or nominate their own scope.
58
- round_heads = git_read.call("rev-list", "--first-parent", "--reverse", "#{base_ref}..HEAD", "--", LEDGER_PATH)
59
- .split("\n").map(&:strip).reject(&:empty?)
60
- # Each round spans (previous ledger boundary, this one] so the work commits that
55
+ # Rounds are cut at the commits that touch the ledger itself, walked along a
56
+ # first-parent line. The partition comes from git alone: an author cannot widen,
57
+ # move, or nominate their own scope.
58
+ #
59
+ # THE SAME PARTITION BEFORE AND AFTER THE MERGE. A branch is judged on its own
60
+ # first-parent line while it is a pull request (CI checks out the branch head),
61
+ # and once merged that whole line sits behind ONE first-parent step of the
62
+ # integration branch. Reading that step as one boundary gave the same history a
63
+ # different partition after it landed — every round on the branch collapsed into
64
+ # one, and a row whose validity depends on its round being narrow (a routing-
65
+ # surface `#description` anchor in a commit that changed nothing else) turned red
66
+ # without a byte of it changing. That is the "verdict moved after it landed"
67
+ # defect in a new coat, and it surfaced on every post-merge evaluation: the push
68
+ # build of the integration branch and the promotion pull request. So a merge that
69
+ # git itself reproduces from its two parents is EXPANDED in place: its second
70
+ # parent's line, from the fork point to the merged head, is walked with the same
71
+ # rule, recursively, and contributes exactly the rounds it had as a branch.
72
+ #
73
+ # A merge is expanded only when git can rebuild it — two parents, a tree equal to
74
+ # `git merge-tree --write-tree` of those parents, and a second parent that is
75
+ # neither already on the base nor already on the line being walked (a sync merge
76
+ # brings nothing that needs a round). Anything else — a hand-resolved merge, a
77
+ # conflicted one, an octopus — keeps today's single boundary at the merge, so
78
+ # content git did not derive from the parents is never left in no round.
79
+ ROUND_WALK_MAX_DEPTH = 8
80
+ ancestor_of = lambda do |commit, tip|
81
+ IO.popen(["git", "-C", root, "merge-base", "--is-ancestor", commit, tip], err: File::NULL, &:read)
82
+ status = $?.exitstatus
83
+ next true if status == 0
84
+ next false if status == 1
85
+ warn "impact_chain_git_failed: git merge-base --is-ancestor #{commit} #{tip} exited #{status}"
86
+ exit 1
87
+ end
88
+ # The tree git produces merging `second` into `first`; nil when that merge
89
+ # conflicts (whoever resolved it was not git). Needs git 2.38+, the same floor
90
+ # the review-ledger binder already requires for the identical invariant.
91
+ automatic_merge_tree = lambda do |first, second|
92
+ out = IO.popen(["git", "-C", root, "merge-tree", "--write-tree", first, second], err: File::NULL, &:read)
93
+ status = $?.exitstatus
94
+ next out.to_s.lines.first.to_s.strip if status == 0
95
+ next nil if status == 1
96
+ warn "impact_chain_git_failed: git merge-tree --write-tree #{first} #{second} exited #{status} (git 2.38 or newer is required)"
97
+ exit 1
98
+ end
99
+ # [first parent, second parent, fork point] when `commit` is a merge git can
100
+ # rebuild from its parents and whose second parent carries a line of its own;
101
+ # nil when the merge keeps today's single-boundary treatment. `line_base` is the
102
+ # base of the line being walked: a merge whose second parent is already below
103
+ # that base is a sync of what the line was cut from (the target advancing under
104
+ # a branch), and is a sync on the branch's own line exactly as it is on the
105
+ # integration line — judging it against the outer base alone would expand it
106
+ # during promotion and strand the branch's earlier work in a rowless span.
107
+ expandable_merge = lambda do |commit, line_base|
108
+ parents = git_read.call("rev-list", "--parents", "-n", "1", commit).split[1..] || []
109
+ next nil unless parents.length == 2
110
+ first, second = parents
111
+ next nil if ancestor_of.call(second, base_ref) || ancestor_of.call(second, line_base) || ancestor_of.call(second, first)
112
+ own_tree = git_read.call("rev-parse", "#{commit}^{tree}").strip
113
+ next nil unless automatic_merge_tree.call(first, second) == own_tree
114
+ # The fork point read FAILS CLOSED like every other git read here: a lookup
115
+ # that errored would otherwise read as "no fork point", skip the expansion,
116
+ # and hand the merge the collapsed span — the lenient verdict — on a git
117
+ # failure nobody sees. Exit 1 is git's own "no common ancestor" and means
118
+ # there is genuinely no line to expand from.
119
+ fork_out = IO.popen(["git", "-C", root, "merge-base", first, second], err: File::NULL, &:read)
120
+ fork_status = $?.exitstatus
121
+ next nil if fork_status == 1
122
+ unless fork_status == 0
123
+ warn "impact_chain_git_failed: git merge-base #{first} #{second} exited #{fork_status}"
124
+ exit 1
125
+ end
126
+ fork = fork_out.to_s.split("\n").first.to_s.strip
127
+ next nil if fork.empty?
128
+ [first, second, fork]
129
+ end
130
+ # Each round spans (previous boundary, this one] so the work commits that
61
131
  # precede a ledger append are inside the round they belong to — landing the change
62
132
  # and appending the row in separate commits is the normal shape, not an evasion.
63
- round_bounds = ([base_ref] + round_heads).each_cons(2).to_a
64
- # The trailing span — owner changes committed after the last ledger append — is a
133
+ # The trailing span — owner changes committed after the last boundary — is a
65
134
  # round too. It holds no rows, so its owners fall through to the presence check
66
- # and the gate still fails closed on undeclared work.
67
- round_bounds << [round_heads.last || base_ref, "HEAD"]
135
+ # and the gate still fails closed on undeclared work. Both hold on every line the
136
+ # walk visits, the integration branch and each expanded merge alike.
137
+ round_bounds_for = lambda do |from, to, depth|
138
+ if depth > ROUND_WALK_MAX_DEPTH
139
+ warn "impact_chain_round_walk_too_deep: merges nested more than #{ROUND_WALK_MAX_DEPTH} levels between #{from} and #{to}"
140
+ exit 1
141
+ end
142
+ list = lambda do |*options, pathspec|
143
+ git_read.call("rev-list", "--first-parent", "--reverse", *options, "#{from}..#{to}", *pathspec)
144
+ .split("\n").map(&:strip).reject(&:empty?)
145
+ end
146
+ line = list.call([])
147
+ ledger_heads = list.call(["--", LEDGER_PATH])
148
+ merges = list.call("--merges", [])
149
+ spans = []
150
+ prev = from
151
+ line.each do |commit|
152
+ expansion = merges.include?(commit) ? expandable_merge.call(commit, from) : nil
153
+ if expansion
154
+ first, second, fork = expansion
155
+ spans << [prev, first] unless prev == first
156
+ spans.concat(round_bounds_for.call(fork, second, depth + 1))
157
+ prev = commit
158
+ elsif ledger_heads.include?(commit)
159
+ spans << [prev, commit]
160
+ prev = commit
161
+ end
162
+ end
163
+ spans << [prev, to]
164
+ spans
165
+ end
166
+ round_bounds = round_bounds_for.call(base_ref, "HEAD", 0)
167
+ # Diagnostic only: print the partition so a verdict can be read against the
168
+ # rounds it was judged in. Off by default so no suite's output assertions move.
169
+ if ENV["CCL_IMPACT_CHAIN_TRACE_ROUNDS"] == "1"
170
+ round_bounds.each { |span_base, span_head| warn "impact_chain_round: #{span_base[0, 12]}..#{span_head[0, 12]}" }
171
+ end
68
172
  # Everything a predicate needs to judge one span. Built lazily per span and
69
173
  # memoized: a round whose rows are all RED-baseline never pays for the rename
70
174
  # derivation.
@@ -1037,9 +1141,80 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
1037
1141
  blob[:content].scan(Regexp.new(Regexp.escape(anchor))).length == 1
1038
1142
  end
1039
1143
  end
1144
+ # Definition/decision tables are firing surfaces too. Recognize explicit
1145
+ # Markdown tables with a header and delimiter; comments and code examples do
1146
+ # not count. This proves location/shape, not the truth of the definition.
1147
+ table_data_anchor_valid = lambda do |scope, parts|
1148
+ prior = blob_at.call(scope.base, parts[:path])
1149
+ next false if prior && prior[:content].include?(parts[:anchor])
1150
+ previous_cells = nil
1151
+ columns = nil
1152
+ fence = nil
1153
+ comment = false
1154
+ raw_html = nil
1155
+ html_block = false
1156
+ head_blob.call(scope, parts[:path])[:content].each_line do |raw_line|
1157
+ line = raw_line.chomp
1158
+ if fence
1159
+ fence = nil if line.match?(/\A {0,3}#{Regexp.escape(fence[0])}{#{fence.length},}\s*\z/)
1160
+ next
1161
+ end
1162
+ if raw_html
1163
+ raw_html = nil if line.match?(raw_html)
1164
+ next
1165
+ end
1166
+ if html_block
1167
+ html_block = false if line.strip.empty?
1168
+ next
1169
+ end
1170
+ hidden = comment || line.include?("<!--") || line.include?("-->")
1171
+ line.scan(/<!--|-->/).each { |marker| comment = marker == "<!--" }
1172
+ if hidden
1173
+ previous_cells = columns = nil
1174
+ next
1175
+ end
1176
+ if (opening = line.match(/\A {0,3}(`{3,}|~{3,})/))
1177
+ fence = opening[1]
1178
+ previous_cells = columns = nil
1179
+ next
1180
+ end
1181
+ terminator = case line
1182
+ when /\A {0,3}<\?/ then /\?>/
1183
+ when /\A {0,3}<!\[CDATA\[/ then /\]\]>/
1184
+ when /\A {0,3}<![A-Z]/ then />/
1185
+ end
1186
+ if terminator
1187
+ raw_html = terminator unless line.match?(terminator)
1188
+ previous_cells = columns = nil
1189
+ next
1190
+ end
1191
+ if line.match?(%r{\A {0,3}</?[A-Za-z][\w-]*(?:\s|>|/|\z)})
1192
+ tag = line[/\A {0,3}<(script|pre|style|textarea)(?:\s|>|\z)/i, 1]
1193
+ raw_html = %r{</#{tag}\s*>}i if tag && !line.match?(%r{</#{tag}\s*>}i)
1194
+ html_block = !tag
1195
+ previous_cells = columns = nil
1196
+ next
1197
+ end
1198
+ unless line.match?(/\A {0,3}\|.*\|\s*\z/)
1199
+ previous_cells = columns = nil
1200
+ next
1201
+ end
1202
+ cells = line.strip[1...-1].split(/(?<!\\)\|/, -1).map(&:strip)
1203
+ delimiter = cells.length >= 2 && cells.all? { |cell| cell.match?(/\A:?-{3,}:?\z/) }
1204
+ if delimiter
1205
+ columns = previous_cells && previous_cells.length == cells.length ? cells.length : nil
1206
+ elsif columns && columns == cells.length
1207
+ break true if cells.any? { |cell| cell.include?(parts[:anchor]) }
1208
+ else
1209
+ columns = nil
1210
+ end
1211
+ previous_cells = cells
1212
+ end == true
1213
+ end
1040
1214
  enforcing_file_locator_valid = lambda do |scope, parts|
1041
1215
  next false unless parts && parts[:kind] == "file"
1042
1216
  next false unless parts[:path].end_with?(".md")
1217
+ next false if parts[:path] == LEDGER_PATH # Evidence cannot certify itself.
1043
1218
  next false unless locator_valid.call(scope, "file:#{parts[:path]}##{parts[:anchor]}")
1044
1219
  line = added_lines_for.call(scope, parts[:path]).find { |added| added.include?(parts[:anchor]) }
1045
1220
  next false unless line
@@ -1057,7 +1232,7 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
1057
1232
  # "the adapter does not support"), and single characters with broad
1058
1233
  # compounds (应/只/别 — 应用/只是/区别).
1059
1234
  normative = line.match?(/(?:\b(?:must|shall|never|do\s+not|don'?t|required?|requires?|block(?:s|ed)?|reject(?:s|ed)?|deny|denied|invalidates?|forbid(?:s|den)?|cannot|enforcement)\b|必须|不得|禁止|拒绝|作废|仅限|只能|应当|应该|务必|不能|不允许|不可)/i)
1060
- list_rule && normative
1235
+ (list_rule && normative) || table_data_anchor_valid.call(scope, parts)
1061
1236
  end
1062
1237
  # A routing-surface-only owner has no changed rule line to anchor on: its whole
1063
1238
  # change is one YAML scalar. The answer is NOT to exempt it — a description edit
@@ -1256,7 +1431,7 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
1256
1431
  firing_parts = locator_parts.call(firing_path)
1257
1432
  firing_path_valid = firing_locator_valid.call(row_scope, firing_parts, owner)
1258
1433
  # The machine-checked core is the FIRING PATH (an owner-scoped anchor on a
1259
- # changed normative rule, or a changed owner executable) plus the
1434
+ # changed normative rule/table data, or a changed owner executable) plus the
1260
1435
  # deterministic wording-only classification. The behavioral-evidence
1261
1436
  # status and observed-failure fields are required author declarations —
1262
1437
  # honest labels, not digest-verified artifacts: a digest-bound evidence
@@ -89,8 +89,13 @@ its own branch and a later round could restore the real one, so judging history
89
89
  with history's tools would let that round's forged ledger stand forever. Using
90
90
  the landing tree's tools means a controller or validator change between a round
91
91
  and the promotion can stop an old round reproducing, and that reads as a refusal
92
- rather than a pass. A round is never itself a chain, so the walk is one level
93
- deep by construction. The detached checkout is released with `git worktree
92
+ rather than a pass. A round's evidence is likewise read from the landing tree,
93
+ not from the round's own checkout: a ledger is evidence because the validator
94
+ accepts it and its candidate hash equals the round's packet, not because of
95
+ where it was committed, so a round that merged without its ledger is bound by a
96
+ later review of the same bytes committed on the integration branch -- and by
97
+ nothing less, since a ledger for any other bytes does not match. A round is
98
+ never itself a chain, so the walk is one level deep by construction. The detached checkout is released with `git worktree
94
99
  remove` and its removal verified against the worktree list; a checkout that
95
100
  cannot be released is an error, never a pass, and nothing prunes registrations
96
101
  this run did not create. The chain is
@@ -581,7 +586,7 @@ def render_manifest(
581
586
 
582
587
  def accepted_ledger_for(
583
588
  evidence: list[tuple[Path, dict]],
584
- repo_root: Path,
589
+ evidence_home: Path,
585
590
  validator: Path,
586
591
  digest: str,
587
592
  rejected: list[str],
@@ -590,14 +595,15 @@ def accepted_ledger_for(
590
595
 
591
596
  The same criterion the single-candidate path uses: a receipt-shaped file is
592
597
  not evidence, only a ledger the validator accepts, because this gate cannot
593
- authenticate that a controller minted what it reads.
598
+ authenticate that a controller minted what it reads. `evidence_home` is the
599
+ tree the evidence was enumerated from, used only to name the ledger.
594
600
  """
595
601
  for path, payload in evidence:
596
602
  if payload.get("candidate_sha256") != digest:
597
603
  continue
598
604
  if "closeout_state" not in payload or "controller_receipts" not in payload:
599
605
  continue
600
- relative = str(path.relative_to(repo_root))
606
+ relative = str(path.relative_to(evidence_home))
601
607
  accepted, output = validator_accepts(validator, path)
602
608
  if accepted:
603
609
  return f"{relative} -- {output}"
@@ -613,6 +619,7 @@ def bind_manifest(
613
619
  excludes: tuple[str, ...],
614
620
  changed_all: list[str],
615
621
  evidence: list[tuple[Path, dict]],
622
+ evidence_home: Path,
616
623
  validator: Path,
617
624
  rejected_ledgers: list[str],
618
625
  ) -> list[str]:
@@ -630,7 +637,7 @@ def bind_manifest(
630
637
  raise ManifestError(
631
638
  f"{label} recorded {recorded[:12]}... but does not reproduce: the candidate now hashes to {actual[:12]}..."
632
639
  )
633
- proof = accepted_ledger_for(evidence, repo_root, validator, actual, rejected_ledgers)
640
+ proof = accepted_ledger_for(evidence, evidence_home, validator, actual, rejected_ledgers)
634
641
  if proof is None:
635
642
  raise ManifestError(f"no accepted ledger binds {label} {actual}")
636
643
  proofs.append(f" {label} {actual[:12]}... <- {proof}")
@@ -889,8 +896,9 @@ def bind_chain(
889
896
  Returns (round count, proof lines) or raises ChainError naming the first
890
897
  step that does not add up. Each round is rebound by this same gate in a
891
898
  detached checkout of its head against its first parent, with THIS tree's
892
- controller and validator (never the round's own), and never as a chain of
893
- its own.
899
+ controller, validator and committed evidence (never the round's own tools;
900
+ the round's own evidence is part of this tree's history and is found there),
901
+ and never as a chain of its own.
894
902
  """
895
903
  steps = walk_first_parent_chain(repo_root, base_tip)
896
904
  proofs: list[str] = []
@@ -902,7 +910,13 @@ def bind_chain(
902
910
  continue
903
911
  with detached_checkout(repo_root, second) as round_root:
904
912
  binding = bind_candidate(
905
- round_root, first, DEFAULT_PATHS, evidence_root, allow_chain=False, tools_root=repo_root
913
+ round_root,
914
+ first,
915
+ DEFAULT_PATHS,
916
+ evidence_root,
917
+ allow_chain=False,
918
+ tools_root=repo_root,
919
+ evidence_tree=repo_root,
906
920
  )
907
921
  subject = git_read(repo_root, ["log", "-1", "--format=%s", merge], f"cannot read {merge[:12]}")
908
922
  if not binding.ok:
@@ -955,15 +969,24 @@ def bind_candidate(
955
969
  evidence_root: str,
956
970
  allow_chain: bool,
957
971
  tools_root: Path | None = None,
972
+ evidence_tree: Path | None = None,
958
973
  ) -> Binding:
959
974
  """Evaluate one checkout against one base: single ledger, then manifest, then chain.
960
975
 
961
976
  `tools_root` names the tree whose controller and validator judge the
962
977
  candidate; it defaults to the checkout itself and is the landing tree when a
963
978
  historical round is rebound, so a round never judges itself with its own tools.
979
+ `evidence_tree` names the tree whose committed evidence is consulted; it too
980
+ defaults to the checkout and is the landing tree when a historical round is
981
+ rebound. Evidence is a validator-accepted closeout bound to the round's own
982
+ candidate hash wherever it was committed: a round that merged without its
983
+ ledger is bound by a later review of the same bytes, committed on the
984
+ integration branch, and by nothing less -- the candidate hash and the
985
+ validator, not the file's location, are what make a ledger evidence.
964
986
  """
965
987
  binding = Binding()
966
988
  tools = tools_root if tools_root is not None else repo_root
989
+ evidence_home = evidence_tree if evidence_tree is not None else repo_root
967
990
  fork, excludes, paths, changed = candidate_scope(repo_root, base_tip, user_paths)
968
991
  binding.fork = fork
969
992
  binding.changed = changed
@@ -987,14 +1010,14 @@ def bind_candidate(
987
1010
  whole_error = str(exc)
988
1011
 
989
1012
  validator = tools / "skills" / "skill-extraction-workflow" / "scripts" / VALIDATOR
990
- evidence = scan(repo_root, evidence_root)
1013
+ evidence = scan(evidence_home, evidence_root)
991
1014
  ledgers: list[str] = []
992
1015
  if expected is not None:
993
1016
  # Only a validator-accepted ledger counts. A receipt-shaped file proves
994
1017
  # nothing on its own: this gate cannot authenticate that a controller
995
1018
  # minted it, so any branch keyed on a self-declared field is a bypass a
996
1019
  # contributor can hand-write.
997
- proof = accepted_ledger_for(evidence, repo_root, validator, expected, ledgers)
1020
+ proof = accepted_ledger_for(evidence, evidence_home, validator, expected, ledgers)
998
1021
  if proof is not None:
999
1022
  binding.ok = True
1000
1023
  binding.summary = (
@@ -1007,10 +1030,10 @@ def bind_candidate(
1007
1030
  for path, payload in evidence:
1008
1031
  if payload.get("kind") != MANIFEST_KIND:
1009
1032
  continue
1010
- relative = str(path.relative_to(repo_root))
1033
+ relative = str(path.relative_to(evidence_home))
1011
1034
  try:
1012
1035
  proofs = bind_manifest(
1013
- module, repo_root, fork, payload, excludes, changed, evidence, validator, ledgers
1036
+ module, repo_root, fork, payload, excludes, changed, evidence, evidence_home, validator, ledgers
1014
1037
  )
1015
1038
  except ManifestError as exc:
1016
1039
  manifests.append(f"{relative}: {exc}")
@@ -52,7 +52,7 @@ Usage:
52
52
  Run in a SCRATCH checkout: the current arm executes the installed hooks/plugins (not just the
53
53
  read-only model tools), so treat it as potentially side-effecting, not inert.
54
54
  """
55
- import argparse, hashlib, json, os, re, subprocess, sys, threading, time
55
+ import argparse, hashlib, json, os, re, signal, subprocess, sys, time
56
56
 
57
57
  HERE = os.path.dirname(os.path.abspath(__file__))
58
58
  DEFAULT_FIXTURES = os.path.normpath(os.path.join(HERE, "..", "..", "..", "eval", "behavior-fixtures.jsonl"))
@@ -100,23 +100,62 @@ def _headless_claude(cmd, prompt, timeout_s):
100
100
  # stderr → DEVNULL: we never read it, and a full stderr pipe would deadlock the child
101
101
  # on a verbose run and time out an otherwise-valid answer.
102
102
  p = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE,
103
- stderr=subprocess.DEVNULL, text=True)
103
+ stderr=subprocess.DEVNULL, text=True,
104
+ start_new_session=(os.name == "posix"))
104
105
  except FileNotFoundError:
105
106
  return None, "claude_not_found", None, []
106
- out = {"s": ""}
107
- t = threading.Thread(target=lambda: out.__setitem__("s", p.stdout.read()))
108
- t.start()
109
107
  try:
110
- p.stdin.write(prompt); p.stdin.close()
111
- except (BrokenPipeError, OSError):
112
- pass
113
- t.join(timeout_s)
114
- if t.is_alive():
115
- p.kill(); t.join(2)
116
- return None, f"timeout_{timeout_s}s", None, []
117
- rc = p.wait()
108
+ # One deadline covers stdin backpressure, stdout collection and process
109
+ # completion. A reader-only timer leaves writes and p.wait() unbounded.
110
+ out, _ = p.communicate(input=prompt, timeout=timeout_s)
111
+ except BaseException as stopped:
112
+ # The group can outlive its leader while a descendant holds stdout open.
113
+ # Kill the recorded group even when the direct child has already exited.
114
+ cleanup_errors = []
115
+ try:
116
+ if os.name == "posix":
117
+ os.killpg(p.pid, signal.SIGKILL)
118
+ else:
119
+ cleanup_errors.append("descendant_cleanup_unsupported")
120
+ p.kill()
121
+ except ProcessLookupError:
122
+ pass
123
+ except PermissionError:
124
+ target = "group" if os.name == "posix" else "process"
125
+ cleanup_errors.append(f"{target}_kill_permission_denied")
126
+ try:
127
+ p.communicate(timeout=1)
128
+ except (subprocess.TimeoutExpired, OSError) as cleanup_error:
129
+ # An escaped descendant may still own a pipe; do not wait for EOF.
130
+ cleanup_errors.append("stdio_timeout" if isinstance(cleanup_error, subprocess.TimeoutExpired) else "stdio_error")
131
+ try:
132
+ p.kill()
133
+ except ProcessLookupError:
134
+ pass
135
+ except PermissionError:
136
+ cleanup_errors.append("process_kill_permission_denied")
137
+ try:
138
+ p.wait(timeout=1)
139
+ except (subprocess.TimeoutExpired, OSError) as cleanup_error:
140
+ cleanup_errors.append("wait_timeout" if isinstance(cleanup_error, subprocess.TimeoutExpired) else "wait_error")
141
+ # A denied kill or incomplete drain/reap is not evidence of cleanup.
142
+ # Return it so callers can save partial results and stop new processes.
143
+ interrupted = not isinstance(stopped, subprocess.TimeoutExpired)
144
+ error = "interrupted" if interrupted else f"timeout_{timeout_s}s"
145
+ if cleanup_errors:
146
+ error += ";cleanup_unconfirmed:" + ",".join(cleanup_errors)
147
+ if interrupted:
148
+ if cleanup_errors:
149
+ print(f"{error}; confirm process termination before resuming", file=sys.stderr)
150
+ raise
151
+ return None, error, None, []
152
+ finally:
153
+ p.stdin.close()
154
+ p.stdout.close()
155
+ rc = p.returncode
118
156
  parts, result_text, result_subtype, util, invoked = [], None, None, None, []
119
- for ln in out["s"].splitlines():
157
+ terminal_results = []
158
+ for ln in out.splitlines():
120
159
  # The stream is external/untrusted: a line may be invalid JSON, deeply nested (RecursionError),
121
160
  # or a shape-drifted value. Wrap the WHOLE per-line parse+extract so any bad line skips itself
122
161
  # and never crashes the eval run (fail-closed-skip). We read only known fields of known event
@@ -127,6 +166,7 @@ def _headless_claude(cmd, prompt, timeout_s):
127
166
  if not isinstance(ev, dict):
128
167
  continue
129
168
  if ev.get("type") == "result":
169
+ terminal_results.append(ev)
130
170
  result_subtype = ev.get("subtype")
131
171
  if result_subtype == "success":
132
172
  result_text = ev.get("result")
@@ -159,8 +199,19 @@ def _headless_claude(cmd, prompt, timeout_s):
159
199
  # teardown failure — the answer is complete, accept. Without that event we do NOT bank the
160
200
  # text, even if some assistant chunks streamed and rc==0: a truncation or format drift before
161
201
  # the terminal event would otherwise be recorded as a valid sample.
162
- if result_subtype == "success":
163
- return ("\n\n".join(parts) if parts else result_text), None, util, invoked
202
+ if len(terminal_results) > 1:
203
+ return None, "invalid_terminal_result", util, invoked
204
+ if result_subtype == "success" and terminal_results:
205
+ terminal = terminal_results[0]
206
+ if (terminal.get("is_error") not in (None, False)
207
+ or terminal.get("permission_denials")
208
+ or terminal.get("api_error_status") not in (None, 0, "0")
209
+ or terminal.get("terminal_reason") not in (None, "completed")):
210
+ return None, "invalid_terminal_result", util, invoked
211
+ text = "\n\n".join(parts) if parts else result_text
212
+ if not isinstance(text, str) or not text.strip():
213
+ return None, "invalid_terminal_result", util, invoked
214
+ return text, None, util, invoked
164
215
  if rc != 0:
165
216
  return None, f"claude_exit_{rc}", util, invoked
166
217
  if result_subtype:
@@ -312,8 +363,12 @@ def build_report(rows, out_dir, do_judge, timeout_s, last_util, stop_util,
312
363
  in the summary (never silently dropped, or the report reads clean when it isn't). Raw responses
313
364
  stay on disk for audit; judgment-ASSIST, never a score. do_judge=False → fill-in scaffold."""
314
365
  verdicts = []
366
+ cleanup_error = None
315
367
  for fx in rows:
316
368
  base = {"id": fx["id"], "axis": fx.get("axis", "")}
369
+ if cleanup_error:
370
+ verdicts.append({**base, "status": "judge-skipped", "note": cleanup_error})
371
+ continue
317
372
  cur = read_saved_response(os.path.join(out_dir, f"{fx['id']}.current.s1.txt"))
318
373
  cand = read_saved_response(os.path.join(out_dir, f"{fx['id']}.candidate.s1.txt"))
319
374
  if not cur or not cand:
@@ -342,6 +397,8 @@ def build_report(rows, out_dir, do_judge, timeout_s, last_util, stop_util,
342
397
  last_util = util
343
398
  if err:
344
399
  verdicts.append({**base, "status": f"judge-error:{err}"})
400
+ if ";cleanup_unconfirmed:" in err:
401
+ cleanup_error = err
345
402
  continue
346
403
  v.update(base); v["status"] = "judged"
347
404
  if fixture_needs_human(fx):
@@ -467,8 +524,10 @@ def main():
467
524
  if not a.no_judge:
468
525
  print(f"--report-only will make up to {len(rows)} LLM-judge claude call(s) "
469
526
  f"(one per fixture with both arms saved). Use --no-judge for a fill-in scaffold.")
470
- build_report(rows, a.out, not a.no_judge, a.timeout, None, a.stop_util, ctext, a.current_tag)
527
+ verdicts = build_report(rows, a.out, not a.no_judge, a.timeout, None, a.stop_util, ctext, a.current_tag)
471
528
  print(f"\nReport: {os.path.join(a.out, 'capability-delta-report.md')}")
529
+ if any(";cleanup_unconfirmed:" in v["status"] for v in verdicts):
530
+ sys.exit(1)
472
531
  return
473
532
  # Fail fast: don't burn every current-arm run and only then discover the candidate
474
533
  # arm has no contract to inject.
@@ -501,6 +560,12 @@ def main():
501
560
  f"{a.stop_util:.0%} — stopping before [{fx['id']}/{arm}/s{s}]. "
502
561
  f"Partial results in {a.out}; rerun to continue, then --report-only.")
503
562
  logf.close(); return
563
+ # A failed replacement must not leave its old response usable.
564
+ # Invalidate only after quota allows this sample to start.
565
+ try:
566
+ os.unlink(rpath)
567
+ except FileNotFoundError:
568
+ pass
504
569
  t0 = time.time()
505
570
  text, err, util, invoked = run_agent(fx["prompt"], a.timeout, arm, a.contract)
506
571
  dt = time.time() - t0
@@ -509,7 +574,13 @@ def main():
509
574
  if err:
510
575
  print(f"[{fx['id']}/{arm}/s{s}] ERROR {err} ({dt:.0f}s)")
511
576
  logf.write(json.dumps({"id": fx["id"], "arm": arm, "sample": s, "error": err}) + "\n")
512
- logf.flush(); continue
577
+ logf.flush()
578
+ if ";cleanup_unconfirmed:" in err:
579
+ print(f"*** ABORT: process cleanup is unconfirmed; no further samples or judges started. "
580
+ f"Partial results in {a.out}. Confirm process termination before resuming.")
581
+ logf.close()
582
+ sys.exit(1)
583
+ continue
513
584
  with open(rpath, "w", encoding="utf-8") as rf:
514
585
  rf.write(f"# sig: {sig}\n")
515
586
  rf.write(f"# {fx['id']} / arm={arm} / sample={s} / {dt:.0f}s / util={last_util}\n")
@@ -528,13 +599,24 @@ def main():
528
599
  if "current" in arms and "candidate" in arms:
529
600
  print("Building capability-delta report (candidate vs current)"
530
601
  + ("" if a.no_judge else " via LLM-judge — this makes more claude calls") + " ...")
531
- build_report(rows, a.out, not a.no_judge, a.timeout, last_util, a.stop_util,
532
- contract_text, a.current_tag)
602
+ verdicts = build_report(rows, a.out, not a.no_judge, a.timeout, last_util, a.stop_util,
603
+ contract_text, a.current_tag)
533
604
  print(f"Report: {os.path.join(a.out, 'capability-delta-report.md')} "
534
605
  f"(confirm every 🔴 HUMAN row by eye; it is judgment-assist, not a score).")
606
+ if any(";cleanup_unconfirmed:" in v["status"] for v in verdicts):
607
+ sys.exit(1)
535
608
  else:
536
609
  print("Single arm — no delta to report. Run --both-arms for the capability-delta report.")
537
610
 
538
611
 
539
612
  if __name__ == "__main__":
540
- main()
613
+ previous_sigterm = signal.getsignal(signal.SIGTERM)
614
+ if previous_sigterm == signal.SIG_DFL:
615
+ def terminate(signum, _frame):
616
+ raise SystemExit(128 + signum)
617
+ signal.signal(signal.SIGTERM, terminate)
618
+ try:
619
+ main()
620
+ finally:
621
+ if previous_sigterm == signal.SIG_DFL:
622
+ signal.signal(signal.SIGTERM, previous_sigterm)