@ccoalm/ccl-skills 0.14.0 → 0.15.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +80 -4
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +16 -4
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +13 -2
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +17 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +14 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +40 -83
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +44 -37
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +9 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +52 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +31 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +188 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +36 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +103 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +261 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_runtime.py +428 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +63 -5
- package/dist/assets/release.json +76 -56
- package/dist/claude-adapter.js +14 -7
- package/dist/codex-host.d.ts +1 -3
- package/dist/codex-host.js +6 -9
- package/dist/host-probe.d.ts +11 -0
- package/dist/host-probe.js +27 -0
- package/dist/opencode-adapter.js +24 -19
- package/dist/unified.js +11 -9
- package/package.json +1 -1
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
# upstream-owner skill must declare a behavioral-evidence status and an
|
|
6
6
|
# observed-failure state, and (for non-wording changes) name an owner-scoped
|
|
7
7
|
# FIRING PATH that resolves to this diff — an anchor on a changed normative
|
|
8
|
-
# rule line, or a changed owner executable. The statuses are required author
|
|
8
|
+
# rule or table data line, or a changed owner executable. The statuses are required author
|
|
9
9
|
# declarations; the firing path and the wording-only classification are the
|
|
10
10
|
# machine-verified core. Extracted from the former inline `ruby -e` block in
|
|
11
11
|
# check-ccl-skills.sh so the program gets normal Ruby tooling and no
|
|
@@ -52,19 +52,123 @@ LEDGER_PATH = "skills/skill-extraction-workflow/references/source-register.md"
|
|
|
52
52
|
# the routing-surface class below was one such patch) only re-instantiates it on
|
|
53
53
|
# the next input. The predicate now reads the round.
|
|
54
54
|
#
|
|
55
|
-
# Rounds are cut at the commits that touch the ledger itself, walked
|
|
56
|
-
#
|
|
57
|
-
#
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
#
|
|
55
|
+
# Rounds are cut at the commits that touch the ledger itself, walked along a
|
|
56
|
+
# first-parent line. The partition comes from git alone: an author cannot widen,
|
|
57
|
+
# move, or nominate their own scope.
|
|
58
|
+
#
|
|
59
|
+
# THE SAME PARTITION BEFORE AND AFTER THE MERGE. A branch is judged on its own
|
|
60
|
+
# first-parent line while it is a pull request (CI checks out the branch head),
|
|
61
|
+
# and once merged that whole line sits behind ONE first-parent step of the
|
|
62
|
+
# integration branch. Reading that step as one boundary gave the same history a
|
|
63
|
+
# different partition after it landed — every round on the branch collapsed into
|
|
64
|
+
# one, and a row whose validity depends on its round being narrow (a routing-
|
|
65
|
+
# surface `#description` anchor in a commit that changed nothing else) turned red
|
|
66
|
+
# without a byte of it changing. That is the "verdict moved after it landed"
|
|
67
|
+
# defect in a new coat, and it surfaced on every post-merge evaluation: the push
|
|
68
|
+
# build of the integration branch and the promotion pull request. So a merge that
|
|
69
|
+
# git itself reproduces from its two parents is EXPANDED in place: its second
|
|
70
|
+
# parent's line, from the fork point to the merged head, is walked with the same
|
|
71
|
+
# rule, recursively, and contributes exactly the rounds it had as a branch.
|
|
72
|
+
#
|
|
73
|
+
# A merge is expanded only when git can rebuild it — two parents, a tree equal to
|
|
74
|
+
# `git merge-tree --write-tree` of those parents, and a second parent that is
|
|
75
|
+
# neither already on the base nor already on the line being walked (a sync merge
|
|
76
|
+
# brings nothing that needs a round). Anything else — a hand-resolved merge, a
|
|
77
|
+
# conflicted one, an octopus — keeps today's single boundary at the merge, so
|
|
78
|
+
# content git did not derive from the parents is never left in no round.
|
|
79
|
+
ROUND_WALK_MAX_DEPTH = 8
|
|
80
|
+
ancestor_of = lambda do |commit, tip|
|
|
81
|
+
IO.popen(["git", "-C", root, "merge-base", "--is-ancestor", commit, tip], err: File::NULL, &:read)
|
|
82
|
+
status = $?.exitstatus
|
|
83
|
+
next true if status == 0
|
|
84
|
+
next false if status == 1
|
|
85
|
+
warn "impact_chain_git_failed: git merge-base --is-ancestor #{commit} #{tip} exited #{status}"
|
|
86
|
+
exit 1
|
|
87
|
+
end
|
|
88
|
+
# The tree git produces merging `second` into `first`; nil when that merge
|
|
89
|
+
# conflicts (whoever resolved it was not git). Needs git 2.38+, the same floor
|
|
90
|
+
# the review-ledger binder already requires for the identical invariant.
|
|
91
|
+
automatic_merge_tree = lambda do |first, second|
|
|
92
|
+
out = IO.popen(["git", "-C", root, "merge-tree", "--write-tree", first, second], err: File::NULL, &:read)
|
|
93
|
+
status = $?.exitstatus
|
|
94
|
+
next out.to_s.lines.first.to_s.strip if status == 0
|
|
95
|
+
next nil if status == 1
|
|
96
|
+
warn "impact_chain_git_failed: git merge-tree --write-tree #{first} #{second} exited #{status} (git 2.38 or newer is required)"
|
|
97
|
+
exit 1
|
|
98
|
+
end
|
|
99
|
+
# [first parent, second parent, fork point] when `commit` is a merge git can
|
|
100
|
+
# rebuild from its parents and whose second parent carries a line of its own;
|
|
101
|
+
# nil when the merge keeps today's single-boundary treatment. `line_base` is the
|
|
102
|
+
# base of the line being walked: a merge whose second parent is already below
|
|
103
|
+
# that base is a sync of what the line was cut from (the target advancing under
|
|
104
|
+
# a branch), and is a sync on the branch's own line exactly as it is on the
|
|
105
|
+
# integration line — judging it against the outer base alone would expand it
|
|
106
|
+
# during promotion and strand the branch's earlier work in a rowless span.
|
|
107
|
+
expandable_merge = lambda do |commit, line_base|
|
|
108
|
+
parents = git_read.call("rev-list", "--parents", "-n", "1", commit).split[1..] || []
|
|
109
|
+
next nil unless parents.length == 2
|
|
110
|
+
first, second = parents
|
|
111
|
+
next nil if ancestor_of.call(second, base_ref) || ancestor_of.call(second, line_base) || ancestor_of.call(second, first)
|
|
112
|
+
own_tree = git_read.call("rev-parse", "#{commit}^{tree}").strip
|
|
113
|
+
next nil unless automatic_merge_tree.call(first, second) == own_tree
|
|
114
|
+
# The fork point read FAILS CLOSED like every other git read here: a lookup
|
|
115
|
+
# that errored would otherwise read as "no fork point", skip the expansion,
|
|
116
|
+
# and hand the merge the collapsed span — the lenient verdict — on a git
|
|
117
|
+
# failure nobody sees. Exit 1 is git's own "no common ancestor" and means
|
|
118
|
+
# there is genuinely no line to expand from.
|
|
119
|
+
fork_out = IO.popen(["git", "-C", root, "merge-base", first, second], err: File::NULL, &:read)
|
|
120
|
+
fork_status = $?.exitstatus
|
|
121
|
+
next nil if fork_status == 1
|
|
122
|
+
unless fork_status == 0
|
|
123
|
+
warn "impact_chain_git_failed: git merge-base #{first} #{second} exited #{fork_status}"
|
|
124
|
+
exit 1
|
|
125
|
+
end
|
|
126
|
+
fork = fork_out.to_s.split("\n").first.to_s.strip
|
|
127
|
+
next nil if fork.empty?
|
|
128
|
+
[first, second, fork]
|
|
129
|
+
end
|
|
130
|
+
# Each round spans (previous boundary, this one] so the work commits that
|
|
61
131
|
# precede a ledger append are inside the round they belong to — landing the change
|
|
62
132
|
# and appending the row in separate commits is the normal shape, not an evasion.
|
|
63
|
-
|
|
64
|
-
# The trailing span — owner changes committed after the last ledger append — is a
|
|
133
|
+
# The trailing span — owner changes committed after the last boundary — is a
|
|
65
134
|
# round too. It holds no rows, so its owners fall through to the presence check
|
|
66
|
-
# and the gate still fails closed on undeclared work.
|
|
67
|
-
|
|
135
|
+
# and the gate still fails closed on undeclared work. Both hold on every line the
|
|
136
|
+
# walk visits, the integration branch and each expanded merge alike.
|
|
137
|
+
round_bounds_for = lambda do |from, to, depth|
|
|
138
|
+
if depth > ROUND_WALK_MAX_DEPTH
|
|
139
|
+
warn "impact_chain_round_walk_too_deep: merges nested more than #{ROUND_WALK_MAX_DEPTH} levels between #{from} and #{to}"
|
|
140
|
+
exit 1
|
|
141
|
+
end
|
|
142
|
+
list = lambda do |*options, pathspec|
|
|
143
|
+
git_read.call("rev-list", "--first-parent", "--reverse", *options, "#{from}..#{to}", *pathspec)
|
|
144
|
+
.split("\n").map(&:strip).reject(&:empty?)
|
|
145
|
+
end
|
|
146
|
+
line = list.call([])
|
|
147
|
+
ledger_heads = list.call(["--", LEDGER_PATH])
|
|
148
|
+
merges = list.call("--merges", [])
|
|
149
|
+
spans = []
|
|
150
|
+
prev = from
|
|
151
|
+
line.each do |commit|
|
|
152
|
+
expansion = merges.include?(commit) ? expandable_merge.call(commit, from) : nil
|
|
153
|
+
if expansion
|
|
154
|
+
first, second, fork = expansion
|
|
155
|
+
spans << [prev, first] unless prev == first
|
|
156
|
+
spans.concat(round_bounds_for.call(fork, second, depth + 1))
|
|
157
|
+
prev = commit
|
|
158
|
+
elsif ledger_heads.include?(commit)
|
|
159
|
+
spans << [prev, commit]
|
|
160
|
+
prev = commit
|
|
161
|
+
end
|
|
162
|
+
end
|
|
163
|
+
spans << [prev, to]
|
|
164
|
+
spans
|
|
165
|
+
end
|
|
166
|
+
round_bounds = round_bounds_for.call(base_ref, "HEAD", 0)
|
|
167
|
+
# Diagnostic only: print the partition so a verdict can be read against the
|
|
168
|
+
# rounds it was judged in. Off by default so no suite's output assertions move.
|
|
169
|
+
if ENV["CCL_IMPACT_CHAIN_TRACE_ROUNDS"] == "1"
|
|
170
|
+
round_bounds.each { |span_base, span_head| warn "impact_chain_round: #{span_base[0, 12]}..#{span_head[0, 12]}" }
|
|
171
|
+
end
|
|
68
172
|
# Everything a predicate needs to judge one span. Built lazily per span and
|
|
69
173
|
# memoized: a round whose rows are all RED-baseline never pays for the rename
|
|
70
174
|
# derivation.
|
|
@@ -1037,9 +1141,80 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
|
|
|
1037
1141
|
blob[:content].scan(Regexp.new(Regexp.escape(anchor))).length == 1
|
|
1038
1142
|
end
|
|
1039
1143
|
end
|
|
1144
|
+
# Definition/decision tables are firing surfaces too. Recognize explicit
|
|
1145
|
+
# Markdown tables with a header and delimiter; comments and code examples do
|
|
1146
|
+
# not count. This proves location/shape, not the truth of the definition.
|
|
1147
|
+
table_data_anchor_valid = lambda do |scope, parts|
|
|
1148
|
+
prior = blob_at.call(scope.base, parts[:path])
|
|
1149
|
+
next false if prior && prior[:content].include?(parts[:anchor])
|
|
1150
|
+
previous_cells = nil
|
|
1151
|
+
columns = nil
|
|
1152
|
+
fence = nil
|
|
1153
|
+
comment = false
|
|
1154
|
+
raw_html = nil
|
|
1155
|
+
html_block = false
|
|
1156
|
+
head_blob.call(scope, parts[:path])[:content].each_line do |raw_line|
|
|
1157
|
+
line = raw_line.chomp
|
|
1158
|
+
if fence
|
|
1159
|
+
fence = nil if line.match?(/\A {0,3}#{Regexp.escape(fence[0])}{#{fence.length},}\s*\z/)
|
|
1160
|
+
next
|
|
1161
|
+
end
|
|
1162
|
+
if raw_html
|
|
1163
|
+
raw_html = nil if line.match?(raw_html)
|
|
1164
|
+
next
|
|
1165
|
+
end
|
|
1166
|
+
if html_block
|
|
1167
|
+
html_block = false if line.strip.empty?
|
|
1168
|
+
next
|
|
1169
|
+
end
|
|
1170
|
+
hidden = comment || line.include?("<!--") || line.include?("-->")
|
|
1171
|
+
line.scan(/<!--|-->/).each { |marker| comment = marker == "<!--" }
|
|
1172
|
+
if hidden
|
|
1173
|
+
previous_cells = columns = nil
|
|
1174
|
+
next
|
|
1175
|
+
end
|
|
1176
|
+
if (opening = line.match(/\A {0,3}(`{3,}|~{3,})/))
|
|
1177
|
+
fence = opening[1]
|
|
1178
|
+
previous_cells = columns = nil
|
|
1179
|
+
next
|
|
1180
|
+
end
|
|
1181
|
+
terminator = case line
|
|
1182
|
+
when /\A {0,3}<\?/ then /\?>/
|
|
1183
|
+
when /\A {0,3}<!\[CDATA\[/ then /\]\]>/
|
|
1184
|
+
when /\A {0,3}<![A-Z]/ then />/
|
|
1185
|
+
end
|
|
1186
|
+
if terminator
|
|
1187
|
+
raw_html = terminator unless line.match?(terminator)
|
|
1188
|
+
previous_cells = columns = nil
|
|
1189
|
+
next
|
|
1190
|
+
end
|
|
1191
|
+
if line.match?(%r{\A {0,3}</?[A-Za-z][\w-]*(?:\s|>|/|\z)})
|
|
1192
|
+
tag = line[/\A {0,3}<(script|pre|style|textarea)(?:\s|>|\z)/i, 1]
|
|
1193
|
+
raw_html = %r{</#{tag}\s*>}i if tag && !line.match?(%r{</#{tag}\s*>}i)
|
|
1194
|
+
html_block = !tag
|
|
1195
|
+
previous_cells = columns = nil
|
|
1196
|
+
next
|
|
1197
|
+
end
|
|
1198
|
+
unless line.match?(/\A {0,3}\|.*\|\s*\z/)
|
|
1199
|
+
previous_cells = columns = nil
|
|
1200
|
+
next
|
|
1201
|
+
end
|
|
1202
|
+
cells = line.strip[1...-1].split(/(?<!\\)\|/, -1).map(&:strip)
|
|
1203
|
+
delimiter = cells.length >= 2 && cells.all? { |cell| cell.match?(/\A:?-{3,}:?\z/) }
|
|
1204
|
+
if delimiter
|
|
1205
|
+
columns = previous_cells && previous_cells.length == cells.length ? cells.length : nil
|
|
1206
|
+
elsif columns && columns == cells.length
|
|
1207
|
+
break true if cells.any? { |cell| cell.include?(parts[:anchor]) }
|
|
1208
|
+
else
|
|
1209
|
+
columns = nil
|
|
1210
|
+
end
|
|
1211
|
+
previous_cells = cells
|
|
1212
|
+
end == true
|
|
1213
|
+
end
|
|
1040
1214
|
enforcing_file_locator_valid = lambda do |scope, parts|
|
|
1041
1215
|
next false unless parts && parts[:kind] == "file"
|
|
1042
1216
|
next false unless parts[:path].end_with?(".md")
|
|
1217
|
+
next false if parts[:path] == LEDGER_PATH # Evidence cannot certify itself.
|
|
1043
1218
|
next false unless locator_valid.call(scope, "file:#{parts[:path]}##{parts[:anchor]}")
|
|
1044
1219
|
line = added_lines_for.call(scope, parts[:path]).find { |added| added.include?(parts[:anchor]) }
|
|
1045
1220
|
next false unless line
|
|
@@ -1057,7 +1232,7 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
|
|
|
1057
1232
|
# "the adapter does not support"), and single characters with broad
|
|
1058
1233
|
# compounds (应/只/别 — 应用/只是/区别).
|
|
1059
1234
|
normative = line.match?(/(?:\b(?:must|shall|never|do\s+not|don'?t|required?|requires?|block(?:s|ed)?|reject(?:s|ed)?|deny|denied|invalidates?|forbid(?:s|den)?|cannot|enforcement)\b|必须|不得|禁止|拒绝|作废|仅限|只能|应当|应该|务必|不能|不允许|不可)/i)
|
|
1060
|
-
list_rule && normative
|
|
1235
|
+
(list_rule && normative) || table_data_anchor_valid.call(scope, parts)
|
|
1061
1236
|
end
|
|
1062
1237
|
# A routing-surface-only owner has no changed rule line to anchor on: its whole
|
|
1063
1238
|
# change is one YAML scalar. The answer is NOT to exempt it — a description edit
|
|
@@ -1256,7 +1431,7 @@ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_
|
|
|
1256
1431
|
firing_parts = locator_parts.call(firing_path)
|
|
1257
1432
|
firing_path_valid = firing_locator_valid.call(row_scope, firing_parts, owner)
|
|
1258
1433
|
# The machine-checked core is the FIRING PATH (an owner-scoped anchor on a
|
|
1259
|
-
# changed normative rule, or a changed owner executable) plus the
|
|
1434
|
+
# changed normative rule/table data, or a changed owner executable) plus the
|
|
1260
1435
|
# deterministic wording-only classification. The behavioral-evidence
|
|
1261
1436
|
# status and observed-failure fields are required author declarations —
|
|
1262
1437
|
# honest labels, not digest-verified artifacts: a digest-bound evidence
|
|
@@ -89,8 +89,13 @@ its own branch and a later round could restore the real one, so judging history
|
|
|
89
89
|
with history's tools would let that round's forged ledger stand forever. Using
|
|
90
90
|
the landing tree's tools means a controller or validator change between a round
|
|
91
91
|
and the promotion can stop an old round reproducing, and that reads as a refusal
|
|
92
|
-
rather than a pass. A round is
|
|
93
|
-
|
|
92
|
+
rather than a pass. A round's evidence is likewise read from the landing tree,
|
|
93
|
+
not from the round's own checkout: a ledger is evidence because the validator
|
|
94
|
+
accepts it and its candidate hash equals the round's packet, not because of
|
|
95
|
+
where it was committed, so a round that merged without its ledger is bound by a
|
|
96
|
+
later review of the same bytes committed on the integration branch -- and by
|
|
97
|
+
nothing less, since a ledger for any other bytes does not match. A round is
|
|
98
|
+
never itself a chain, so the walk is one level deep by construction. The detached checkout is released with `git worktree
|
|
94
99
|
remove` and its removal verified against the worktree list; a checkout that
|
|
95
100
|
cannot be released is an error, never a pass, and nothing prunes registrations
|
|
96
101
|
this run did not create. The chain is
|
|
@@ -581,7 +586,7 @@ def render_manifest(
|
|
|
581
586
|
|
|
582
587
|
def accepted_ledger_for(
|
|
583
588
|
evidence: list[tuple[Path, dict]],
|
|
584
|
-
|
|
589
|
+
evidence_home: Path,
|
|
585
590
|
validator: Path,
|
|
586
591
|
digest: str,
|
|
587
592
|
rejected: list[str],
|
|
@@ -590,14 +595,15 @@ def accepted_ledger_for(
|
|
|
590
595
|
|
|
591
596
|
The same criterion the single-candidate path uses: a receipt-shaped file is
|
|
592
597
|
not evidence, only a ledger the validator accepts, because this gate cannot
|
|
593
|
-
authenticate that a controller minted what it reads.
|
|
598
|
+
authenticate that a controller minted what it reads. `evidence_home` is the
|
|
599
|
+
tree the evidence was enumerated from, used only to name the ledger.
|
|
594
600
|
"""
|
|
595
601
|
for path, payload in evidence:
|
|
596
602
|
if payload.get("candidate_sha256") != digest:
|
|
597
603
|
continue
|
|
598
604
|
if "closeout_state" not in payload or "controller_receipts" not in payload:
|
|
599
605
|
continue
|
|
600
|
-
relative = str(path.relative_to(
|
|
606
|
+
relative = str(path.relative_to(evidence_home))
|
|
601
607
|
accepted, output = validator_accepts(validator, path)
|
|
602
608
|
if accepted:
|
|
603
609
|
return f"{relative} -- {output}"
|
|
@@ -613,6 +619,7 @@ def bind_manifest(
|
|
|
613
619
|
excludes: tuple[str, ...],
|
|
614
620
|
changed_all: list[str],
|
|
615
621
|
evidence: list[tuple[Path, dict]],
|
|
622
|
+
evidence_home: Path,
|
|
616
623
|
validator: Path,
|
|
617
624
|
rejected_ledgers: list[str],
|
|
618
625
|
) -> list[str]:
|
|
@@ -630,7 +637,7 @@ def bind_manifest(
|
|
|
630
637
|
raise ManifestError(
|
|
631
638
|
f"{label} recorded {recorded[:12]}... but does not reproduce: the candidate now hashes to {actual[:12]}..."
|
|
632
639
|
)
|
|
633
|
-
proof = accepted_ledger_for(evidence,
|
|
640
|
+
proof = accepted_ledger_for(evidence, evidence_home, validator, actual, rejected_ledgers)
|
|
634
641
|
if proof is None:
|
|
635
642
|
raise ManifestError(f"no accepted ledger binds {label} {actual}")
|
|
636
643
|
proofs.append(f" {label} {actual[:12]}... <- {proof}")
|
|
@@ -889,8 +896,9 @@ def bind_chain(
|
|
|
889
896
|
Returns (round count, proof lines) or raises ChainError naming the first
|
|
890
897
|
step that does not add up. Each round is rebound by this same gate in a
|
|
891
898
|
detached checkout of its head against its first parent, with THIS tree's
|
|
892
|
-
controller and
|
|
893
|
-
|
|
899
|
+
controller, validator and committed evidence (never the round's own tools;
|
|
900
|
+
the round's own evidence is part of this tree's history and is found there),
|
|
901
|
+
and never as a chain of its own.
|
|
894
902
|
"""
|
|
895
903
|
steps = walk_first_parent_chain(repo_root, base_tip)
|
|
896
904
|
proofs: list[str] = []
|
|
@@ -902,7 +910,13 @@ def bind_chain(
|
|
|
902
910
|
continue
|
|
903
911
|
with detached_checkout(repo_root, second) as round_root:
|
|
904
912
|
binding = bind_candidate(
|
|
905
|
-
round_root,
|
|
913
|
+
round_root,
|
|
914
|
+
first,
|
|
915
|
+
DEFAULT_PATHS,
|
|
916
|
+
evidence_root,
|
|
917
|
+
allow_chain=False,
|
|
918
|
+
tools_root=repo_root,
|
|
919
|
+
evidence_tree=repo_root,
|
|
906
920
|
)
|
|
907
921
|
subject = git_read(repo_root, ["log", "-1", "--format=%s", merge], f"cannot read {merge[:12]}")
|
|
908
922
|
if not binding.ok:
|
|
@@ -955,15 +969,24 @@ def bind_candidate(
|
|
|
955
969
|
evidence_root: str,
|
|
956
970
|
allow_chain: bool,
|
|
957
971
|
tools_root: Path | None = None,
|
|
972
|
+
evidence_tree: Path | None = None,
|
|
958
973
|
) -> Binding:
|
|
959
974
|
"""Evaluate one checkout against one base: single ledger, then manifest, then chain.
|
|
960
975
|
|
|
961
976
|
`tools_root` names the tree whose controller and validator judge the
|
|
962
977
|
candidate; it defaults to the checkout itself and is the landing tree when a
|
|
963
978
|
historical round is rebound, so a round never judges itself with its own tools.
|
|
979
|
+
`evidence_tree` names the tree whose committed evidence is consulted; it too
|
|
980
|
+
defaults to the checkout and is the landing tree when a historical round is
|
|
981
|
+
rebound. Evidence is a validator-accepted closeout bound to the round's own
|
|
982
|
+
candidate hash wherever it was committed: a round that merged without its
|
|
983
|
+
ledger is bound by a later review of the same bytes, committed on the
|
|
984
|
+
integration branch, and by nothing less -- the candidate hash and the
|
|
985
|
+
validator, not the file's location, are what make a ledger evidence.
|
|
964
986
|
"""
|
|
965
987
|
binding = Binding()
|
|
966
988
|
tools = tools_root if tools_root is not None else repo_root
|
|
989
|
+
evidence_home = evidence_tree if evidence_tree is not None else repo_root
|
|
967
990
|
fork, excludes, paths, changed = candidate_scope(repo_root, base_tip, user_paths)
|
|
968
991
|
binding.fork = fork
|
|
969
992
|
binding.changed = changed
|
|
@@ -987,14 +1010,14 @@ def bind_candidate(
|
|
|
987
1010
|
whole_error = str(exc)
|
|
988
1011
|
|
|
989
1012
|
validator = tools / "skills" / "skill-extraction-workflow" / "scripts" / VALIDATOR
|
|
990
|
-
evidence = scan(
|
|
1013
|
+
evidence = scan(evidence_home, evidence_root)
|
|
991
1014
|
ledgers: list[str] = []
|
|
992
1015
|
if expected is not None:
|
|
993
1016
|
# Only a validator-accepted ledger counts. A receipt-shaped file proves
|
|
994
1017
|
# nothing on its own: this gate cannot authenticate that a controller
|
|
995
1018
|
# minted it, so any branch keyed on a self-declared field is a bypass a
|
|
996
1019
|
# contributor can hand-write.
|
|
997
|
-
proof = accepted_ledger_for(evidence,
|
|
1020
|
+
proof = accepted_ledger_for(evidence, evidence_home, validator, expected, ledgers)
|
|
998
1021
|
if proof is not None:
|
|
999
1022
|
binding.ok = True
|
|
1000
1023
|
binding.summary = (
|
|
@@ -1007,10 +1030,10 @@ def bind_candidate(
|
|
|
1007
1030
|
for path, payload in evidence:
|
|
1008
1031
|
if payload.get("kind") != MANIFEST_KIND:
|
|
1009
1032
|
continue
|
|
1010
|
-
relative = str(path.relative_to(
|
|
1033
|
+
relative = str(path.relative_to(evidence_home))
|
|
1011
1034
|
try:
|
|
1012
1035
|
proofs = bind_manifest(
|
|
1013
|
-
module, repo_root, fork, payload, excludes, changed, evidence, validator, ledgers
|
|
1036
|
+
module, repo_root, fork, payload, excludes, changed, evidence, evidence_home, validator, ledgers
|
|
1014
1037
|
)
|
|
1015
1038
|
except ManifestError as exc:
|
|
1016
1039
|
manifests.append(f"{relative}: {exc}")
|
|
@@ -52,7 +52,7 @@ Usage:
|
|
|
52
52
|
Run in a SCRATCH checkout: the current arm executes the installed hooks/plugins (not just the
|
|
53
53
|
read-only model tools), so treat it as potentially side-effecting, not inert.
|
|
54
54
|
"""
|
|
55
|
-
import argparse, hashlib, json, os, re, subprocess, sys,
|
|
55
|
+
import argparse, hashlib, json, os, re, signal, subprocess, sys, time
|
|
56
56
|
|
|
57
57
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
58
58
|
DEFAULT_FIXTURES = os.path.normpath(os.path.join(HERE, "..", "..", "..", "eval", "behavior-fixtures.jsonl"))
|
|
@@ -100,23 +100,62 @@ def _headless_claude(cmd, prompt, timeout_s):
|
|
|
100
100
|
# stderr → DEVNULL: we never read it, and a full stderr pipe would deadlock the child
|
|
101
101
|
# on a verbose run and time out an otherwise-valid answer.
|
|
102
102
|
p = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE,
|
|
103
|
-
stderr=subprocess.DEVNULL, text=True
|
|
103
|
+
stderr=subprocess.DEVNULL, text=True,
|
|
104
|
+
start_new_session=(os.name == "posix"))
|
|
104
105
|
except FileNotFoundError:
|
|
105
106
|
return None, "claude_not_found", None, []
|
|
106
|
-
out = {"s": ""}
|
|
107
|
-
t = threading.Thread(target=lambda: out.__setitem__("s", p.stdout.read()))
|
|
108
|
-
t.start()
|
|
109
107
|
try:
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
108
|
+
# One deadline covers stdin backpressure, stdout collection and process
|
|
109
|
+
# completion. A reader-only timer leaves writes and p.wait() unbounded.
|
|
110
|
+
out, _ = p.communicate(input=prompt, timeout=timeout_s)
|
|
111
|
+
except BaseException as stopped:
|
|
112
|
+
# The group can outlive its leader while a descendant holds stdout open.
|
|
113
|
+
# Kill the recorded group even when the direct child has already exited.
|
|
114
|
+
cleanup_errors = []
|
|
115
|
+
try:
|
|
116
|
+
if os.name == "posix":
|
|
117
|
+
os.killpg(p.pid, signal.SIGKILL)
|
|
118
|
+
else:
|
|
119
|
+
cleanup_errors.append("descendant_cleanup_unsupported")
|
|
120
|
+
p.kill()
|
|
121
|
+
except ProcessLookupError:
|
|
122
|
+
pass
|
|
123
|
+
except PermissionError:
|
|
124
|
+
target = "group" if os.name == "posix" else "process"
|
|
125
|
+
cleanup_errors.append(f"{target}_kill_permission_denied")
|
|
126
|
+
try:
|
|
127
|
+
p.communicate(timeout=1)
|
|
128
|
+
except (subprocess.TimeoutExpired, OSError) as cleanup_error:
|
|
129
|
+
# An escaped descendant may still own a pipe; do not wait for EOF.
|
|
130
|
+
cleanup_errors.append("stdio_timeout" if isinstance(cleanup_error, subprocess.TimeoutExpired) else "stdio_error")
|
|
131
|
+
try:
|
|
132
|
+
p.kill()
|
|
133
|
+
except ProcessLookupError:
|
|
134
|
+
pass
|
|
135
|
+
except PermissionError:
|
|
136
|
+
cleanup_errors.append("process_kill_permission_denied")
|
|
137
|
+
try:
|
|
138
|
+
p.wait(timeout=1)
|
|
139
|
+
except (subprocess.TimeoutExpired, OSError) as cleanup_error:
|
|
140
|
+
cleanup_errors.append("wait_timeout" if isinstance(cleanup_error, subprocess.TimeoutExpired) else "wait_error")
|
|
141
|
+
# A denied kill or incomplete drain/reap is not evidence of cleanup.
|
|
142
|
+
# Return it so callers can save partial results and stop new processes.
|
|
143
|
+
interrupted = not isinstance(stopped, subprocess.TimeoutExpired)
|
|
144
|
+
error = "interrupted" if interrupted else f"timeout_{timeout_s}s"
|
|
145
|
+
if cleanup_errors:
|
|
146
|
+
error += ";cleanup_unconfirmed:" + ",".join(cleanup_errors)
|
|
147
|
+
if interrupted:
|
|
148
|
+
if cleanup_errors:
|
|
149
|
+
print(f"{error}; confirm process termination before resuming", file=sys.stderr)
|
|
150
|
+
raise
|
|
151
|
+
return None, error, None, []
|
|
152
|
+
finally:
|
|
153
|
+
p.stdin.close()
|
|
154
|
+
p.stdout.close()
|
|
155
|
+
rc = p.returncode
|
|
118
156
|
parts, result_text, result_subtype, util, invoked = [], None, None, None, []
|
|
119
|
-
|
|
157
|
+
terminal_results = []
|
|
158
|
+
for ln in out.splitlines():
|
|
120
159
|
# The stream is external/untrusted: a line may be invalid JSON, deeply nested (RecursionError),
|
|
121
160
|
# or a shape-drifted value. Wrap the WHOLE per-line parse+extract so any bad line skips itself
|
|
122
161
|
# and never crashes the eval run (fail-closed-skip). We read only known fields of known event
|
|
@@ -127,6 +166,7 @@ def _headless_claude(cmd, prompt, timeout_s):
|
|
|
127
166
|
if not isinstance(ev, dict):
|
|
128
167
|
continue
|
|
129
168
|
if ev.get("type") == "result":
|
|
169
|
+
terminal_results.append(ev)
|
|
130
170
|
result_subtype = ev.get("subtype")
|
|
131
171
|
if result_subtype == "success":
|
|
132
172
|
result_text = ev.get("result")
|
|
@@ -159,8 +199,19 @@ def _headless_claude(cmd, prompt, timeout_s):
|
|
|
159
199
|
# teardown failure — the answer is complete, accept. Without that event we do NOT bank the
|
|
160
200
|
# text, even if some assistant chunks streamed and rc==0: a truncation or format drift before
|
|
161
201
|
# the terminal event would otherwise be recorded as a valid sample.
|
|
162
|
-
if
|
|
163
|
-
return
|
|
202
|
+
if len(terminal_results) > 1:
|
|
203
|
+
return None, "invalid_terminal_result", util, invoked
|
|
204
|
+
if result_subtype == "success" and terminal_results:
|
|
205
|
+
terminal = terminal_results[0]
|
|
206
|
+
if (terminal.get("is_error") not in (None, False)
|
|
207
|
+
or terminal.get("permission_denials")
|
|
208
|
+
or terminal.get("api_error_status") not in (None, 0, "0")
|
|
209
|
+
or terminal.get("terminal_reason") not in (None, "completed")):
|
|
210
|
+
return None, "invalid_terminal_result", util, invoked
|
|
211
|
+
text = "\n\n".join(parts) if parts else result_text
|
|
212
|
+
if not isinstance(text, str) or not text.strip():
|
|
213
|
+
return None, "invalid_terminal_result", util, invoked
|
|
214
|
+
return text, None, util, invoked
|
|
164
215
|
if rc != 0:
|
|
165
216
|
return None, f"claude_exit_{rc}", util, invoked
|
|
166
217
|
if result_subtype:
|
|
@@ -312,8 +363,12 @@ def build_report(rows, out_dir, do_judge, timeout_s, last_util, stop_util,
|
|
|
312
363
|
in the summary (never silently dropped, or the report reads clean when it isn't). Raw responses
|
|
313
364
|
stay on disk for audit; judgment-ASSIST, never a score. do_judge=False → fill-in scaffold."""
|
|
314
365
|
verdicts = []
|
|
366
|
+
cleanup_error = None
|
|
315
367
|
for fx in rows:
|
|
316
368
|
base = {"id": fx["id"], "axis": fx.get("axis", "")}
|
|
369
|
+
if cleanup_error:
|
|
370
|
+
verdicts.append({**base, "status": "judge-skipped", "note": cleanup_error})
|
|
371
|
+
continue
|
|
317
372
|
cur = read_saved_response(os.path.join(out_dir, f"{fx['id']}.current.s1.txt"))
|
|
318
373
|
cand = read_saved_response(os.path.join(out_dir, f"{fx['id']}.candidate.s1.txt"))
|
|
319
374
|
if not cur or not cand:
|
|
@@ -342,6 +397,8 @@ def build_report(rows, out_dir, do_judge, timeout_s, last_util, stop_util,
|
|
|
342
397
|
last_util = util
|
|
343
398
|
if err:
|
|
344
399
|
verdicts.append({**base, "status": f"judge-error:{err}"})
|
|
400
|
+
if ";cleanup_unconfirmed:" in err:
|
|
401
|
+
cleanup_error = err
|
|
345
402
|
continue
|
|
346
403
|
v.update(base); v["status"] = "judged"
|
|
347
404
|
if fixture_needs_human(fx):
|
|
@@ -467,8 +524,10 @@ def main():
|
|
|
467
524
|
if not a.no_judge:
|
|
468
525
|
print(f"--report-only will make up to {len(rows)} LLM-judge claude call(s) "
|
|
469
526
|
f"(one per fixture with both arms saved). Use --no-judge for a fill-in scaffold.")
|
|
470
|
-
build_report(rows, a.out, not a.no_judge, a.timeout, None, a.stop_util, ctext, a.current_tag)
|
|
527
|
+
verdicts = build_report(rows, a.out, not a.no_judge, a.timeout, None, a.stop_util, ctext, a.current_tag)
|
|
471
528
|
print(f"\nReport: {os.path.join(a.out, 'capability-delta-report.md')}")
|
|
529
|
+
if any(";cleanup_unconfirmed:" in v["status"] for v in verdicts):
|
|
530
|
+
sys.exit(1)
|
|
472
531
|
return
|
|
473
532
|
# Fail fast: don't burn every current-arm run and only then discover the candidate
|
|
474
533
|
# arm has no contract to inject.
|
|
@@ -501,6 +560,12 @@ def main():
|
|
|
501
560
|
f"{a.stop_util:.0%} — stopping before [{fx['id']}/{arm}/s{s}]. "
|
|
502
561
|
f"Partial results in {a.out}; rerun to continue, then --report-only.")
|
|
503
562
|
logf.close(); return
|
|
563
|
+
# A failed replacement must not leave its old response usable.
|
|
564
|
+
# Invalidate only after quota allows this sample to start.
|
|
565
|
+
try:
|
|
566
|
+
os.unlink(rpath)
|
|
567
|
+
except FileNotFoundError:
|
|
568
|
+
pass
|
|
504
569
|
t0 = time.time()
|
|
505
570
|
text, err, util, invoked = run_agent(fx["prompt"], a.timeout, arm, a.contract)
|
|
506
571
|
dt = time.time() - t0
|
|
@@ -509,7 +574,13 @@ def main():
|
|
|
509
574
|
if err:
|
|
510
575
|
print(f"[{fx['id']}/{arm}/s{s}] ERROR {err} ({dt:.0f}s)")
|
|
511
576
|
logf.write(json.dumps({"id": fx["id"], "arm": arm, "sample": s, "error": err}) + "\n")
|
|
512
|
-
logf.flush()
|
|
577
|
+
logf.flush()
|
|
578
|
+
if ";cleanup_unconfirmed:" in err:
|
|
579
|
+
print(f"*** ABORT: process cleanup is unconfirmed; no further samples or judges started. "
|
|
580
|
+
f"Partial results in {a.out}. Confirm process termination before resuming.")
|
|
581
|
+
logf.close()
|
|
582
|
+
sys.exit(1)
|
|
583
|
+
continue
|
|
513
584
|
with open(rpath, "w", encoding="utf-8") as rf:
|
|
514
585
|
rf.write(f"# sig: {sig}\n")
|
|
515
586
|
rf.write(f"# {fx['id']} / arm={arm} / sample={s} / {dt:.0f}s / util={last_util}\n")
|
|
@@ -528,13 +599,24 @@ def main():
|
|
|
528
599
|
if "current" in arms and "candidate" in arms:
|
|
529
600
|
print("Building capability-delta report (candidate vs current)"
|
|
530
601
|
+ ("" if a.no_judge else " via LLM-judge — this makes more claude calls") + " ...")
|
|
531
|
-
build_report(rows, a.out, not a.no_judge, a.timeout, last_util, a.stop_util,
|
|
532
|
-
|
|
602
|
+
verdicts = build_report(rows, a.out, not a.no_judge, a.timeout, last_util, a.stop_util,
|
|
603
|
+
contract_text, a.current_tag)
|
|
533
604
|
print(f"Report: {os.path.join(a.out, 'capability-delta-report.md')} "
|
|
534
605
|
f"(confirm every 🔴 HUMAN row by eye; it is judgment-assist, not a score).")
|
|
606
|
+
if any(";cleanup_unconfirmed:" in v["status"] for v in verdicts):
|
|
607
|
+
sys.exit(1)
|
|
535
608
|
else:
|
|
536
609
|
print("Single arm — no delta to report. Run --both-arms for the capability-delta report.")
|
|
537
610
|
|
|
538
611
|
|
|
539
612
|
if __name__ == "__main__":
|
|
540
|
-
|
|
613
|
+
previous_sigterm = signal.getsignal(signal.SIGTERM)
|
|
614
|
+
if previous_sigterm == signal.SIG_DFL:
|
|
615
|
+
def terminate(signum, _frame):
|
|
616
|
+
raise SystemExit(128 + signum)
|
|
617
|
+
signal.signal(signal.SIGTERM, terminate)
|
|
618
|
+
try:
|
|
619
|
+
main()
|
|
620
|
+
finally:
|
|
621
|
+
if previous_sigterm == signal.SIG_DFL:
|
|
622
|
+
signal.signal(signal.SIGTERM, previous_sigterm)
|