@ccoalm/ccl-skills 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +10 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +14 -41
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/correction-routing-map.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/coverage-exhaustion-traps.md +7 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/incident-postmortem-extraction.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +73 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/entrypoint_form_census.py +169 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +114 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/reference-access-census.sh +157 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +483 -109
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +188 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_form_census.sh +174 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_reference_access_census.sh +209 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +394 -5
- package/dist/assets/release.json +77 -47
- package/package.json +1 -1
|
@@ -42,6 +42,14 @@ require "shellwords"
|
|
|
42
42
|
require "timeout"
|
|
43
43
|
require "digest"
|
|
44
44
|
|
|
45
|
+
# Resolution floor for ACTING on a measured case. A report taken below it
|
|
46
|
+
# LOCATES candidates; it does not license a description edit, because at three
|
|
47
|
+
# replicas a case that routes correctly 80-90% of the time reads as failing and
|
|
48
|
+
# a one-off deviation reads as a finding. `references/eval-routing.md` owns the
|
|
49
|
+
# rule and states the same number; test_eval_routing_bank_resolution.sh pins the
|
|
50
|
+
# two sides together so they cannot drift apart.
|
|
51
|
+
ACTION_RESOLUTION_MIN_REPLICAS = 10
|
|
52
|
+
|
|
45
53
|
def arg(flag, default = nil)
|
|
46
54
|
i = ARGV.index(flag)
|
|
47
55
|
i ? ARGV[i + 1] : default
|
|
@@ -183,15 +191,25 @@ desc_changed = changed_files.any? { |f| f.end_with?("/SKILL.md") }
|
|
|
183
191
|
co_change = bank_changed && desc_changed
|
|
184
192
|
|
|
185
193
|
# --- build skill routing surface (the same descriptions the agent routes on) -
|
|
186
|
-
|
|
194
|
+
# One filtered list feeds BOTH the prompt and the set of selectable names, so a
|
|
195
|
+
# skill the prompt never offered (no frontmatter, empty description) cannot be a
|
|
196
|
+
# valid selection: review noted the two were built by separate transformations
|
|
197
|
+
# whose filtering could drift apart, and a name allowed but never shown is exactly
|
|
198
|
+
# the shape that drift would let a verdict claim.
|
|
199
|
+
# map + compact rather than filter_map: the runner has to work on the oldest
|
|
200
|
+
# Ruby a host ships (macOS system Ruby is 2.6), and filter_map is 2.7+.
|
|
201
|
+
catalog_entries = Dir[File.join(root, "skills", "*", "SKILL.md")].sort.map do |path|
|
|
187
202
|
name = File.basename(File.dirname(path))
|
|
188
203
|
m = File.read(path).match(/\A---\s*\n(.*?)\n---\s*\n/m)
|
|
189
204
|
next unless m
|
|
190
205
|
desc = (YAML.safe_load(m[1]) rescue {})["description"].to_s.strip
|
|
191
206
|
next if desc.empty?
|
|
192
207
|
desc = desc[0, desc_budget] if desc_budget && desc.length > desc_budget
|
|
193
|
-
"### #{name}\n#{desc}"
|
|
194
|
-
end.compact
|
|
208
|
+
[name, "### #{name}\n#{desc}"]
|
|
209
|
+
end.compact
|
|
210
|
+
catalog = catalog_entries.map(&:last).join("\n\n")
|
|
211
|
+
# The set of names a verdict may legitimately select; "none" is accepted separately.
|
|
212
|
+
catalog_names = catalog_entries.map(&:first)
|
|
195
213
|
|
|
196
214
|
# --- optional always-on entry-routing layer -----------------------------------
|
|
197
215
|
# The catalog above is the description-only surface. Hosts ALSO inject
|
|
@@ -401,6 +419,16 @@ tasks.each do |t|
|
|
|
401
419
|
break
|
|
402
420
|
end
|
|
403
421
|
selected = parsed && parsed["selected_skill"]
|
|
422
|
+
# A parseable answer is not yet a usable observation. The prompt's contract is
|
|
423
|
+
# an exact catalog name or "none"; anything else -- a missing key, a name the
|
|
424
|
+
# catalog does not carry, a non-string -- is grader output that says nothing
|
|
425
|
+
# about routing, and counting it as a verdict would let it feed the per-case
|
|
426
|
+
# resolution floor exactly as a well-formed one does. It is an ERROR, not a
|
|
427
|
+
# FAIL: a FAIL is evidence against the route, this is absence of evidence.
|
|
428
|
+
if parsed && error.nil? && !(selected == "none" || catalog_names.include?(selected))
|
|
429
|
+
error = "invalid_selection: #{selected.inspect[0, 80]}"
|
|
430
|
+
parsed = nil
|
|
431
|
+
end
|
|
404
432
|
clarify = parsed && parsed["clarify"] == true
|
|
405
433
|
confidence = parsed && parsed["confidence"]
|
|
406
434
|
v_status =
|
|
@@ -519,6 +547,30 @@ if baseline_path && File.file?(baseline_path)
|
|
|
519
547
|
end
|
|
520
548
|
end
|
|
521
549
|
|
|
550
|
+
# The floor is on VALID observations, not on the requested replica count: a run
|
|
551
|
+
# asked for ten replicas can come back with seven usable verdicts once a grader
|
|
552
|
+
# times out or returns unparsable output, and a report that called itself
|
|
553
|
+
# actionable on the request alone would license an edit the evidence cannot
|
|
554
|
+
# support. Observed in this repository: a fourteen-task run at --replicas 10
|
|
555
|
+
# returned six grader errors and left two tasks at seven valid observations.
|
|
556
|
+
#
|
|
557
|
+
# The verdict is PER CASE, because an edit is licensed per case. A report-wide
|
|
558
|
+
# flag alone is unsound in the other direction: a subset run over case A can be
|
|
559
|
+
# actionable while saying nothing about case B, and a consumer reading only the
|
|
560
|
+
# top-level boolean would take it as licence for an edit to B. Each result
|
|
561
|
+
# therefore carries its own `actionable`, and the report-level field is the
|
|
562
|
+
# conjunction over the cases the run actually measured -- true only when every
|
|
563
|
+
# measured case clears the floor, and never a statement about a case absent from
|
|
564
|
+
# `results`.
|
|
565
|
+
results.each do |r|
|
|
566
|
+
valid = r[:verdicts].count { |v| v[:status] != "ERROR" }
|
|
567
|
+
r[:valid_observations] = valid
|
|
568
|
+
r[:actionable] = replicas >= ACTION_RESOLUTION_MIN_REPLICAS &&
|
|
569
|
+
valid >= ACTION_RESOLUTION_MIN_REPLICAS
|
|
570
|
+
end
|
|
571
|
+
min_valid_observations = results.map { |r| r[:valid_observations] }.min.to_i
|
|
572
|
+
action_resolution = !results.empty? && results.all? { |r| r[:actionable] }
|
|
573
|
+
|
|
522
574
|
report = {
|
|
523
575
|
model: model, tasks: results.size, pass: passes, fail: fails.size, error: errors.size,
|
|
524
576
|
replicas: replicas, verdicts: all_observed.size,
|
|
@@ -526,6 +578,10 @@ report = {
|
|
|
526
578
|
clarify_count: clarify_count, low_confidence_count: low_conf_count,
|
|
527
579
|
replica_agreement: (replicas >= 2 ? { agree: agreement_agree, measured: agreement_measured } : nil),
|
|
528
580
|
desc_budget_chars: desc_budget, routing_surface: routing_surface,
|
|
581
|
+
action_resolution: action_resolution,
|
|
582
|
+
action_resolution_scope: "cases measured by this run only; see each result's actionable field",
|
|
583
|
+
action_resolution_min_replicas: ACTION_RESOLUTION_MIN_REPLICAS,
|
|
584
|
+
min_valid_observations: min_valid_observations,
|
|
529
585
|
co_change_bank_and_descriptions: co_change, co_change_check_available: co_change_check_ok,
|
|
530
586
|
frozen_drift: drift.map { |r| r[:id] },
|
|
531
587
|
baseline_comparable: baseline_comparable,
|
|
@@ -535,6 +591,9 @@ report = {
|
|
|
535
591
|
File.write(json_path, JSON.pretty_generate(report)) if json_path
|
|
536
592
|
|
|
537
593
|
puts "eval-routing-bank (#{model}): #{passes}/#{results.size} pass, #{fails.size} fail, #{errors.size} grader-error"
|
|
594
|
+
unless action_resolution
|
|
595
|
+
puts " \u26a0 screening_resolution_only: replicas=#{replicas}, weakest task has #{min_valid_observations} valid observations, floor #{ACTION_RESOLUTION_MIN_REPLICAS} — this report locates candidates, it does not license a description edit; a per-case edit needs #{ACTION_RESOLUTION_MIN_REPLICAS} valid observations of that case (references/eval-routing.md)"
|
|
596
|
+
end
|
|
538
597
|
puts " arm: desc-budget-chars=#{desc_budget}" if desc_budget
|
|
539
598
|
unless all_observed.empty?
|
|
540
599
|
line = " clarify: #{clarify_count}/#{all_observed.size} verdicts, low-confidence(<0.5): #{low_conf_count}/#{all_observed.size}"
|
|
@@ -52,19 +52,123 @@ LEDGER_PATH = "skills/skill-extraction-workflow/references/source-register.md"
|
|
|
52
52
|
# the routing-surface class below was one such patch) only re-instantiates it on
|
|
53
53
|
# the next input. The predicate now reads the round.
|
|
54
54
|
#
|
|
55
|
-
# Rounds are cut at the commits that touch the ledger itself, walked
|
|
56
|
-
#
|
|
57
|
-
#
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
#
|
|
55
|
+
# Rounds are cut at the commits that touch the ledger itself, walked along a
|
|
56
|
+
# first-parent line. The partition comes from git alone: an author cannot widen,
|
|
57
|
+
# move, or nominate their own scope.
|
|
58
|
+
#
|
|
59
|
+
# THE SAME PARTITION BEFORE AND AFTER THE MERGE. A branch is judged on its own
|
|
60
|
+
# first-parent line while it is a pull request (CI checks out the branch head),
|
|
61
|
+
# and once merged that whole line sits behind ONE first-parent step of the
|
|
62
|
+
# integration branch. Reading that step as one boundary gave the same history a
|
|
63
|
+
# different partition after it landed — every round on the branch collapsed into
|
|
64
|
+
# one, and a row whose validity depends on its round being narrow (a routing-
|
|
65
|
+
# surface `#description` anchor in a commit that changed nothing else) turned red
|
|
66
|
+
# without a byte of it changing. That is the "verdict moved after it landed"
|
|
67
|
+
# defect in a new coat, and it surfaced on every post-merge evaluation: the push
|
|
68
|
+
# build of the integration branch and the promotion pull request. So a merge that
|
|
69
|
+
# git itself reproduces from its two parents is EXPANDED in place: its second
|
|
70
|
+
# parent's line, from the fork point to the merged head, is walked with the same
|
|
71
|
+
# rule, recursively, and contributes exactly the rounds it had as a branch.
|
|
72
|
+
#
|
|
73
|
+
# A merge is expanded only when git can rebuild it — two parents, a tree equal to
|
|
74
|
+
# `git merge-tree --write-tree` of those parents, and a second parent that is
|
|
75
|
+
# neither already on the base nor already on the line being walked (a sync merge
|
|
76
|
+
# brings nothing that needs a round). Anything else — a hand-resolved merge, a
|
|
77
|
+
# conflicted one, an octopus — keeps today's single boundary at the merge, so
|
|
78
|
+
# content git did not derive from the parents is never left in no round.
|
|
79
|
+
ROUND_WALK_MAX_DEPTH = 8
|
|
80
|
+
ancestor_of = lambda do |commit, tip|
|
|
81
|
+
IO.popen(["git", "-C", root, "merge-base", "--is-ancestor", commit, tip], err: File::NULL, &:read)
|
|
82
|
+
status = $?.exitstatus
|
|
83
|
+
next true if status == 0
|
|
84
|
+
next false if status == 1
|
|
85
|
+
warn "impact_chain_git_failed: git merge-base --is-ancestor #{commit} #{tip} exited #{status}"
|
|
86
|
+
exit 1
|
|
87
|
+
end
|
|
88
|
+
# The tree git produces merging `second` into `first`; nil when that merge
|
|
89
|
+
# conflicts (whoever resolved it was not git). Needs git 2.38+, the same floor
|
|
90
|
+
# the review-ledger binder already requires for the identical invariant.
|
|
91
|
+
automatic_merge_tree = lambda do |first, second|
|
|
92
|
+
out = IO.popen(["git", "-C", root, "merge-tree", "--write-tree", first, second], err: File::NULL, &:read)
|
|
93
|
+
status = $?.exitstatus
|
|
94
|
+
next out.to_s.lines.first.to_s.strip if status == 0
|
|
95
|
+
next nil if status == 1
|
|
96
|
+
warn "impact_chain_git_failed: git merge-tree --write-tree #{first} #{second} exited #{status} (git 2.38 or newer is required)"
|
|
97
|
+
exit 1
|
|
98
|
+
end
|
|
99
|
+
# [first parent, second parent, fork point] when `commit` is a merge git can
|
|
100
|
+
# rebuild from its parents and whose second parent carries a line of its own;
|
|
101
|
+
# nil when the merge keeps today's single-boundary treatment. `line_base` is the
|
|
102
|
+
# base of the line being walked: a merge whose second parent is already below
|
|
103
|
+
# that base is a sync of what the line was cut from (the target advancing under
|
|
104
|
+
# a branch), and is a sync on the branch's own line exactly as it is on the
|
|
105
|
+
# integration line — judging it against the outer base alone would expand it
|
|
106
|
+
# during promotion and strand the branch's earlier work in a rowless span.
|
|
107
|
+
expandable_merge = lambda do |commit, line_base|
|
|
108
|
+
parents = git_read.call("rev-list", "--parents", "-n", "1", commit).split[1..] || []
|
|
109
|
+
next nil unless parents.length == 2
|
|
110
|
+
first, second = parents
|
|
111
|
+
next nil if ancestor_of.call(second, base_ref) || ancestor_of.call(second, line_base) || ancestor_of.call(second, first)
|
|
112
|
+
own_tree = git_read.call("rev-parse", "#{commit}^{tree}").strip
|
|
113
|
+
next nil unless automatic_merge_tree.call(first, second) == own_tree
|
|
114
|
+
# The fork point read FAILS CLOSED like every other git read here: a lookup
|
|
115
|
+
# that errored would otherwise read as "no fork point", skip the expansion,
|
|
116
|
+
# and hand the merge the collapsed span — the lenient verdict — on a git
|
|
117
|
+
# failure nobody sees. Exit 1 is git's own "no common ancestor" and means
|
|
118
|
+
# there is genuinely no line to expand from.
|
|
119
|
+
fork_out = IO.popen(["git", "-C", root, "merge-base", first, second], err: File::NULL, &:read)
|
|
120
|
+
fork_status = $?.exitstatus
|
|
121
|
+
next nil if fork_status == 1
|
|
122
|
+
unless fork_status == 0
|
|
123
|
+
warn "impact_chain_git_failed: git merge-base #{first} #{second} exited #{fork_status}"
|
|
124
|
+
exit 1
|
|
125
|
+
end
|
|
126
|
+
fork = fork_out.to_s.split("\n").first.to_s.strip
|
|
127
|
+
next nil if fork.empty?
|
|
128
|
+
[first, second, fork]
|
|
129
|
+
end
|
|
130
|
+
# Each round spans (previous boundary, this one] so the work commits that
|
|
61
131
|
# precede a ledger append are inside the round they belong to — landing the change
|
|
62
132
|
# and appending the row in separate commits is the normal shape, not an evasion.
|
|
63
|
-
|
|
64
|
-
# The trailing span — owner changes committed after the last ledger append — is a
|
|
133
|
+
# The trailing span — owner changes committed after the last boundary — is a
|
|
65
134
|
# round too. It holds no rows, so its owners fall through to the presence check
|
|
66
|
-
# and the gate still fails closed on undeclared work.
|
|
67
|
-
|
|
135
|
+
# and the gate still fails closed on undeclared work. Both hold on every line the
|
|
136
|
+
# walk visits, the integration branch and each expanded merge alike.
|
|
137
|
+
round_bounds_for = lambda do |from, to, depth|
|
|
138
|
+
if depth > ROUND_WALK_MAX_DEPTH
|
|
139
|
+
warn "impact_chain_round_walk_too_deep: merges nested more than #{ROUND_WALK_MAX_DEPTH} levels between #{from} and #{to}"
|
|
140
|
+
exit 1
|
|
141
|
+
end
|
|
142
|
+
list = lambda do |*options, pathspec|
|
|
143
|
+
git_read.call("rev-list", "--first-parent", "--reverse", *options, "#{from}..#{to}", *pathspec)
|
|
144
|
+
.split("\n").map(&:strip).reject(&:empty?)
|
|
145
|
+
end
|
|
146
|
+
line = list.call([])
|
|
147
|
+
ledger_heads = list.call(["--", LEDGER_PATH])
|
|
148
|
+
merges = list.call("--merges", [])
|
|
149
|
+
spans = []
|
|
150
|
+
prev = from
|
|
151
|
+
line.each do |commit|
|
|
152
|
+
expansion = merges.include?(commit) ? expandable_merge.call(commit, from) : nil
|
|
153
|
+
if expansion
|
|
154
|
+
first, second, fork = expansion
|
|
155
|
+
spans << [prev, first] unless prev == first
|
|
156
|
+
spans.concat(round_bounds_for.call(fork, second, depth + 1))
|
|
157
|
+
prev = commit
|
|
158
|
+
elsif ledger_heads.include?(commit)
|
|
159
|
+
spans << [prev, commit]
|
|
160
|
+
prev = commit
|
|
161
|
+
end
|
|
162
|
+
end
|
|
163
|
+
spans << [prev, to]
|
|
164
|
+
spans
|
|
165
|
+
end
|
|
166
|
+
round_bounds = round_bounds_for.call(base_ref, "HEAD", 0)
|
|
167
|
+
# Diagnostic only: print the partition so a verdict can be read against the
|
|
168
|
+
# rounds it was judged in. Off by default so no suite's output assertions move.
|
|
169
|
+
if ENV["CCL_IMPACT_CHAIN_TRACE_ROUNDS"] == "1"
|
|
170
|
+
round_bounds.each { |span_base, span_head| warn "impact_chain_round: #{span_base[0, 12]}..#{span_head[0, 12]}" }
|
|
171
|
+
end
|
|
68
172
|
# Everything a predicate needs to judge one span. Built lazily per span and
|
|
69
173
|
# memoized: a round whose rows are all RED-baseline never pays for the rename
|
|
70
174
|
# derivation.
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Reference-access census: which of a skill package's files were actually
|
|
3
|
+
# touched by real agent sessions on THIS host, over a lookback window.
|
|
4
|
+
#
|
|
5
|
+
# Why this exists: a rule set must not grow monotonically, but "retire what is
|
|
6
|
+
# not pulling its weight" needs a signal that is not the author's opinion. The
|
|
7
|
+
# closest observable proxy on a local host is the per-file access record in the
|
|
8
|
+
# agent's own session transcripts (Claude Code `~/.claude/projects/**/*.jsonl`,
|
|
9
|
+
# Codex `~/.codex/sessions/**/*.jsonl`): a reference that no session mentioned
|
|
10
|
+
# in N days is a relocation/retirement candidate; one mentioned in most sessions
|
|
11
|
+
# that loaded the skill is a candidate for promotion into the entrypoint. This
|
|
12
|
+
# is the local instantiation of the usage counters that context-evolution
|
|
13
|
+
# methods keep per bullet (helpful/harmful counts) and of the "ignored content
|
|
14
|
+
# is unnecessary or poorly signaled" observation in the official skill-authoring
|
|
15
|
+
# guidance — advisory only, never a gate (Goodhart: a count that becomes a
|
|
16
|
+
# target gets gamed by mentioning files).
|
|
17
|
+
#
|
|
18
|
+
# Privacy contract: the transcripts are private per-host data. This script
|
|
19
|
+
# prints ONLY repo-relative skill file paths, per-file session counts, and
|
|
20
|
+
# last-touched dates. It never prints transcript text, prompts, absolute paths
|
|
21
|
+
# outside the skill tree, or session ids. Its output is safe to paste into a
|
|
22
|
+
# private charter; it is still not shared-tree content by itself.
|
|
23
|
+
#
|
|
24
|
+
# Counting unit: a SESSION (one transcript file) counts once per skill file it
|
|
25
|
+
# mentions, whatever the tool (Read, sed, grep, Skill load). "Mentioned" is a
|
|
26
|
+
# superset of "read to depth" — treat a count as an upper bound on real use.
|
|
27
|
+
#
|
|
28
|
+
# Usage:
|
|
29
|
+
# reference-access-census.sh [--skill <name>] [--days <n>] [--repo-root <dir>]
|
|
30
|
+
# [--logs <dir>[,<dir>...]]
|
|
31
|
+
# Defaults: skill=skill-extraction-workflow, days=60, repo-root=cwd-derived,
|
|
32
|
+
# logs=$HOME/.claude/projects,$HOME/.codex/sessions
|
|
33
|
+
# Exit 0 on a completed census or an honest unevaluated result (no transcripts);
|
|
34
|
+
# exit 2 on usage errors and on input errors (unreadable/vanished/unexecutable
|
|
35
|
+
# inputs) — counts are withheld rather than printed as zeros.
|
|
36
|
+
set -euo pipefail
|
|
37
|
+
|
|
38
|
+
skill="skill-extraction-workflow"
|
|
39
|
+
days=60
|
|
40
|
+
repo_root=""
|
|
41
|
+
logs="${HOME:+${HOME}/.claude/projects,${HOME}/.codex/sessions}"
|
|
42
|
+
logs_explicit=0
|
|
43
|
+
while [ $# -gt 0 ]; do
|
|
44
|
+
case "$1" in
|
|
45
|
+
--skill|--days|--repo-root|--logs)
|
|
46
|
+
if [ $# -lt 2 ]; then echo "reference_access_census_usage_error: $1 needs a value" >&2; exit 2; fi ;;
|
|
47
|
+
esac
|
|
48
|
+
case "$1" in
|
|
49
|
+
--skill) skill="$2"; shift 2 ;;
|
|
50
|
+
--days) days="$2"; shift 2 ;;
|
|
51
|
+
--repo-root) repo_root="$2"; shift 2 ;;
|
|
52
|
+
--logs) logs="$2"; logs_explicit=1; shift 2 ;;
|
|
53
|
+
-h|--help) sed -n '2,32p' "$0"; exit 0 ;;
|
|
54
|
+
*) echo "reference_access_census_usage_error: unknown argument (see --help)" >&2; exit 2 ;;
|
|
55
|
+
esac
|
|
56
|
+
done
|
|
57
|
+
if [ -z "$repo_root" ]; then
|
|
58
|
+
repo_root="$(git rev-parse --show-toplevel 2>/dev/null || pwd)"
|
|
59
|
+
fi
|
|
60
|
+
skill_dir="$repo_root/skills/$skill"
|
|
61
|
+
if [ ! -d "$skill_dir" ]; then
|
|
62
|
+
echo "reference_access_census_usage_error: --skill names no tracked skill package under the repo root" >&2; exit 2
|
|
63
|
+
fi
|
|
64
|
+
case "$days" in ''|*[!0-9]*) echo "reference_access_census_usage_error: --days must be an integer" >&2; exit 2 ;; esac
|
|
65
|
+
|
|
66
|
+
# Inventory: every tracked markdown file in the package (entrypoint + references).
|
|
67
|
+
files=()
|
|
68
|
+
while IFS= read -r f; do files+=("$f"); done < <(cd "$repo_root" && git ls-files "skills/$skill/SKILL.md" "skills/$skill/references/*.md" 2>/dev/null | sort)
|
|
69
|
+
if [ "${#files[@]}" -eq 0 ]; then
|
|
70
|
+
echo "reference_access_census_usage_error: --skill names a package with no tracked SKILL.md or references" >&2; exit 2
|
|
71
|
+
fi
|
|
72
|
+
|
|
73
|
+
# Candidate transcripts: any .jsonl under the log roots modified within the window.
|
|
74
|
+
tmp="$(mktemp -d)"; trap 'rm -rf "$tmp"' EXIT
|
|
75
|
+
roots=(); if [ -n "$logs" ]; then IFS=',' read -r -a roots <<<"$logs"; fi
|
|
76
|
+
: > "$tmp/candidates0"; : > "$tmp/errors"
|
|
77
|
+
# Input errors are never zeros: every scan phase collects stderr, and a non-empty
|
|
78
|
+
# error log withholds the table (exit 2) instead of printing counts that would
|
|
79
|
+
# read as evidence. grep's no-match status writes nothing to stderr, so it never
|
|
80
|
+
# trips this; unreadable, vanished, or unexecutable inputs do. The error log is
|
|
81
|
+
# summarized as a count only — file names would be private paths.
|
|
82
|
+
withhold_if_errors() {
|
|
83
|
+
if [ -s "$tmp/errors" ]; then
|
|
84
|
+
n="$(wc -l < "$tmp/errors" | tr -d ' ')"
|
|
85
|
+
echo "reference_access_census_unevaluated: $n input error(s) while scanning transcripts (unreadable, vanished, or unexecutable inputs); counts withheld"
|
|
86
|
+
exit 2
|
|
87
|
+
fi
|
|
88
|
+
}
|
|
89
|
+
# A default root that is absent is normal (the host may run only one agent); an
|
|
90
|
+
# explicitly supplied root that is absent is an input error — the caller asked
|
|
91
|
+
# for a scan that cannot happen, and skipping it would read as a real count.
|
|
92
|
+
for r in "${roots[@]}"; do
|
|
93
|
+
[ -n "$r" ] || continue
|
|
94
|
+
if [ ! -d "$r" ]; then
|
|
95
|
+
if [ "$logs_explicit" -eq 1 ]; then echo "missing explicit log root" >> "$tmp/errors"; fi
|
|
96
|
+
continue
|
|
97
|
+
fi
|
|
98
|
+
find "$r" -type f -name '*.jsonl' -mtime "-$days" -print0 >> "$tmp/candidates0" 2>> "$tmp/errors" || echo "transcript listing failed under one log root" >> "$tmp/errors"
|
|
99
|
+
done
|
|
100
|
+
withhold_if_errors
|
|
101
|
+
# The inventory is NUL-delimited (a pathname may contain a newline) and
|
|
102
|
+
# deduplicated by pathname: overlapping roots (a root and its own subtree, or
|
|
103
|
+
# the same root twice) list a transcript more than once, and a session must
|
|
104
|
+
# count once. Two names for one file (symlinked roots) are not canonicalized.
|
|
105
|
+
sort -zu "$tmp/candidates0" -o "$tmp/candidates0"
|
|
106
|
+
total_sessions="$(tr -cd '\0' < "$tmp/candidates0" | wc -c | tr -d ' ')"
|
|
107
|
+
if [ "$total_sessions" -eq 0 ]; then
|
|
108
|
+
echo "reference_access_census_unevaluated: no transcripts within ${days}d under ${#roots[@]} log root(s); counts withheld"
|
|
109
|
+
exit 0
|
|
110
|
+
fi
|
|
111
|
+
|
|
112
|
+
# Sessions that mention the package at all (fixed-string prefilter keeps this fast).
|
|
113
|
+
# xargs batches keep this under ARG_MAX; grep exit 1 (no match in a batch) is not an error.
|
|
114
|
+
# grep exit 1 (no match in a batch) is success; any other non-zero status —
|
|
115
|
+
# with or without a stderr line — is an input error, so each batch runs under a
|
|
116
|
+
# small wrapper that normalizes 1 to 0 and lets xargs report anything else.
|
|
117
|
+
# `|| rc=$?` keeps the batch alive when the caller exported SHELLOPTS=errexit:
|
|
118
|
+
# a bare `grep; rc=$?` would exit the child at status 1 before normalizing it.
|
|
119
|
+
grep_batch() { local rc=0; grep -lF --null "$@" || rc=$?; [ "$rc" -le 1 ] && return 0; return "$rc"; }
|
|
120
|
+
export -f grep_batch 2>/dev/null || true
|
|
121
|
+
if ! xargs -0 -n 200 bash -c 'grep_batch "$@"' _ "skills/$skill/" < "$tmp/candidates0" > "$tmp/touching0" 2>> "$tmp/errors"; then
|
|
122
|
+
echo "transcript scan failed in at least one batch" >> "$tmp/errors"
|
|
123
|
+
fi
|
|
124
|
+
withhold_if_errors
|
|
125
|
+
touching="$(tr -cd '\0' < "$tmp/touching0" | wc -c | tr -d ' ')"
|
|
126
|
+
|
|
127
|
+
# Per-file session count + last-touched date (mtime of the newest touching transcript).
|
|
128
|
+
: > "$tmp/rows"
|
|
129
|
+
for f in "${files[@]}"; do
|
|
130
|
+
rel="${f#skills/$skill/}"
|
|
131
|
+
if [ "$touching" -eq 0 ]; then n=0; last="-"; else
|
|
132
|
+
if ! xargs -0 -n 200 bash -c 'grep_batch "$@"' _ "$f" < "$tmp/touching0" > "$tmp/hits0" 2>> "$tmp/errors"; then
|
|
133
|
+
echo "transcript scan failed in at least one batch" >> "$tmp/errors"
|
|
134
|
+
fi
|
|
135
|
+
withhold_if_errors
|
|
136
|
+
n="$(tr -cd '\0' < "$tmp/hits0" | wc -c | tr -d ' ')"
|
|
137
|
+
if [ "$n" -gt 0 ]; then
|
|
138
|
+
# BSD stat first, GNU stat as the fallback; only the fallback's failure is an
|
|
139
|
+
# error. Every substitution is guarded so a failing pipeline cannot abort the
|
|
140
|
+
# script under set -e before withhold_if_errors runs.
|
|
141
|
+
last="$(xargs -0 stat -f '%Sm' -t '%Y-%m-%d' < "$tmp/hits0" 2>/dev/null | sort | tail -1)" || last=""
|
|
142
|
+
if [ -z "$last" ]; then
|
|
143
|
+
last="$(xargs -0 stat -c '%y' < "$tmp/hits0" 2>> "$tmp/errors" | cut -c1-10 | sort | tail -1)" || last=""
|
|
144
|
+
fi
|
|
145
|
+
if [ -z "$last" ]; then
|
|
146
|
+
[ -s "$tmp/errors" ] || echo "stat produced no timestamp for a touching transcript" >> "$tmp/errors"
|
|
147
|
+
withhold_if_errors
|
|
148
|
+
fi
|
|
149
|
+
else last="-"; fi
|
|
150
|
+
fi
|
|
151
|
+
if [ "$touching" -gt 0 ]; then share="$(( n * 100 / touching ))%"; else share="-"; fi
|
|
152
|
+
printf '%s | %s | %s | %s\n' "$rel" "$n" "$last" "$share" >> "$tmp/rows"
|
|
153
|
+
done
|
|
154
|
+
printf '%s\n' "reference_access_census: skill=$skill window=${days}d transcripts=$total_sessions sessions_touching_package=$touching"
|
|
155
|
+
printf '%s\n' "file | sessions | last_touched | share_of_touching"
|
|
156
|
+
sort -t'|' -k2,2nr "$tmp/rows"
|
|
157
|
+
echo "reference_access_census_ok"
|