@ccoalm/ccl-skills 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +8 -11
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +9 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +29 -31
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +32 -0
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
  33. package/dist/assets/release.json +79 -24
  34. package/package.json +1 -1
@@ -21,6 +21,18 @@
21
21
  # advisory = no signal). A agent-context/session-start.md absent from base is treated as a NEW
22
22
  # every-session injection and blocks like a new severe entrypoint.
23
23
  #
24
+ # Blocking (delta-scoped, skills/*/references/**/*.md — the reference-file line
25
+ # ratchet; design invariants in references/attention-budget-ratchet.md):
26
+ # - a changed reference that is NEW or CROSSES over 500 physical lines (base
27
+ # was not over) => block ("new reference over line budget")
28
+ # - a changed reference already over 500 lines in base that GREW
29
+ # (head_lines > base_lines) => block ("over-limit reference grew")
30
+ # references/source-register.md is structurally excluded (append-only ledger;
31
+ # its growth is governed by the append-only contract, not a line cap) — the
32
+ # exclusion lives in this gate, never in the candidate. A NEW reference over
33
+ # 100 lines with no "##" section headings draws an advisory token (never
34
+ # blocks). Rename credit is path-paired and non-growing, same as entrypoints.
35
+ #
24
36
  # Advisory (never blocks on its own): the recommended-band body-char debt
25
37
  # counters, severe-debt counters, per-file deltas, and the agent-context/session-start.md 13000B
26
38
  # byte tripwire BAND remain VISIBILITY / DEBT-MANAGEMENT only (the band flags the
@@ -113,6 +125,11 @@ end
113
125
  WORD_BUDGET_MAX = 5000
114
126
  WORD_TOKEN_PATTERN = /\p{Han}|\uFFFD|(?:(?!\p{Han})[\p{L}\p{N}])+(?:[\u0027\u2019_-](?:(?!\p{Han})[\p{L}\p{N}])+)*/u
115
127
 
128
+ REF_LINE_MAX = 500
129
+ REF_NAV_MIN_LINES = 100
130
+ REF_PATH_RE = %r{\Askills/[^/]+/references/.+\.md\z}
131
+ REF_LEDGER_RE = %r{\Askills/[^/]+/references/source-register\.md\z}
132
+
116
133
  Metric = Struct.new(:body_chars, :body_words, :bytes, keyword_init: true)
117
134
 
118
135
  def metric_for_text(text)
@@ -195,6 +212,42 @@ def head_metric_for_path(root, rel)
195
212
  metric_for_file(full)
196
213
  end
197
214
 
215
+ def ref_info_for_text(raw)
216
+ text = raw.b.dup.force_encoding(Encoding::UTF_8).scrub { |invalid| "�" * invalid.bytesize }
217
+ # Line-ending normalization comes first: a CR-only or CRLF file must measure the
218
+ # same as the LF file a reader sees, otherwise a 501-line CR-delimited file reads
219
+ # as one line and earns a clean verdict. Headings count EXACT H2 only, because an
220
+ # H3-only file gives a chunked read no section structure to navigate by.
221
+ text = text.gsub(/\r\n?/, "\n")
222
+ lines = text.lines
223
+ { lines: lines.length, headings: lines.count { |l| l.match?(/\A##[ \t]/) } }
224
+ end
225
+
226
+ def base_ref_info_for_path(root, base, rel)
227
+ return :unknown if base.nil?
228
+ return :missing unless git_success?("cat-file", "-e", "#{base}:#{rel}", root: root)
229
+ ref_info_for_text(git_text("show", "#{base}:#{rel}", root: root))
230
+ end
231
+
232
+ def head_ref_info_for_path(root, rel)
233
+ full = File.join(root, rel)
234
+ return :missing unless File.file?(full)
235
+ ref_info_for_text(File.binread(full))
236
+ end
237
+
238
+ def fmt_ref_value(info)
239
+ case info
240
+ when :unknown then "unknown"
241
+ when :missing then "missing"
242
+ else info[:lines].to_s
243
+ end
244
+ end
245
+
246
+ def fmt_ref_delta(base_info, head_info)
247
+ return "unknown" unless base_info.is_a?(Hash) && head_info.is_a?(Hash)
248
+ format("%+d", head_info[:lines] - base_info[:lines])
249
+ end
250
+
198
251
  def fmt_metric_value(metric, field)
199
252
  case metric
200
253
  when :unknown then "unknown"
@@ -437,6 +490,144 @@ if bootstrap_changed
437
490
  # :missing (deleted) => nothing left to size-gate, same as a deleted entrypoint
438
491
  end
439
492
 
493
+ # --- reference-file line-budget ratchet (delta-scoped) ------------------------
494
+ # Design invariants: references/attention-budget-ratchet.md. Physical lines are
495
+ # the stable proxy metric; the append-only ledger references/source-register.md
496
+ # is excluded by gate design (its growth is governed by the append-only
497
+ # contract) and the exclusion lives HERE so a candidate cannot nominate its own
498
+ # exemption.
499
+ ref_blocks = []
500
+ ref_partials = []
501
+ ref_changed = []
502
+ ref_scan_error = nil
503
+ begin
504
+ if system("git", "-C", root, "rev-parse", "--is-inside-work-tree", out: File::NULL, err: File::NULL)
505
+ rpaths = []
506
+ rpaths.concat(git_lines("diff", "--name-only", base, "HEAD", "--", "skills", root: root)) if base
507
+ rpaths.concat(git_lines("diff", "--name-only", "--", "skills", root: root))
508
+ rpaths.concat(git_lines("diff", "--cached", "--name-only", "--", "skills", root: root))
509
+ rpaths.concat(git_lines("ls-files", "--others", "--exclude-standard", "--", "skills", root: root))
510
+ ref_changed = rpaths.select { |p| p.match?(REF_PATH_RE) }.reject { |p| p.match?(REF_LEDGER_RE) }.uniq
511
+ end
512
+ rescue StandardError => e
513
+ ref_scan_error = "#{e.class.name} #{e.message.lines.first.to_s.strip}"
514
+ warn "changed_reference_line_scan_skipped: #{ref_scan_error}"
515
+ end
516
+ # A scan failure must not silently read as "no references changed" — that would
517
+ # false-green a real growth, so it fails closed as a partial.
518
+ ref_partials << "skills/*/references: changed-detection failed (#{ref_scan_error}) — line-budget check unavailable (fail-closed)" if ref_scan_error
519
+
520
+ ref_move_map = {}
521
+ begin
522
+ if system("git", "-C", root, "rev-parse", "--is-inside-work-tree", out: File::NULL, err: File::NULL)
523
+ status_args = []
524
+ status_args << ["diff", "--name-status", base, "HEAD"] if base
525
+ status_args << ["diff", "--name-status", "--cached"]
526
+ status_args << ["diff", "--name-status"]
527
+ status_args.each do |cmd|
528
+ git_lines(*cmd, "--", "skills", root: root).each do |line|
529
+ status, *paths = line.split("\t")
530
+ next unless status.start_with?("R") && paths.length >= 2
531
+ next unless status[1..].to_i >= 90
532
+ old_rel = paths[0].match?(REF_PATH_RE) ? paths[0] : nil
533
+ new_rel = paths[1].match?(REF_PATH_RE) ? paths[1] : nil
534
+ ref_move_map[new_rel] = old_rel if old_rel && new_rel
535
+ end
536
+ end
537
+ end
538
+ rescue StandardError => e
539
+ warn "reference_move_pair_scan_skipped: #{e.class.name}"
540
+ end
541
+
542
+ # The census is visibility only, so an unreadable or vanished file must degrade the
543
+ # COUNTER to unknown rather than abort the program: aborting here would skip the
544
+ # per-file fail-closed partials below and emit no verdict token at all.
545
+ head_ref_over_count = 0
546
+ head_ref_census_error = nil
547
+ begin
548
+ Dir[File.join(root, "skills", "*", "references", "**", "*.md")].each do |path|
549
+ rel = path.start_with?(root + "/") ? path[(root.length + 1)..] : path.sub(%r{\A\./}, "")
550
+ next unless rel.match?(REF_PATH_RE)
551
+ next if rel.match?(REF_LEDGER_RE)
552
+ info = begin
553
+ head_ref_info_for_path(root, rel)
554
+ rescue StandardError => e
555
+ head_ref_census_error ||= "#{rel}: #{e.class.name}"
556
+ next
557
+ end
558
+ head_ref_over_count += 1 if info.is_a?(Hash) && info[:lines] > REF_LINE_MAX
559
+ end
560
+ rescue StandardError => e
561
+ head_ref_census_error ||= e.class.name
562
+ end
563
+ if head_ref_census_error
564
+ warn "reference_line_census_partial: head over-limit census incomplete (#{head_ref_census_error}) — counter reported unknown; per-file verdicts below are unaffected"
565
+ head_ref_over_count = nil
566
+ end
567
+ base_ref_over_count = nil
568
+ if base
569
+ begin
570
+ base_ref_over_count = 0
571
+ git_lines("grep", "-c", "-e", "", base, "--", "skills", root: root).each do |line|
572
+ rest, _, cnt = line.rpartition(":")
573
+ _, _, rel = rest.partition(":")
574
+ next unless rel.match?(REF_PATH_RE)
575
+ next if rel.match?(REF_LEDGER_RE)
576
+ base_ref_over_count += 1 if cnt.to_i > REF_LINE_MAX
577
+ end
578
+ rescue StandardError => e
579
+ base_ref_over_count = nil
580
+ warn "reference_line_trend_partial: base=unknown reason=#{e.class.name}"
581
+ end
582
+ end
583
+ puts "reference_line_over_limit_count_base=#{base_ref_over_count.nil? ? "unknown" : base_ref_over_count}"
584
+ puts "reference_line_over_limit_count_head=#{head_ref_over_count.nil? ? "unknown" : head_ref_over_count}"
585
+ puts "reference_line_over_limit_count_delta=#{(base_ref_over_count.nil? || head_ref_over_count.nil?) ? "unknown" : format("%+d", head_ref_over_count - base_ref_over_count)}"
586
+
587
+ # Ledger visibility: excluded from the ratchet, still reported when over.
588
+ Dir[File.join(root, "skills", "*", "references", "source-register.md")].sort.each do |path|
589
+ rel = path.start_with?(root + "/") ? path[(root.length + 1)..] : path.sub(%r{\A\./}, "")
590
+ info = begin
591
+ head_ref_info_for_path(root, rel)
592
+ rescue StandardError
593
+ :missing
594
+ end
595
+ if info.is_a?(Hash) && info[:lines] > REF_LINE_MAX
596
+ puts "reference_line_budget_ledger_excluded: #{rel} lines=#{info[:lines]} — append-only ledger, excluded by gate design (growth governed by the append-only contract)"
597
+ end
598
+ end
599
+
600
+ ref_changed.sort.each do |rel|
601
+ begin
602
+ head_info = head_ref_info_for_path(root, rel)
603
+ rescue StandardError => e
604
+ ref_partials << "#{rel}: head lines unreadable (#{e.class.name}) — line-budget check unavailable (fail-closed)"
605
+ next
606
+ end
607
+ next if head_info == :missing # deleted reference: nothing left to gate
608
+ base_info = base_ref_info_for_path(root, base, rel)
609
+ puts "changed_reference_line_delta: #{rel} base_lines=#{fmt_ref_value(base_info)} head_lines=#{head_info[:lines]} delta_lines=#{fmt_ref_delta(base_info, head_info)}"
610
+ if base_info == :missing && head_info[:lines] > REF_NAV_MIN_LINES && head_info[:headings].zero?
611
+ warn "reference_nav_advisory: #{rel} lines=#{head_info[:lines]} headings=0 — a new reference over #{REF_NAV_MIN_LINES} lines needs ## section structure so chunked reads and greps can navigate (advisory, never blocks)"
612
+ end
613
+ next unless head_info[:lines] > REF_LINE_MAX
614
+ if base_info == :unknown
615
+ ref_partials << "#{rel}: base unknown — reference line-budget check unavailable (fail-closed)"
616
+ elsif base_info == :missing || base_info[:lines] <= REF_LINE_MAX
617
+ source_rel = ref_move_map[rel]
618
+ source_info = source_rel ? base_ref_info_for_path(root, base, source_rel) : nil
619
+ if source_info.is_a?(Hash) && source_info[:lines] > REF_LINE_MAX && head_info[:lines] <= source_info[:lines]
620
+ puts "reference_line_budget_move_ok: #{rel} head_lines=#{head_info[:lines]} allowed_lines=#{source_info[:lines]} moved_from=#{source_rel}"
621
+ else
622
+ ref_blocks << "#{rel}: new reference over line budget head_lines=#{head_info[:lines]} (> #{REF_LINE_MAX}) — split it by subtopic before landing (see references/attention-budget-ratchet.md; there is no exempt marker or waiver flag)"
623
+ end
624
+ elsif head_info[:lines] > base_info[:lines]
625
+ ref_blocks << "#{rel}: over-limit reference grew base_lines=#{base_info[:lines]} head_lines=#{head_info[:lines]} — shrink or stay level; fund additions by consolidating text in the same file"
626
+ else
627
+ puts "reference_line_budget_legacy_ok: #{rel} base_lines=#{base_info[:lines]} head_lines=#{head_info[:lines]} allowed_lines=#{base_info[:lines]}"
628
+ end
629
+ end
630
+
440
631
  # Advisory markers only when the blocking verdict is clean — a consumer reading
441
632
  # the preserved legacy token must never see ok next to a block.
442
633
  #
@@ -451,18 +642,20 @@ end
451
642
  # — but the token says un-evaluated, not ok.
452
643
  # NOTE: this comment lives inside the single-quoted `ruby -e` program; an
453
644
  # apostrophe here terminates the shell quote and breaks the script.
454
- if blocks.empty? && partials.empty? && word_blocks.empty? && word_partials.empty?
645
+ if blocks.empty? && partials.empty? && word_blocks.empty? && word_partials.empty? && ref_blocks.empty? && ref_partials.empty?
455
646
  puts "size_budget_advisory_#{size_state}"
456
647
  puts "entrypoint_size_budget_advisory_ok"
457
648
  if base.nil?
458
649
  puts "entrypoint_size_blocking_unevaluated: base=unknown — committed changes were NOT delta-checked; this is not a pass"
459
650
  puts "entrypoint_word_budget_blocking_unevaluated: base=unknown — committed changes were NOT delta-checked; this is not a pass"
651
+ puts "reference_line_budget_blocking_unevaluated: base=unknown — committed changes were NOT delta-checked; this is not a pass"
460
652
  if File.file?(File.join(root, "agent-context/session-start.md"))
461
653
  puts "bootstrap_size_delta_unevaluated: base=unknown — agent-context/session-start.md committed changes were NOT delta-checked; this is not a pass"
462
654
  end
463
655
  else
464
656
  puts "entrypoint_size_blocking_ok"
465
657
  puts "entrypoint_word_budget_blocking_ok"
658
+ puts "reference_line_budget_blocking_ok"
466
659
  end
467
660
  exit 0
468
661
  end
@@ -470,7 +663,10 @@ blocks.each { |b| warn "entrypoint_size_block: #{b}" }
470
663
  partials.each { |b| warn "entrypoint_size_block_partial: #{b}" }
471
664
  word_blocks.each { |b| warn "entrypoint_word_budget_block: #{b}" }
472
665
  word_partials.each { |b| warn "entrypoint_word_budget_block_partial: #{b}" }
666
+ ref_blocks.each { |b| warn "reference_line_block: #{b}" }
667
+ ref_partials.each { |b| warn "reference_line_block_partial: #{b}" }
473
668
  puts "entrypoint_word_budget_blocking_failed" unless word_blocks.empty? && word_partials.empty?
669
+ puts "reference_line_budget_blocking_failed" unless ref_blocks.empty? && ref_partials.empty?
474
670
  # Legacy aggregate token for every blocking verdict owned by this size-budget
475
671
  # script, including the body-word rule. Keep it unconditional so existing
476
672
  # consumers cannot miss a new word-only failure.
@@ -0,0 +1,15 @@
1
+ # Contract-anchor table for check-contract-anchors.sh (same directory).
2
+ # One row per pinned load-bearing contract literal:
3
+ # id <TAB> repo-relative path <TAB> pinned literal (exactly-once, >=16 chars) <TAB> note
4
+ # The note names the sanitized provenance label and why the wording is load-bearing.
5
+ # To change pinned wording intentionally: edit the contract sentence AND this row
6
+ # in the same MR. Keep this set small — load-bearing contracts only, never style.
7
+ verdict-taxonomy-discriminator skills/testing-strategy/references/ci-fixtures-and-flake-control.md One discriminating predicate decides the verdict 071-chainA-r1f1: fault-origin discriminator sentence; deleting it reverts verdicts to case-by-case judgment
8
+ verdict-taxonomy-false-green skills/testing-strategy/references/ci-fixtures-and-flake-control.md excludes `infra-error` cases to present a clean total is a false-green report 071-chainA-r1f1: report-completeness clause; deleting it legalizes infra-error-excluding clean totals
9
+ stop-predicate-materially-differing skills/product-rd-workflow/SKILL.md materially differing viable approaches (none dominant-and-reversible) 071-chainB-r1f1: stop-condition predicate wording; inverting it flips blocked/continue classification
10
+ stop-predicate-evidenced-cause skills/product-rd-workflow/SKILL.md a fix lacking evidenced cause 071-chainB-r1f1: stop-condition predicate wording; weakening it admits plausible-but-unevidenced fixes
11
+ burn-rate-tiers-entry skills/platform-observability/SKILL.md page 14.4× 1h/5m, page 6× 6h/30m, ticket 1× 3d/6h 071-chainC-r1f4: SRE Workbook multiwindow tiers, externally verified value (specs/071 source-verification); silent renumbering must not pass
12
+ burn-rate-page-row-reference skills/platform-observability/references/sli-slo-design.md | Page (P0) | 1h | 5m | 14.4 | 071-chainC-r1f4: reference-side copy of the page tier; pinning both sides makes entry/reference drift visible
13
+ coverage-tier-provenance skills/testing-strategy/references/test-code-authoring-patterns.md 60% acceptable / 75% commendable / 90% exemplary 分档 071-chainC-r1f4: externally verified coverage tiers (specs/071 source-verification)
14
+ pairwise-trigger-range skills/test-artifact-management/references/classical-test-design-techniques.md 2-way 累计触发 53–97% 071-chainC-r1f4: NIST SP 800-142 empirical range, externally verified (specs/071 source-verification)
15
+ bva-two-vs-three-value skills/test-artifact-management/references/classical-test-design-techniques.md 2-value(边界 + 下一格)和 3-value(边界 + 两侧) 071-chainC-r1f4: ISTQB v4 BVA variant definitions, externally verified (specs/071 source-verification)
@@ -16,13 +16,23 @@
16
16
  # Usage:
17
17
  # eval-routing-bank.rb <repo-root> [--bank <path>] [--model <name>] [--limit N]
18
18
  # [--dry-run] [--json <path>] [--baseline <path>] [--timeout S]
19
- # [--desc-budget-chars N] [--with-bootstrap]
19
+ # [--desc-budget-chars N] [--with-bootstrap] [--replicas N]
20
20
  #
21
21
  # --desc-budget-chars N simulates a consumer that truncates each skill description
22
22
  # to its first N characters before routing (e.g. Codex compresses the skill listing
23
23
  # under a ~2%-of-context budget, 8000 chars when the window is unknown, shortening
24
24
  # descriptions first). Run the bank once plain and once with a budget to see which
25
25
  # routes only survive on the description tail — those triggers need front-loading.
26
+ #
27
+ # expected_skill "none" marks negative controls (out-of-library utterances) and
28
+ # coverage-gap probes (in-domain utterances no skill owns): the correct outcome
29
+ # is rejection, and a catalog skill claiming them is labeled "absorbed".
30
+ # --replicas N grades each task N times: task status is the conservative
31
+ # consensus (every observed replica must PASS), replica top1 agreement is
32
+ # reported, and disagreeing replicas label the task "ownership_split". clarify
33
+ # and low-confidence counts are first-class report fields either way.
34
+ # Reports from different (bank, replicas) configurations are different rulers —
35
+ # do not diff them as a regression signal.
26
36
  # Exit: 0 = ran (advisory); 2 = usage error; 3 = grader entirely unavailable.
27
37
 
28
38
  require "yaml"
@@ -39,7 +49,7 @@ end
39
49
 
40
50
  root = ARGV[0]
41
51
  if root.nil? || root.start_with?("-")
42
- warn "usage: eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--timeout S] [--desc-budget-chars N]"
52
+ warn "usage: eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--timeout S] [--desc-budget-chars N] [--replicas N]"
43
53
  exit 2
44
54
  end
45
55
  bank_path = arg("--bank", File.join(root, "eval", "routing-tasks.jsonl"))
@@ -65,6 +75,17 @@ if ARGV.include?("--desc-budget-chars")
65
75
  desc_budget = b.to_i
66
76
  end
67
77
  timeout_s = (t = arg("--timeout")) ? t.to_i : 60
78
+ replicas = 1
79
+ if ARGV.include?("--replicas")
80
+ r = arg("--replicas")
81
+ # Same strictness as --desc-budget-chars: a silent to_i coercion would turn a
82
+ # typo into "1 replica" and the agreement metric would quietly measure nothing.
83
+ unless r && r =~ /\A[1-9]\d*\z/
84
+ warn "--replicas requires a positive integer value, got #{r.inspect}"
85
+ exit 2
86
+ end
87
+ replicas = r.to_i
88
+ end
68
89
 
69
90
  unless File.file?(bank_path)
70
91
  warn "eval_bank_missing: #{bank_path}"
@@ -89,16 +110,57 @@ end
89
110
  tasks = tasks.first(limit) if limit
90
111
 
91
112
  # Schema validation: a task must be answerable, and must not be self-contradictory.
113
+ # expected_skill "none" is the negative-control / coverage-gap sentinel: the
114
+ # correct routing outcome is that NO catalog skill claims the utterance.
115
+ all_outcomes = Dir[File.join(root, "skills", "*", "SKILL.md")]
116
+ .map { |p| File.basename(File.dirname(p)) } + ["none"]
92
117
  tasks.each do |t|
93
118
  id = t["id"] || "(no id)"
94
119
  if t["utterance"].to_s.strip.empty? || t["expected_skill"].to_s.strip.empty?
95
120
  warn "eval_bank_invalid_task: #{id}: utterance and expected_skill are required"
96
121
  exit 2
97
122
  end
123
+ # Type-check the ORIGINAL values FIRST: a present-but-non-list field (e.g. ""
124
+ # or a bare string or a number) must fail as a usage error before any
125
+ # membership check touches it — Ruby strings are truthy and respond to
126
+ # include? (substring semantics), and non-strings would raise a bare
127
+ # NoMethodError instead of the documented invalid-bank diagnostic.
128
+ %w[must_not_route_to acceptable].each do |f|
129
+ next unless t.key?(f)
130
+ unless t[f].is_a?(Array)
131
+ warn "eval_bank_invalid_task: #{id}: #{f} must be a list, got #{t[f].inspect}"
132
+ exit 2
133
+ end
134
+ end
98
135
  if (t["must_not_route_to"] || []).include?(t["expected_skill"])
99
136
  warn "eval_bank_invalid_task: #{id}: expected_skill is also in must_not_route_to (impossible)"
100
137
  exit 2
101
138
  end
139
+ if (t["must_not_route_to"] || []).include?("none")
140
+ warn "eval_bank_invalid_task: #{id}: \"none\" is a sentinel outcome, not a routable target for must_not_route_to"
141
+ exit 2
142
+ end
143
+ # Optional acceptable[] names defensible alternate outcomes (a skill name or
144
+ # "none") for utterances with more than one correct route — e.g. a coverage-gap
145
+ # ask where both coordinator intake and rejection are right. It must not
146
+ # restate expected_skill or contradict must_not_route_to.
147
+ acc = t["acceptable"] || []
148
+ if acc.include?(t["expected_skill"])
149
+ warn "eval_bank_invalid_task: #{id}: acceptable restates expected_skill"
150
+ exit 2
151
+ end
152
+ unless (acc & (t["must_not_route_to"] || [])).empty?
153
+ warn "eval_bank_invalid_task: #{id}: acceptable and must_not_route_to overlap (contradictory)"
154
+ exit 2
155
+ end
156
+ # Anti-gaming: expected + acceptable must leave at least one outcome that
157
+ # would FAIL the row. A row covering the complete catalog-and-none outcome
158
+ # space passes on every valid grader selection — a vacuous fixture that fakes
159
+ # green regardless of description behavior.
160
+ if (all_outcomes - ([t["expected_skill"]] + acc)).empty?
161
+ warn "eval_bank_invalid_task: #{id}: expected_skill plus acceptable cover every possible outcome (vacuous row)"
162
+ exit 2
163
+ end
102
164
  end
103
165
 
104
166
  # --- anti-game ---------------------------------------------------------------
@@ -188,9 +250,12 @@ def build_prompt(catalog, utterance, bootstrap_layer = nil)
188
250
  user utterance and the catalog of skills (name + description), pick the SINGLE
189
251
  best skill that should handle it. Use only the catalog; route by the
190
252
  descriptions' Use-when / Proactively / Skip rules.
253
+ If NO skill's description covers the utterance, answer "none" — do not
254
+ force-fit the nearest neighbor. Set "clarify" to true only when you would
255
+ need to ask the user a clarifying question before committing to a route.
191
256
 
192
257
  Output ONLY a JSON object on one line, no other text:
193
- {"selected_skill": "<exact skill name from the catalog>", "confidence": <0.0-1.0>, "rationale_short": "<one short clause>"}
258
+ {"selected_skill": "<exact skill name from the catalog, or none>", "clarify": <true|false>, "confidence": <0.0-1.0>, "rationale_short": "<one short clause>"}
194
259
 
195
260
  #{bootstrap_layer ? "== ALWAYS-ON ENTRY ROUTING LAYER (injected into every session; takes precedence for entry routing) ==\n#{bootstrap_layer}\n" : ""}
196
261
  == SKILL CATALOG ==
@@ -309,39 +374,77 @@ grader_available = true
309
374
  tasks.each do |t|
310
375
  exp = t["expected_skill"]
311
376
  must_not = t["must_not_route_to"] || []
377
+ acceptable = t["acceptable"] || []
312
378
  frozen_ref = t["frozen_at_sha"] == "root" ? `git -C #{Shellwords.escape(root)} rev-list --max-parents=0 HEAD`.lines.first.to_s.strip : t["frozen_at_sha"]
313
379
  frozen_ok = ancestor?(root, frozen_ref)
314
380
  prompt = build_prompt(catalog, t["utterance"], bootstrap_layer)
315
- parsed, error = grade(model, timeout_s, prompt)
316
- # A task that produced no observation contributes nothing to the totals, and an
317
- # unmeasured task is indistinguishable from a routing failure in the numbers
318
- # this bank reports. Both recoverable causes are sampling accidents rather than
319
- # verdicts — a hard timeout is usually machine load, and unparseable output is
320
- # usually one stray quote in a generated rationale — so each gets one retry.
321
- # Repairing malformed output instead of re-asking for it was tried and removed:
322
- # interpreting text that is by definition malformed has no natural boundary,
323
- # and three review rounds each found a different shape that a repair would read
324
- # as a verdict. Re-asking needs no such interpretation. Nothing else is retried:
325
- # an auth failure and a missing CLI do reproduce on a second call.
326
- if error&.start_with?("grader_timeout_") || error&.start_with?("no_json_in_output")
327
- parsed, retry_error = grade(model, timeout_s, prompt)
328
- error = parsed ? nil : retry_error
329
- end
330
- if error == "claude_not_found"
331
- grader_available = false
332
- break
381
+ verdicts = []
382
+ replicas.times do |ri|
383
+ parsed, error = grade(model, timeout_s, prompt)
384
+ # A verdict that produced no observation contributes nothing to the totals,
385
+ # and an unmeasured task is indistinguishable from a routing failure in the
386
+ # numbers this bank reports. Both recoverable causes are sampling accidents
387
+ # rather than verdicts — a hard timeout is usually machine load, and
388
+ # unparseable output is usually one stray quote in a generated rationale —
389
+ # so each gets one retry.
390
+ # Repairing malformed output instead of re-asking for it was tried and removed:
391
+ # interpreting text that is by definition malformed has no natural boundary,
392
+ # and three review rounds each found a different shape that a repair would read
393
+ # as a verdict. Re-asking needs no such interpretation. Nothing else is retried:
394
+ # an auth failure and a missing CLI do reproduce on a second call.
395
+ if error&.start_with?("grader_timeout_") || error&.start_with?("no_json_in_output")
396
+ parsed, retry_error = grade(model, timeout_s, prompt)
397
+ error = parsed ? nil : retry_error
398
+ end
399
+ if error == "claude_not_found"
400
+ grader_available = false
401
+ break
402
+ end
403
+ selected = parsed && parsed["selected_skill"]
404
+ clarify = parsed && parsed["clarify"] == true
405
+ confidence = parsed && parsed["confidence"]
406
+ v_status =
407
+ if error then "ERROR"
408
+ elsif (selected == exp || acceptable.include?(selected)) && !must_not.include?(selected) then "PASS"
409
+ else "FAIL"
410
+ end
411
+ verdicts << { replica: ri + 1, selected: selected, clarify: clarify,
412
+ confidence: confidence, status: v_status, error: error }
333
413
  end
334
- selected = parsed && parsed["selected_skill"]
335
- confidence = parsed && parsed["confidence"]
414
+ break unless grader_available
415
+ # Task-level consensus over replicas (replicas=1 reproduces the old per-task
416
+ # semantics exactly). Conservative: one failing replica fails the task —
417
+ # a route that only sometimes lands is not a stable route.
418
+ observed = verdicts.reject { |v| v[:status] == "ERROR" }
336
419
  status =
337
- if error then "ERROR"
338
- elsif selected == exp && !must_not.include?(selected) then "PASS"
420
+ if observed.empty? then "ERROR"
421
+ elsif observed.all? { |v| v[:status] == "PASS" } then "PASS"
339
422
  else "FAIL"
340
423
  end
424
+ selections = observed.map { |v| v[:selected] }.uniq
425
+ # Failure-mode labels (vocabulary in references/eval-routing.md):
426
+ # absorbed — an utterance that should be rejected (expected "none")
427
+ # or kept away from named bait neighbors (must_not_route_to)
428
+ # was claimed by such a skill anyway
429
+ # ownership_split — replicas disagreed on the top pick (unstable ownership)
430
+ labels = []
431
+ absorbed = (exp == "none" && observed.any? { |v| v[:selected] && v[:selected] != "none" }) ||
432
+ observed.any? { |v| must_not.include?(v[:selected]) }
433
+ labels << "absorbed" if absorbed
434
+ labels << "ownership_split" if selections.size > 1
435
+ # A task where some replicas erred but others graded is PARTIALLY measured:
436
+ # the consensus above sees only the observed verdicts, so without this label
437
+ # a PASS+ERROR pair would read as a clean PASS and the run as fully sampled.
438
+ labels << "partial_error" if verdicts.any? { |v| v[:status] == "ERROR" } && !observed.empty?
341
439
  results << {
342
- id: t["id"], utterance: t["utterance"], expected: exp, selected: selected,
343
- confidence: confidence, must_not_route_to: must_not, status: status,
344
- error: error, frozen_at_sha_is_ancestor: frozen_ok
440
+ id: t["id"], utterance: t["utterance"], expected: exp, acceptable: acceptable,
441
+ selected: observed.first && observed.first[:selected],
442
+ confidence: observed.first && observed.first[:confidence],
443
+ clarify: observed.any? { |v| v[:clarify] },
444
+ acceptable_hit: observed.any? { |v| acceptable.include?(v[:selected]) },
445
+ must_not_route_to: must_not, status: status, labels: labels,
446
+ verdicts: verdicts, error: (verdicts.find { |v| v[:error] } || {})[:error],
447
+ frozen_at_sha_is_ancestor: frozen_ok
345
448
  }
346
449
  end
347
450
 
@@ -355,43 +458,114 @@ fails = results.select { |r| r[:status] == "FAIL" }
355
458
  errors = results.select { |r| r[:status] == "ERROR" }
356
459
  drift = results.reject { |r| r[:frozen_at_sha_is_ancestor] }
357
460
 
461
+ # First-class routing-quality metrics beyond pass/fail (counted over observed,
462
+ # non-ERROR verdicts): clarify rate, low-confidence rate, and — when replicas
463
+ # >= 2 — how often all replicas of one task picked the same top skill.
464
+ all_observed = results.flat_map { |r| r[:verdicts].reject { |v| v[:status] == "ERROR" } }
465
+ clarify_count = all_observed.count { |v| v[:clarify] }
466
+ low_conf_count = all_observed.count { |v| v[:confidence].is_a?(Numeric) && v[:confidence] < 0.5 }
467
+ # Replica-level error accounting: task-level `error` counts only fully
468
+ # unmeasured tasks, so a PASS+ERROR pair would otherwise report zero
469
+ # grader-errors while a replica silently went missing.
470
+ error_verdicts = results.sum { |r| r[:verdicts].count { |v| v[:status] == "ERROR" } }
471
+ partial_error_ids = results.select { |r| r[:labels].include?("partial_error") }.map { |r| r[:id] }
472
+ agreement_measured = 0
473
+ agreement_agree = 0
474
+ if replicas >= 2
475
+ results.each do |r|
476
+ obs = r[:verdicts].reject { |v| v[:status] == "ERROR" }
477
+ next if obs.size < 2
478
+ agreement_measured += 1
479
+ agreement_agree += 1 if obs.map { |v| v[:selected] }.uniq.size == 1
480
+ end
481
+ end
482
+
358
483
  # baseline diff (newly failed / newly passed). The baseline is a prior report
359
- # (a hash with a "results" array) or a bare results array.
484
+ # (a hash with a "results" array) or a bare results array. Reports from a
485
+ # different bank content or replica count are DIFFERENT RULERS (documented
486
+ # above): comparing them emits false regressions/improvements, so a
487
+ # demonstrated mismatch suppresses the diff instead of computing it. A bare
488
+ # results array carries no fingerprint — its comparability is unverifiable and
489
+ # is flagged as such rather than silently trusted.
360
490
  newly_failed = []
361
491
  newly_passed = []
492
+ baseline_comparable = nil
493
+ baseline_incomparable_reason = nil
362
494
  if baseline_path && File.file?(baseline_path)
363
495
  base_json = JSON.parse(File.read(baseline_path))
364
- base_results = base_json.is_a?(Hash) ? (base_json["results"] || []) : base_json
365
- base_status = base_results.to_h { |r| [r["id"], r["status"]] }
366
- results.each do |r|
367
- was = base_status[r[:id]]
368
- newly_failed << r[:id] if was == "PASS" && r[:status] == "FAIL"
369
- newly_passed << r[:id] if was == "FAIL" && r[:status] == "PASS"
496
+ if base_json.is_a?(Hash)
497
+ base_bank_sha = base_json.dig("routing_surface", "bank_sha256")
498
+ base_replicas = base_json["replicas"] || 1 # legacy reports predate --replicas
499
+ if base_bank_sha && base_bank_sha != routing_surface[:bank_sha256]
500
+ baseline_comparable = false
501
+ baseline_incomparable_reason = "bank content differs (baseline #{base_bank_sha[0, 12]}… vs current #{routing_surface[:bank_sha256][0, 12]}…)"
502
+ elsif base_replicas != replicas
503
+ baseline_comparable = false
504
+ baseline_incomparable_reason = "replica count differs (baseline #{base_replicas} vs current #{replicas})"
505
+ else
506
+ baseline_comparable = base_bank_sha ? true : "unverified (baseline carries no bank fingerprint)"
507
+ end
508
+ else
509
+ baseline_comparable = "unverified (bare results array carries no fingerprint)"
510
+ end
511
+ unless baseline_comparable == false
512
+ base_results = base_json.is_a?(Hash) ? (base_json["results"] || []) : base_json
513
+ base_status = base_results.to_h { |r| [r["id"], r["status"]] }
514
+ results.each do |r|
515
+ was = base_status[r[:id]]
516
+ newly_failed << r[:id] if was == "PASS" && r[:status] == "FAIL"
517
+ newly_passed << r[:id] if was == "FAIL" && r[:status] == "PASS"
518
+ end
370
519
  end
371
520
  end
372
521
 
373
522
  report = {
374
523
  model: model, tasks: results.size, pass: passes, fail: fails.size, error: errors.size,
524
+ replicas: replicas, verdicts: all_observed.size,
525
+ error_verdicts: error_verdicts, partial_error_tasks: partial_error_ids,
526
+ clarify_count: clarify_count, low_confidence_count: low_conf_count,
527
+ replica_agreement: (replicas >= 2 ? { agree: agreement_agree, measured: agreement_measured } : nil),
375
528
  desc_budget_chars: desc_budget, routing_surface: routing_surface,
376
529
  co_change_bank_and_descriptions: co_change, co_change_check_available: co_change_check_ok,
377
530
  frozen_drift: drift.map { |r| r[:id] },
531
+ baseline_comparable: baseline_comparable,
532
+ baseline_incomparable_reason: baseline_incomparable_reason,
378
533
  newly_failed: newly_failed, newly_passed: newly_passed, results: results
379
534
  }
380
535
  File.write(json_path, JSON.pretty_generate(report)) if json_path
381
536
 
382
537
  puts "eval-routing-bank (#{model}): #{passes}/#{results.size} pass, #{fails.size} fail, #{errors.size} grader-error"
383
538
  puts " arm: desc-budget-chars=#{desc_budget}" if desc_budget
539
+ unless all_observed.empty?
540
+ line = " clarify: #{clarify_count}/#{all_observed.size} verdicts, low-confidence(<0.5): #{low_conf_count}/#{all_observed.size}"
541
+ line += ", replica top1 agreement: #{agreement_agree}/#{agreement_measured}" if replicas >= 2
542
+ puts line
543
+ end
544
+ if error_verdicts.positive?
545
+ puts " ⚠ grader-error verdicts: #{error_verdicts}#{partial_error_ids.empty? ? '' : " (partially measured tasks: #{partial_error_ids.join(', ')})"}"
546
+ end
384
547
  puts " ⚠ co-change check unavailable: base ref #{base.inspect} not resolvable — could not verify bank/description co-change" unless co_change_check_ok
385
548
  puts " ⚠ co-change: this change touches BOTH the bank and a SKILL.md description (verify the bank was not edited to pass)" if co_change
386
549
  puts " ⚠ frozen-drift (not ancestor of HEAD, excluded from regression judgment): #{drift.map { |r| r[:id] }.join(', ')}" unless drift.empty?
387
550
  unless fails.empty?
388
551
  puts " FAIL:"
389
- fails.each { |r| puts " - #{r[:id]}: expected #{r[:expected]} got #{r[:selected].inspect} (conf #{r[:confidence]})" }
552
+ fails.each do |r|
553
+ tag = r[:labels].empty? ? "" : " [#{r[:labels].join(',')}]"
554
+ obs = r[:verdicts].reject { |v| v[:status] == "ERROR" }
555
+ got = obs.map { |v| v[:selected].inspect }.uniq.join(" | ")
556
+ conf = obs.map { |v| v[:confidence] }.join(",")
557
+ puts " - #{r[:id]}: expected #{r[:expected]} got #{got} (conf #{conf})#{tag}"
558
+ end
390
559
  end
391
560
  unless errors.empty?
392
561
  puts " GRADER-ERROR:"
393
562
  errors.each { |r| puts " - #{r[:id]}: #{r[:error]}" }
394
563
  end
564
+ if baseline_comparable == false
565
+ puts " ⚠ baseline not compared — different ruler: #{baseline_incomparable_reason}"
566
+ elsif baseline_comparable.is_a?(String)
567
+ puts " ⚠ baseline comparability #{baseline_comparable}"
568
+ end
395
569
  unless newly_failed.empty? && newly_passed.empty?
396
570
  puts " vs baseline: newly_failed=#{newly_failed.join(',')} newly_passed=#{newly_passed.join(',')}"
397
571
  end
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env bash
2
- # Extraction-owned autonomous review wrapper: one review plus two challenges.
2
+ # Extraction-owned autonomous review wrapper: one review plus one challenge.
3
3
  set -euo pipefail
4
4
 
5
5
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
@@ -8,7 +8,7 @@ CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
8
8
  for arg in "$@"; do
9
9
  case "$arg" in
10
10
  --challenge-b*)
11
- echo "extraction_review_gate_error: challenge budget is fixed at 2" >&2
11
+ echo "extraction_review_gate_error: challenge budget is fixed at 1" >&2
12
12
  exit 2
13
13
  ;;
14
14
  esac
@@ -19,4 +19,4 @@ if [[ ! -x "$CONTROLLER" ]]; then
19
19
  exit 2
20
20
  fi
21
21
 
22
- exec bash "$CONTROLLER" --challenge-budget 2 "$@"
22
+ exec bash "$CONTROLLER" --challenge-budget 1 "$@"