@ccoalm/ccl-skills 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +8 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +29 -31
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +32 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
- package/dist/assets/release.json +79 -24
- package/package.json +1 -1
|
@@ -21,6 +21,18 @@
|
|
|
21
21
|
# advisory = no signal). A agent-context/session-start.md absent from base is treated as a NEW
|
|
22
22
|
# every-session injection and blocks like a new severe entrypoint.
|
|
23
23
|
#
|
|
24
|
+
# Blocking (delta-scoped, skills/*/references/**/*.md — the reference-file line
|
|
25
|
+
# ratchet; design invariants in references/attention-budget-ratchet.md):
|
|
26
|
+
# - a changed reference that is NEW or CROSSES over 500 physical lines (base
|
|
27
|
+
# was not over) => block ("new reference over line budget")
|
|
28
|
+
# - a changed reference already over 500 lines in base that GREW
|
|
29
|
+
# (head_lines > base_lines) => block ("over-limit reference grew")
|
|
30
|
+
# references/source-register.md is structurally excluded (append-only ledger;
|
|
31
|
+
# its growth is governed by the append-only contract, not a line cap) — the
|
|
32
|
+
# exclusion lives in this gate, never in the candidate. A NEW reference over
|
|
33
|
+
# 100 lines with no "##" section headings draws an advisory token (never
|
|
34
|
+
# blocks). Rename credit is path-paired and non-growing, same as entrypoints.
|
|
35
|
+
#
|
|
24
36
|
# Advisory (never blocks on its own): the recommended-band body-char debt
|
|
25
37
|
# counters, severe-debt counters, per-file deltas, and the agent-context/session-start.md 13000B
|
|
26
38
|
# byte tripwire BAND remain VISIBILITY / DEBT-MANAGEMENT only (the band flags the
|
|
@@ -113,6 +125,11 @@ end
|
|
|
113
125
|
WORD_BUDGET_MAX = 5000
|
|
114
126
|
WORD_TOKEN_PATTERN = /\p{Han}|\uFFFD|(?:(?!\p{Han})[\p{L}\p{N}])+(?:[\u0027\u2019_-](?:(?!\p{Han})[\p{L}\p{N}])+)*/u
|
|
115
127
|
|
|
128
|
+
REF_LINE_MAX = 500
|
|
129
|
+
REF_NAV_MIN_LINES = 100
|
|
130
|
+
REF_PATH_RE = %r{\Askills/[^/]+/references/.+\.md\z}
|
|
131
|
+
REF_LEDGER_RE = %r{\Askills/[^/]+/references/source-register\.md\z}
|
|
132
|
+
|
|
116
133
|
Metric = Struct.new(:body_chars, :body_words, :bytes, keyword_init: true)
|
|
117
134
|
|
|
118
135
|
def metric_for_text(text)
|
|
@@ -195,6 +212,42 @@ def head_metric_for_path(root, rel)
|
|
|
195
212
|
metric_for_file(full)
|
|
196
213
|
end
|
|
197
214
|
|
|
215
|
+
def ref_info_for_text(raw)
|
|
216
|
+
text = raw.b.dup.force_encoding(Encoding::UTF_8).scrub { |invalid| "�" * invalid.bytesize }
|
|
217
|
+
# Line-ending normalization comes first: a CR-only or CRLF file must measure the
|
|
218
|
+
# same as the LF file a reader sees, otherwise a 501-line CR-delimited file reads
|
|
219
|
+
# as one line and earns a clean verdict. Headings count EXACT H2 only, because an
|
|
220
|
+
# H3-only file gives a chunked read no section structure to navigate by.
|
|
221
|
+
text = text.gsub(/\r\n?/, "\n")
|
|
222
|
+
lines = text.lines
|
|
223
|
+
{ lines: lines.length, headings: lines.count { |l| l.match?(/\A##[ \t]/) } }
|
|
224
|
+
end
|
|
225
|
+
|
|
226
|
+
def base_ref_info_for_path(root, base, rel)
|
|
227
|
+
return :unknown if base.nil?
|
|
228
|
+
return :missing unless git_success?("cat-file", "-e", "#{base}:#{rel}", root: root)
|
|
229
|
+
ref_info_for_text(git_text("show", "#{base}:#{rel}", root: root))
|
|
230
|
+
end
|
|
231
|
+
|
|
232
|
+
def head_ref_info_for_path(root, rel)
|
|
233
|
+
full = File.join(root, rel)
|
|
234
|
+
return :missing unless File.file?(full)
|
|
235
|
+
ref_info_for_text(File.binread(full))
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
def fmt_ref_value(info)
|
|
239
|
+
case info
|
|
240
|
+
when :unknown then "unknown"
|
|
241
|
+
when :missing then "missing"
|
|
242
|
+
else info[:lines].to_s
|
|
243
|
+
end
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
def fmt_ref_delta(base_info, head_info)
|
|
247
|
+
return "unknown" unless base_info.is_a?(Hash) && head_info.is_a?(Hash)
|
|
248
|
+
format("%+d", head_info[:lines] - base_info[:lines])
|
|
249
|
+
end
|
|
250
|
+
|
|
198
251
|
def fmt_metric_value(metric, field)
|
|
199
252
|
case metric
|
|
200
253
|
when :unknown then "unknown"
|
|
@@ -437,6 +490,144 @@ if bootstrap_changed
|
|
|
437
490
|
# :missing (deleted) => nothing left to size-gate, same as a deleted entrypoint
|
|
438
491
|
end
|
|
439
492
|
|
|
493
|
+
# --- reference-file line-budget ratchet (delta-scoped) ------------------------
|
|
494
|
+
# Design invariants: references/attention-budget-ratchet.md. Physical lines are
|
|
495
|
+
# the stable proxy metric; the append-only ledger references/source-register.md
|
|
496
|
+
# is excluded by gate design (its growth is governed by the append-only
|
|
497
|
+
# contract) and the exclusion lives HERE so a candidate cannot nominate its own
|
|
498
|
+
# exemption.
|
|
499
|
+
ref_blocks = []
|
|
500
|
+
ref_partials = []
|
|
501
|
+
ref_changed = []
|
|
502
|
+
ref_scan_error = nil
|
|
503
|
+
begin
|
|
504
|
+
if system("git", "-C", root, "rev-parse", "--is-inside-work-tree", out: File::NULL, err: File::NULL)
|
|
505
|
+
rpaths = []
|
|
506
|
+
rpaths.concat(git_lines("diff", "--name-only", base, "HEAD", "--", "skills", root: root)) if base
|
|
507
|
+
rpaths.concat(git_lines("diff", "--name-only", "--", "skills", root: root))
|
|
508
|
+
rpaths.concat(git_lines("diff", "--cached", "--name-only", "--", "skills", root: root))
|
|
509
|
+
rpaths.concat(git_lines("ls-files", "--others", "--exclude-standard", "--", "skills", root: root))
|
|
510
|
+
ref_changed = rpaths.select { |p| p.match?(REF_PATH_RE) }.reject { |p| p.match?(REF_LEDGER_RE) }.uniq
|
|
511
|
+
end
|
|
512
|
+
rescue StandardError => e
|
|
513
|
+
ref_scan_error = "#{e.class.name} #{e.message.lines.first.to_s.strip}"
|
|
514
|
+
warn "changed_reference_line_scan_skipped: #{ref_scan_error}"
|
|
515
|
+
end
|
|
516
|
+
# A scan failure must not silently read as "no references changed" — that would
|
|
517
|
+
# false-green a real growth, so it fails closed as a partial.
|
|
518
|
+
ref_partials << "skills/*/references: changed-detection failed (#{ref_scan_error}) — line-budget check unavailable (fail-closed)" if ref_scan_error
|
|
519
|
+
|
|
520
|
+
ref_move_map = {}
|
|
521
|
+
begin
|
|
522
|
+
if system("git", "-C", root, "rev-parse", "--is-inside-work-tree", out: File::NULL, err: File::NULL)
|
|
523
|
+
status_args = []
|
|
524
|
+
status_args << ["diff", "--name-status", base, "HEAD"] if base
|
|
525
|
+
status_args << ["diff", "--name-status", "--cached"]
|
|
526
|
+
status_args << ["diff", "--name-status"]
|
|
527
|
+
status_args.each do |cmd|
|
|
528
|
+
git_lines(*cmd, "--", "skills", root: root).each do |line|
|
|
529
|
+
status, *paths = line.split("\t")
|
|
530
|
+
next unless status.start_with?("R") && paths.length >= 2
|
|
531
|
+
next unless status[1..].to_i >= 90
|
|
532
|
+
old_rel = paths[0].match?(REF_PATH_RE) ? paths[0] : nil
|
|
533
|
+
new_rel = paths[1].match?(REF_PATH_RE) ? paths[1] : nil
|
|
534
|
+
ref_move_map[new_rel] = old_rel if old_rel && new_rel
|
|
535
|
+
end
|
|
536
|
+
end
|
|
537
|
+
end
|
|
538
|
+
rescue StandardError => e
|
|
539
|
+
warn "reference_move_pair_scan_skipped: #{e.class.name}"
|
|
540
|
+
end
|
|
541
|
+
|
|
542
|
+
# The census is visibility only, so an unreadable or vanished file must degrade the
|
|
543
|
+
# COUNTER to unknown rather than abort the program: aborting here would skip the
|
|
544
|
+
# per-file fail-closed partials below and emit no verdict token at all.
|
|
545
|
+
head_ref_over_count = 0
|
|
546
|
+
head_ref_census_error = nil
|
|
547
|
+
begin
|
|
548
|
+
Dir[File.join(root, "skills", "*", "references", "**", "*.md")].each do |path|
|
|
549
|
+
rel = path.start_with?(root + "/") ? path[(root.length + 1)..] : path.sub(%r{\A\./}, "")
|
|
550
|
+
next unless rel.match?(REF_PATH_RE)
|
|
551
|
+
next if rel.match?(REF_LEDGER_RE)
|
|
552
|
+
info = begin
|
|
553
|
+
head_ref_info_for_path(root, rel)
|
|
554
|
+
rescue StandardError => e
|
|
555
|
+
head_ref_census_error ||= "#{rel}: #{e.class.name}"
|
|
556
|
+
next
|
|
557
|
+
end
|
|
558
|
+
head_ref_over_count += 1 if info.is_a?(Hash) && info[:lines] > REF_LINE_MAX
|
|
559
|
+
end
|
|
560
|
+
rescue StandardError => e
|
|
561
|
+
head_ref_census_error ||= e.class.name
|
|
562
|
+
end
|
|
563
|
+
if head_ref_census_error
|
|
564
|
+
warn "reference_line_census_partial: head over-limit census incomplete (#{head_ref_census_error}) — counter reported unknown; per-file verdicts below are unaffected"
|
|
565
|
+
head_ref_over_count = nil
|
|
566
|
+
end
|
|
567
|
+
base_ref_over_count = nil
|
|
568
|
+
if base
|
|
569
|
+
begin
|
|
570
|
+
base_ref_over_count = 0
|
|
571
|
+
git_lines("grep", "-c", "-e", "", base, "--", "skills", root: root).each do |line|
|
|
572
|
+
rest, _, cnt = line.rpartition(":")
|
|
573
|
+
_, _, rel = rest.partition(":")
|
|
574
|
+
next unless rel.match?(REF_PATH_RE)
|
|
575
|
+
next if rel.match?(REF_LEDGER_RE)
|
|
576
|
+
base_ref_over_count += 1 if cnt.to_i > REF_LINE_MAX
|
|
577
|
+
end
|
|
578
|
+
rescue StandardError => e
|
|
579
|
+
base_ref_over_count = nil
|
|
580
|
+
warn "reference_line_trend_partial: base=unknown reason=#{e.class.name}"
|
|
581
|
+
end
|
|
582
|
+
end
|
|
583
|
+
puts "reference_line_over_limit_count_base=#{base_ref_over_count.nil? ? "unknown" : base_ref_over_count}"
|
|
584
|
+
puts "reference_line_over_limit_count_head=#{head_ref_over_count.nil? ? "unknown" : head_ref_over_count}"
|
|
585
|
+
puts "reference_line_over_limit_count_delta=#{(base_ref_over_count.nil? || head_ref_over_count.nil?) ? "unknown" : format("%+d", head_ref_over_count - base_ref_over_count)}"
|
|
586
|
+
|
|
587
|
+
# Ledger visibility: excluded from the ratchet, still reported when over.
|
|
588
|
+
Dir[File.join(root, "skills", "*", "references", "source-register.md")].sort.each do |path|
|
|
589
|
+
rel = path.start_with?(root + "/") ? path[(root.length + 1)..] : path.sub(%r{\A\./}, "")
|
|
590
|
+
info = begin
|
|
591
|
+
head_ref_info_for_path(root, rel)
|
|
592
|
+
rescue StandardError
|
|
593
|
+
:missing
|
|
594
|
+
end
|
|
595
|
+
if info.is_a?(Hash) && info[:lines] > REF_LINE_MAX
|
|
596
|
+
puts "reference_line_budget_ledger_excluded: #{rel} lines=#{info[:lines]} — append-only ledger, excluded by gate design (growth governed by the append-only contract)"
|
|
597
|
+
end
|
|
598
|
+
end
|
|
599
|
+
|
|
600
|
+
ref_changed.sort.each do |rel|
|
|
601
|
+
begin
|
|
602
|
+
head_info = head_ref_info_for_path(root, rel)
|
|
603
|
+
rescue StandardError => e
|
|
604
|
+
ref_partials << "#{rel}: head lines unreadable (#{e.class.name}) — line-budget check unavailable (fail-closed)"
|
|
605
|
+
next
|
|
606
|
+
end
|
|
607
|
+
next if head_info == :missing # deleted reference: nothing left to gate
|
|
608
|
+
base_info = base_ref_info_for_path(root, base, rel)
|
|
609
|
+
puts "changed_reference_line_delta: #{rel} base_lines=#{fmt_ref_value(base_info)} head_lines=#{head_info[:lines]} delta_lines=#{fmt_ref_delta(base_info, head_info)}"
|
|
610
|
+
if base_info == :missing && head_info[:lines] > REF_NAV_MIN_LINES && head_info[:headings].zero?
|
|
611
|
+
warn "reference_nav_advisory: #{rel} lines=#{head_info[:lines]} headings=0 — a new reference over #{REF_NAV_MIN_LINES} lines needs ## section structure so chunked reads and greps can navigate (advisory, never blocks)"
|
|
612
|
+
end
|
|
613
|
+
next unless head_info[:lines] > REF_LINE_MAX
|
|
614
|
+
if base_info == :unknown
|
|
615
|
+
ref_partials << "#{rel}: base unknown — reference line-budget check unavailable (fail-closed)"
|
|
616
|
+
elsif base_info == :missing || base_info[:lines] <= REF_LINE_MAX
|
|
617
|
+
source_rel = ref_move_map[rel]
|
|
618
|
+
source_info = source_rel ? base_ref_info_for_path(root, base, source_rel) : nil
|
|
619
|
+
if source_info.is_a?(Hash) && source_info[:lines] > REF_LINE_MAX && head_info[:lines] <= source_info[:lines]
|
|
620
|
+
puts "reference_line_budget_move_ok: #{rel} head_lines=#{head_info[:lines]} allowed_lines=#{source_info[:lines]} moved_from=#{source_rel}"
|
|
621
|
+
else
|
|
622
|
+
ref_blocks << "#{rel}: new reference over line budget head_lines=#{head_info[:lines]} (> #{REF_LINE_MAX}) — split it by subtopic before landing (see references/attention-budget-ratchet.md; there is no exempt marker or waiver flag)"
|
|
623
|
+
end
|
|
624
|
+
elsif head_info[:lines] > base_info[:lines]
|
|
625
|
+
ref_blocks << "#{rel}: over-limit reference grew base_lines=#{base_info[:lines]} head_lines=#{head_info[:lines]} — shrink or stay level; fund additions by consolidating text in the same file"
|
|
626
|
+
else
|
|
627
|
+
puts "reference_line_budget_legacy_ok: #{rel} base_lines=#{base_info[:lines]} head_lines=#{head_info[:lines]} allowed_lines=#{base_info[:lines]}"
|
|
628
|
+
end
|
|
629
|
+
end
|
|
630
|
+
|
|
440
631
|
# Advisory markers only when the blocking verdict is clean — a consumer reading
|
|
441
632
|
# the preserved legacy token must never see ok next to a block.
|
|
442
633
|
#
|
|
@@ -451,18 +642,20 @@ end
|
|
|
451
642
|
# — but the token says un-evaluated, not ok.
|
|
452
643
|
# NOTE: this comment lives inside the single-quoted `ruby -e` program; an
|
|
453
644
|
# apostrophe here terminates the shell quote and breaks the script.
|
|
454
|
-
if blocks.empty? && partials.empty? && word_blocks.empty? && word_partials.empty?
|
|
645
|
+
if blocks.empty? && partials.empty? && word_blocks.empty? && word_partials.empty? && ref_blocks.empty? && ref_partials.empty?
|
|
455
646
|
puts "size_budget_advisory_#{size_state}"
|
|
456
647
|
puts "entrypoint_size_budget_advisory_ok"
|
|
457
648
|
if base.nil?
|
|
458
649
|
puts "entrypoint_size_blocking_unevaluated: base=unknown — committed changes were NOT delta-checked; this is not a pass"
|
|
459
650
|
puts "entrypoint_word_budget_blocking_unevaluated: base=unknown — committed changes were NOT delta-checked; this is not a pass"
|
|
651
|
+
puts "reference_line_budget_blocking_unevaluated: base=unknown — committed changes were NOT delta-checked; this is not a pass"
|
|
460
652
|
if File.file?(File.join(root, "agent-context/session-start.md"))
|
|
461
653
|
puts "bootstrap_size_delta_unevaluated: base=unknown — agent-context/session-start.md committed changes were NOT delta-checked; this is not a pass"
|
|
462
654
|
end
|
|
463
655
|
else
|
|
464
656
|
puts "entrypoint_size_blocking_ok"
|
|
465
657
|
puts "entrypoint_word_budget_blocking_ok"
|
|
658
|
+
puts "reference_line_budget_blocking_ok"
|
|
466
659
|
end
|
|
467
660
|
exit 0
|
|
468
661
|
end
|
|
@@ -470,7 +663,10 @@ blocks.each { |b| warn "entrypoint_size_block: #{b}" }
|
|
|
470
663
|
partials.each { |b| warn "entrypoint_size_block_partial: #{b}" }
|
|
471
664
|
word_blocks.each { |b| warn "entrypoint_word_budget_block: #{b}" }
|
|
472
665
|
word_partials.each { |b| warn "entrypoint_word_budget_block_partial: #{b}" }
|
|
666
|
+
ref_blocks.each { |b| warn "reference_line_block: #{b}" }
|
|
667
|
+
ref_partials.each { |b| warn "reference_line_block_partial: #{b}" }
|
|
473
668
|
puts "entrypoint_word_budget_blocking_failed" unless word_blocks.empty? && word_partials.empty?
|
|
669
|
+
puts "reference_line_budget_blocking_failed" unless ref_blocks.empty? && ref_partials.empty?
|
|
474
670
|
# Legacy aggregate token for every blocking verdict owned by this size-budget
|
|
475
671
|
# script, including the body-word rule. Keep it unconditional so existing
|
|
476
672
|
# consumers cannot miss a new word-only failure.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# Contract-anchor table for check-contract-anchors.sh (same directory).
|
|
2
|
+
# One row per pinned load-bearing contract literal:
|
|
3
|
+
# id <TAB> repo-relative path <TAB> pinned literal (exactly-once, >=16 chars) <TAB> note
|
|
4
|
+
# The note names the sanitized provenance label and why the wording is load-bearing.
|
|
5
|
+
# To change pinned wording intentionally: edit the contract sentence AND this row
|
|
6
|
+
# in the same MR. Keep this set small — load-bearing contracts only, never style.
|
|
7
|
+
verdict-taxonomy-discriminator skills/testing-strategy/references/ci-fixtures-and-flake-control.md One discriminating predicate decides the verdict 071-chainA-r1f1: fault-origin discriminator sentence; deleting it reverts verdicts to case-by-case judgment
|
|
8
|
+
verdict-taxonomy-false-green skills/testing-strategy/references/ci-fixtures-and-flake-control.md excludes `infra-error` cases to present a clean total is a false-green report 071-chainA-r1f1: report-completeness clause; deleting it legalizes infra-error-excluding clean totals
|
|
9
|
+
stop-predicate-materially-differing skills/product-rd-workflow/SKILL.md materially differing viable approaches (none dominant-and-reversible) 071-chainB-r1f1: stop-condition predicate wording; inverting it flips blocked/continue classification
|
|
10
|
+
stop-predicate-evidenced-cause skills/product-rd-workflow/SKILL.md a fix lacking evidenced cause 071-chainB-r1f1: stop-condition predicate wording; weakening it admits plausible-but-unevidenced fixes
|
|
11
|
+
burn-rate-tiers-entry skills/platform-observability/SKILL.md page 14.4× 1h/5m, page 6× 6h/30m, ticket 1× 3d/6h 071-chainC-r1f4: SRE Workbook multiwindow tiers, externally verified value (specs/071 source-verification); silent renumbering must not pass
|
|
12
|
+
burn-rate-page-row-reference skills/platform-observability/references/sli-slo-design.md | Page (P0) | 1h | 5m | 14.4 | 071-chainC-r1f4: reference-side copy of the page tier; pinning both sides makes entry/reference drift visible
|
|
13
|
+
coverage-tier-provenance skills/testing-strategy/references/test-code-authoring-patterns.md 60% acceptable / 75% commendable / 90% exemplary 分档 071-chainC-r1f4: externally verified coverage tiers (specs/071 source-verification)
|
|
14
|
+
pairwise-trigger-range skills/test-artifact-management/references/classical-test-design-techniques.md 2-way 累计触发 53–97% 071-chainC-r1f4: NIST SP 800-142 empirical range, externally verified (specs/071 source-verification)
|
|
15
|
+
bva-two-vs-three-value skills/test-artifact-management/references/classical-test-design-techniques.md 2-value(边界 + 下一格)和 3-value(边界 + 两侧) 071-chainC-r1f4: ISTQB v4 BVA variant definitions, externally verified (specs/071 source-verification)
|
|
@@ -16,13 +16,23 @@
|
|
|
16
16
|
# Usage:
|
|
17
17
|
# eval-routing-bank.rb <repo-root> [--bank <path>] [--model <name>] [--limit N]
|
|
18
18
|
# [--dry-run] [--json <path>] [--baseline <path>] [--timeout S]
|
|
19
|
-
# [--desc-budget-chars N] [--with-bootstrap]
|
|
19
|
+
# [--desc-budget-chars N] [--with-bootstrap] [--replicas N]
|
|
20
20
|
#
|
|
21
21
|
# --desc-budget-chars N simulates a consumer that truncates each skill description
|
|
22
22
|
# to its first N characters before routing (e.g. Codex compresses the skill listing
|
|
23
23
|
# under a ~2%-of-context budget, 8000 chars when the window is unknown, shortening
|
|
24
24
|
# descriptions first). Run the bank once plain and once with a budget to see which
|
|
25
25
|
# routes only survive on the description tail — those triggers need front-loading.
|
|
26
|
+
#
|
|
27
|
+
# expected_skill "none" marks negative controls (out-of-library utterances) and
|
|
28
|
+
# coverage-gap probes (in-domain utterances no skill owns): the correct outcome
|
|
29
|
+
# is rejection, and a catalog skill claiming them is labeled "absorbed".
|
|
30
|
+
# --replicas N grades each task N times: task status is the conservative
|
|
31
|
+
# consensus (every observed replica must PASS), replica top1 agreement is
|
|
32
|
+
# reported, and disagreeing replicas label the task "ownership_split". clarify
|
|
33
|
+
# and low-confidence counts are first-class report fields either way.
|
|
34
|
+
# Reports from different (bank, replicas) configurations are different rulers —
|
|
35
|
+
# do not diff them as a regression signal.
|
|
26
36
|
# Exit: 0 = ran (advisory); 2 = usage error; 3 = grader entirely unavailable.
|
|
27
37
|
|
|
28
38
|
require "yaml"
|
|
@@ -39,7 +49,7 @@ end
|
|
|
39
49
|
|
|
40
50
|
root = ARGV[0]
|
|
41
51
|
if root.nil? || root.start_with?("-")
|
|
42
|
-
warn "usage: eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--timeout S] [--desc-budget-chars N]"
|
|
52
|
+
warn "usage: eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--timeout S] [--desc-budget-chars N] [--replicas N]"
|
|
43
53
|
exit 2
|
|
44
54
|
end
|
|
45
55
|
bank_path = arg("--bank", File.join(root, "eval", "routing-tasks.jsonl"))
|
|
@@ -65,6 +75,17 @@ if ARGV.include?("--desc-budget-chars")
|
|
|
65
75
|
desc_budget = b.to_i
|
|
66
76
|
end
|
|
67
77
|
timeout_s = (t = arg("--timeout")) ? t.to_i : 60
|
|
78
|
+
replicas = 1
|
|
79
|
+
if ARGV.include?("--replicas")
|
|
80
|
+
r = arg("--replicas")
|
|
81
|
+
# Same strictness as --desc-budget-chars: a silent to_i coercion would turn a
|
|
82
|
+
# typo into "1 replica" and the agreement metric would quietly measure nothing.
|
|
83
|
+
unless r && r =~ /\A[1-9]\d*\z/
|
|
84
|
+
warn "--replicas requires a positive integer value, got #{r.inspect}"
|
|
85
|
+
exit 2
|
|
86
|
+
end
|
|
87
|
+
replicas = r.to_i
|
|
88
|
+
end
|
|
68
89
|
|
|
69
90
|
unless File.file?(bank_path)
|
|
70
91
|
warn "eval_bank_missing: #{bank_path}"
|
|
@@ -89,16 +110,57 @@ end
|
|
|
89
110
|
tasks = tasks.first(limit) if limit
|
|
90
111
|
|
|
91
112
|
# Schema validation: a task must be answerable, and must not be self-contradictory.
|
|
113
|
+
# expected_skill "none" is the negative-control / coverage-gap sentinel: the
|
|
114
|
+
# correct routing outcome is that NO catalog skill claims the utterance.
|
|
115
|
+
all_outcomes = Dir[File.join(root, "skills", "*", "SKILL.md")]
|
|
116
|
+
.map { |p| File.basename(File.dirname(p)) } + ["none"]
|
|
92
117
|
tasks.each do |t|
|
|
93
118
|
id = t["id"] || "(no id)"
|
|
94
119
|
if t["utterance"].to_s.strip.empty? || t["expected_skill"].to_s.strip.empty?
|
|
95
120
|
warn "eval_bank_invalid_task: #{id}: utterance and expected_skill are required"
|
|
96
121
|
exit 2
|
|
97
122
|
end
|
|
123
|
+
# Type-check the ORIGINAL values FIRST: a present-but-non-list field (e.g. ""
|
|
124
|
+
# or a bare string or a number) must fail as a usage error before any
|
|
125
|
+
# membership check touches it — Ruby strings are truthy and respond to
|
|
126
|
+
# include? (substring semantics), and non-strings would raise a bare
|
|
127
|
+
# NoMethodError instead of the documented invalid-bank diagnostic.
|
|
128
|
+
%w[must_not_route_to acceptable].each do |f|
|
|
129
|
+
next unless t.key?(f)
|
|
130
|
+
unless t[f].is_a?(Array)
|
|
131
|
+
warn "eval_bank_invalid_task: #{id}: #{f} must be a list, got #{t[f].inspect}"
|
|
132
|
+
exit 2
|
|
133
|
+
end
|
|
134
|
+
end
|
|
98
135
|
if (t["must_not_route_to"] || []).include?(t["expected_skill"])
|
|
99
136
|
warn "eval_bank_invalid_task: #{id}: expected_skill is also in must_not_route_to (impossible)"
|
|
100
137
|
exit 2
|
|
101
138
|
end
|
|
139
|
+
if (t["must_not_route_to"] || []).include?("none")
|
|
140
|
+
warn "eval_bank_invalid_task: #{id}: \"none\" is a sentinel outcome, not a routable target for must_not_route_to"
|
|
141
|
+
exit 2
|
|
142
|
+
end
|
|
143
|
+
# Optional acceptable[] names defensible alternate outcomes (a skill name or
|
|
144
|
+
# "none") for utterances with more than one correct route — e.g. a coverage-gap
|
|
145
|
+
# ask where both coordinator intake and rejection are right. It must not
|
|
146
|
+
# restate expected_skill or contradict must_not_route_to.
|
|
147
|
+
acc = t["acceptable"] || []
|
|
148
|
+
if acc.include?(t["expected_skill"])
|
|
149
|
+
warn "eval_bank_invalid_task: #{id}: acceptable restates expected_skill"
|
|
150
|
+
exit 2
|
|
151
|
+
end
|
|
152
|
+
unless (acc & (t["must_not_route_to"] || [])).empty?
|
|
153
|
+
warn "eval_bank_invalid_task: #{id}: acceptable and must_not_route_to overlap (contradictory)"
|
|
154
|
+
exit 2
|
|
155
|
+
end
|
|
156
|
+
# Anti-gaming: expected + acceptable must leave at least one outcome that
|
|
157
|
+
# would FAIL the row. A row covering the complete catalog-and-none outcome
|
|
158
|
+
# space passes on every valid grader selection — a vacuous fixture that fakes
|
|
159
|
+
# green regardless of description behavior.
|
|
160
|
+
if (all_outcomes - ([t["expected_skill"]] + acc)).empty?
|
|
161
|
+
warn "eval_bank_invalid_task: #{id}: expected_skill plus acceptable cover every possible outcome (vacuous row)"
|
|
162
|
+
exit 2
|
|
163
|
+
end
|
|
102
164
|
end
|
|
103
165
|
|
|
104
166
|
# --- anti-game ---------------------------------------------------------------
|
|
@@ -188,9 +250,12 @@ def build_prompt(catalog, utterance, bootstrap_layer = nil)
|
|
|
188
250
|
user utterance and the catalog of skills (name + description), pick the SINGLE
|
|
189
251
|
best skill that should handle it. Use only the catalog; route by the
|
|
190
252
|
descriptions' Use-when / Proactively / Skip rules.
|
|
253
|
+
If NO skill's description covers the utterance, answer "none" — do not
|
|
254
|
+
force-fit the nearest neighbor. Set "clarify" to true only when you would
|
|
255
|
+
need to ask the user a clarifying question before committing to a route.
|
|
191
256
|
|
|
192
257
|
Output ONLY a JSON object on one line, no other text:
|
|
193
|
-
{"selected_skill": "<exact skill name from the catalog>", "confidence": <0.0-1.0>, "rationale_short": "<one short clause>"}
|
|
258
|
+
{"selected_skill": "<exact skill name from the catalog, or none>", "clarify": <true|false>, "confidence": <0.0-1.0>, "rationale_short": "<one short clause>"}
|
|
194
259
|
|
|
195
260
|
#{bootstrap_layer ? "== ALWAYS-ON ENTRY ROUTING LAYER (injected into every session; takes precedence for entry routing) ==\n#{bootstrap_layer}\n" : ""}
|
|
196
261
|
== SKILL CATALOG ==
|
|
@@ -309,39 +374,77 @@ grader_available = true
|
|
|
309
374
|
tasks.each do |t|
|
|
310
375
|
exp = t["expected_skill"]
|
|
311
376
|
must_not = t["must_not_route_to"] || []
|
|
377
|
+
acceptable = t["acceptable"] || []
|
|
312
378
|
frozen_ref = t["frozen_at_sha"] == "root" ? `git -C #{Shellwords.escape(root)} rev-list --max-parents=0 HEAD`.lines.first.to_s.strip : t["frozen_at_sha"]
|
|
313
379
|
frozen_ok = ancestor?(root, frozen_ref)
|
|
314
380
|
prompt = build_prompt(catalog, t["utterance"], bootstrap_layer)
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
381
|
+
verdicts = []
|
|
382
|
+
replicas.times do |ri|
|
|
383
|
+
parsed, error = grade(model, timeout_s, prompt)
|
|
384
|
+
# A verdict that produced no observation contributes nothing to the totals,
|
|
385
|
+
# and an unmeasured task is indistinguishable from a routing failure in the
|
|
386
|
+
# numbers this bank reports. Both recoverable causes are sampling accidents
|
|
387
|
+
# rather than verdicts — a hard timeout is usually machine load, and
|
|
388
|
+
# unparseable output is usually one stray quote in a generated rationale —
|
|
389
|
+
# so each gets one retry.
|
|
390
|
+
# Repairing malformed output instead of re-asking for it was tried and removed:
|
|
391
|
+
# interpreting text that is by definition malformed has no natural boundary,
|
|
392
|
+
# and three review rounds each found a different shape that a repair would read
|
|
393
|
+
# as a verdict. Re-asking needs no such interpretation. Nothing else is retried:
|
|
394
|
+
# an auth failure and a missing CLI do reproduce on a second call.
|
|
395
|
+
if error&.start_with?("grader_timeout_") || error&.start_with?("no_json_in_output")
|
|
396
|
+
parsed, retry_error = grade(model, timeout_s, prompt)
|
|
397
|
+
error = parsed ? nil : retry_error
|
|
398
|
+
end
|
|
399
|
+
if error == "claude_not_found"
|
|
400
|
+
grader_available = false
|
|
401
|
+
break
|
|
402
|
+
end
|
|
403
|
+
selected = parsed && parsed["selected_skill"]
|
|
404
|
+
clarify = parsed && parsed["clarify"] == true
|
|
405
|
+
confidence = parsed && parsed["confidence"]
|
|
406
|
+
v_status =
|
|
407
|
+
if error then "ERROR"
|
|
408
|
+
elsif (selected == exp || acceptable.include?(selected)) && !must_not.include?(selected) then "PASS"
|
|
409
|
+
else "FAIL"
|
|
410
|
+
end
|
|
411
|
+
verdicts << { replica: ri + 1, selected: selected, clarify: clarify,
|
|
412
|
+
confidence: confidence, status: v_status, error: error }
|
|
333
413
|
end
|
|
334
|
-
|
|
335
|
-
|
|
414
|
+
break unless grader_available
|
|
415
|
+
# Task-level consensus over replicas (replicas=1 reproduces the old per-task
|
|
416
|
+
# semantics exactly). Conservative: one failing replica fails the task —
|
|
417
|
+
# a route that only sometimes lands is not a stable route.
|
|
418
|
+
observed = verdicts.reject { |v| v[:status] == "ERROR" }
|
|
336
419
|
status =
|
|
337
|
-
if
|
|
338
|
-
elsif
|
|
420
|
+
if observed.empty? then "ERROR"
|
|
421
|
+
elsif observed.all? { |v| v[:status] == "PASS" } then "PASS"
|
|
339
422
|
else "FAIL"
|
|
340
423
|
end
|
|
424
|
+
selections = observed.map { |v| v[:selected] }.uniq
|
|
425
|
+
# Failure-mode labels (vocabulary in references/eval-routing.md):
|
|
426
|
+
# absorbed — an utterance that should be rejected (expected "none")
|
|
427
|
+
# or kept away from named bait neighbors (must_not_route_to)
|
|
428
|
+
# was claimed by such a skill anyway
|
|
429
|
+
# ownership_split — replicas disagreed on the top pick (unstable ownership)
|
|
430
|
+
labels = []
|
|
431
|
+
absorbed = (exp == "none" && observed.any? { |v| v[:selected] && v[:selected] != "none" }) ||
|
|
432
|
+
observed.any? { |v| must_not.include?(v[:selected]) }
|
|
433
|
+
labels << "absorbed" if absorbed
|
|
434
|
+
labels << "ownership_split" if selections.size > 1
|
|
435
|
+
# A task where some replicas erred but others graded is PARTIALLY measured:
|
|
436
|
+
# the consensus above sees only the observed verdicts, so without this label
|
|
437
|
+
# a PASS+ERROR pair would read as a clean PASS and the run as fully sampled.
|
|
438
|
+
labels << "partial_error" if verdicts.any? { |v| v[:status] == "ERROR" } && !observed.empty?
|
|
341
439
|
results << {
|
|
342
|
-
id: t["id"], utterance: t["utterance"], expected: exp,
|
|
343
|
-
|
|
344
|
-
|
|
440
|
+
id: t["id"], utterance: t["utterance"], expected: exp, acceptable: acceptable,
|
|
441
|
+
selected: observed.first && observed.first[:selected],
|
|
442
|
+
confidence: observed.first && observed.first[:confidence],
|
|
443
|
+
clarify: observed.any? { |v| v[:clarify] },
|
|
444
|
+
acceptable_hit: observed.any? { |v| acceptable.include?(v[:selected]) },
|
|
445
|
+
must_not_route_to: must_not, status: status, labels: labels,
|
|
446
|
+
verdicts: verdicts, error: (verdicts.find { |v| v[:error] } || {})[:error],
|
|
447
|
+
frozen_at_sha_is_ancestor: frozen_ok
|
|
345
448
|
}
|
|
346
449
|
end
|
|
347
450
|
|
|
@@ -355,43 +458,114 @@ fails = results.select { |r| r[:status] == "FAIL" }
|
|
|
355
458
|
errors = results.select { |r| r[:status] == "ERROR" }
|
|
356
459
|
drift = results.reject { |r| r[:frozen_at_sha_is_ancestor] }
|
|
357
460
|
|
|
461
|
+
# First-class routing-quality metrics beyond pass/fail (counted over observed,
|
|
462
|
+
# non-ERROR verdicts): clarify rate, low-confidence rate, and — when replicas
|
|
463
|
+
# >= 2 — how often all replicas of one task picked the same top skill.
|
|
464
|
+
all_observed = results.flat_map { |r| r[:verdicts].reject { |v| v[:status] == "ERROR" } }
|
|
465
|
+
clarify_count = all_observed.count { |v| v[:clarify] }
|
|
466
|
+
low_conf_count = all_observed.count { |v| v[:confidence].is_a?(Numeric) && v[:confidence] < 0.5 }
|
|
467
|
+
# Replica-level error accounting: task-level `error` counts only fully
|
|
468
|
+
# unmeasured tasks, so a PASS+ERROR pair would otherwise report zero
|
|
469
|
+
# grader-errors while a replica silently went missing.
|
|
470
|
+
error_verdicts = results.sum { |r| r[:verdicts].count { |v| v[:status] == "ERROR" } }
|
|
471
|
+
partial_error_ids = results.select { |r| r[:labels].include?("partial_error") }.map { |r| r[:id] }
|
|
472
|
+
agreement_measured = 0
|
|
473
|
+
agreement_agree = 0
|
|
474
|
+
if replicas >= 2
|
|
475
|
+
results.each do |r|
|
|
476
|
+
obs = r[:verdicts].reject { |v| v[:status] == "ERROR" }
|
|
477
|
+
next if obs.size < 2
|
|
478
|
+
agreement_measured += 1
|
|
479
|
+
agreement_agree += 1 if obs.map { |v| v[:selected] }.uniq.size == 1
|
|
480
|
+
end
|
|
481
|
+
end
|
|
482
|
+
|
|
358
483
|
# baseline diff (newly failed / newly passed). The baseline is a prior report
|
|
359
|
-
# (a hash with a "results" array) or a bare results array.
|
|
484
|
+
# (a hash with a "results" array) or a bare results array. Reports from a
|
|
485
|
+
# different bank content or replica count are DIFFERENT RULERS (documented
|
|
486
|
+
# above): comparing them emits false regressions/improvements, so a
|
|
487
|
+
# demonstrated mismatch suppresses the diff instead of computing it. A bare
|
|
488
|
+
# results array carries no fingerprint — its comparability is unverifiable and
|
|
489
|
+
# is flagged as such rather than silently trusted.
|
|
360
490
|
newly_failed = []
|
|
361
491
|
newly_passed = []
|
|
492
|
+
baseline_comparable = nil
|
|
493
|
+
baseline_incomparable_reason = nil
|
|
362
494
|
if baseline_path && File.file?(baseline_path)
|
|
363
495
|
base_json = JSON.parse(File.read(baseline_path))
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
496
|
+
if base_json.is_a?(Hash)
|
|
497
|
+
base_bank_sha = base_json.dig("routing_surface", "bank_sha256")
|
|
498
|
+
base_replicas = base_json["replicas"] || 1 # legacy reports predate --replicas
|
|
499
|
+
if base_bank_sha && base_bank_sha != routing_surface[:bank_sha256]
|
|
500
|
+
baseline_comparable = false
|
|
501
|
+
baseline_incomparable_reason = "bank content differs (baseline #{base_bank_sha[0, 12]}… vs current #{routing_surface[:bank_sha256][0, 12]}…)"
|
|
502
|
+
elsif base_replicas != replicas
|
|
503
|
+
baseline_comparable = false
|
|
504
|
+
baseline_incomparable_reason = "replica count differs (baseline #{base_replicas} vs current #{replicas})"
|
|
505
|
+
else
|
|
506
|
+
baseline_comparable = base_bank_sha ? true : "unverified (baseline carries no bank fingerprint)"
|
|
507
|
+
end
|
|
508
|
+
else
|
|
509
|
+
baseline_comparable = "unverified (bare results array carries no fingerprint)"
|
|
510
|
+
end
|
|
511
|
+
unless baseline_comparable == false
|
|
512
|
+
base_results = base_json.is_a?(Hash) ? (base_json["results"] || []) : base_json
|
|
513
|
+
base_status = base_results.to_h { |r| [r["id"], r["status"]] }
|
|
514
|
+
results.each do |r|
|
|
515
|
+
was = base_status[r[:id]]
|
|
516
|
+
newly_failed << r[:id] if was == "PASS" && r[:status] == "FAIL"
|
|
517
|
+
newly_passed << r[:id] if was == "FAIL" && r[:status] == "PASS"
|
|
518
|
+
end
|
|
370
519
|
end
|
|
371
520
|
end
|
|
372
521
|
|
|
373
522
|
report = {
|
|
374
523
|
model: model, tasks: results.size, pass: passes, fail: fails.size, error: errors.size,
|
|
524
|
+
replicas: replicas, verdicts: all_observed.size,
|
|
525
|
+
error_verdicts: error_verdicts, partial_error_tasks: partial_error_ids,
|
|
526
|
+
clarify_count: clarify_count, low_confidence_count: low_conf_count,
|
|
527
|
+
replica_agreement: (replicas >= 2 ? { agree: agreement_agree, measured: agreement_measured } : nil),
|
|
375
528
|
desc_budget_chars: desc_budget, routing_surface: routing_surface,
|
|
376
529
|
co_change_bank_and_descriptions: co_change, co_change_check_available: co_change_check_ok,
|
|
377
530
|
frozen_drift: drift.map { |r| r[:id] },
|
|
531
|
+
baseline_comparable: baseline_comparable,
|
|
532
|
+
baseline_incomparable_reason: baseline_incomparable_reason,
|
|
378
533
|
newly_failed: newly_failed, newly_passed: newly_passed, results: results
|
|
379
534
|
}
|
|
380
535
|
File.write(json_path, JSON.pretty_generate(report)) if json_path
|
|
381
536
|
|
|
382
537
|
puts "eval-routing-bank (#{model}): #{passes}/#{results.size} pass, #{fails.size} fail, #{errors.size} grader-error"
|
|
383
538
|
puts " arm: desc-budget-chars=#{desc_budget}" if desc_budget
|
|
539
|
+
unless all_observed.empty?
|
|
540
|
+
line = " clarify: #{clarify_count}/#{all_observed.size} verdicts, low-confidence(<0.5): #{low_conf_count}/#{all_observed.size}"
|
|
541
|
+
line += ", replica top1 agreement: #{agreement_agree}/#{agreement_measured}" if replicas >= 2
|
|
542
|
+
puts line
|
|
543
|
+
end
|
|
544
|
+
if error_verdicts.positive?
|
|
545
|
+
puts " ⚠ grader-error verdicts: #{error_verdicts}#{partial_error_ids.empty? ? '' : " (partially measured tasks: #{partial_error_ids.join(', ')})"}"
|
|
546
|
+
end
|
|
384
547
|
puts " ⚠ co-change check unavailable: base ref #{base.inspect} not resolvable — could not verify bank/description co-change" unless co_change_check_ok
|
|
385
548
|
puts " ⚠ co-change: this change touches BOTH the bank and a SKILL.md description (verify the bank was not edited to pass)" if co_change
|
|
386
549
|
puts " ⚠ frozen-drift (not ancestor of HEAD, excluded from regression judgment): #{drift.map { |r| r[:id] }.join(', ')}" unless drift.empty?
|
|
387
550
|
unless fails.empty?
|
|
388
551
|
puts " FAIL:"
|
|
389
|
-
fails.each
|
|
552
|
+
fails.each do |r|
|
|
553
|
+
tag = r[:labels].empty? ? "" : " [#{r[:labels].join(',')}]"
|
|
554
|
+
obs = r[:verdicts].reject { |v| v[:status] == "ERROR" }
|
|
555
|
+
got = obs.map { |v| v[:selected].inspect }.uniq.join(" | ")
|
|
556
|
+
conf = obs.map { |v| v[:confidence] }.join(",")
|
|
557
|
+
puts " - #{r[:id]}: expected #{r[:expected]} got #{got} (conf #{conf})#{tag}"
|
|
558
|
+
end
|
|
390
559
|
end
|
|
391
560
|
unless errors.empty?
|
|
392
561
|
puts " GRADER-ERROR:"
|
|
393
562
|
errors.each { |r| puts " - #{r[:id]}: #{r[:error]}" }
|
|
394
563
|
end
|
|
564
|
+
if baseline_comparable == false
|
|
565
|
+
puts " ⚠ baseline not compared — different ruler: #{baseline_incomparable_reason}"
|
|
566
|
+
elsif baseline_comparable.is_a?(String)
|
|
567
|
+
puts " ⚠ baseline comparability #{baseline_comparable}"
|
|
568
|
+
end
|
|
395
569
|
unless newly_failed.empty? && newly_passed.empty?
|
|
396
570
|
puts " vs baseline: newly_failed=#{newly_failed.join(',')} newly_passed=#{newly_passed.join(',')}"
|
|
397
571
|
end
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
|
-
# Extraction-owned autonomous review wrapper: one review plus
|
|
2
|
+
# Extraction-owned autonomous review wrapper: one review plus one challenge.
|
|
3
3
|
set -euo pipefail
|
|
4
4
|
|
|
5
5
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
@@ -8,7 +8,7 @@ CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
|
|
|
8
8
|
for arg in "$@"; do
|
|
9
9
|
case "$arg" in
|
|
10
10
|
--challenge-b*)
|
|
11
|
-
echo "extraction_review_gate_error: challenge budget is fixed at
|
|
11
|
+
echo "extraction_review_gate_error: challenge budget is fixed at 1" >&2
|
|
12
12
|
exit 2
|
|
13
13
|
;;
|
|
14
14
|
esac
|
|
@@ -19,4 +19,4 @@ if [[ ! -x "$CONTROLLER" ]]; then
|
|
|
19
19
|
exit 2
|
|
20
20
|
fi
|
|
21
21
|
|
|
22
|
-
exec bash "$CONTROLLER" --challenge-budget
|
|
22
|
+
exec bash "$CONTROLLER" --challenge-budget 1 "$@"
|