@ccoalm/ccl-skills 0.8.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +5 -0
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +6 -0
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/feature-risk-router/SKILL.md +3 -1
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/architecture-playbook.md +1 -1
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/multi-tenant-isolation.md +1 -1
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/state-machine-task-patterns.md +2 -0
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/inference-capacity-operations.md +24 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/llm-client-gateway.md +1 -1
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +4 -1
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/contracts-and-state.md +5 -0
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +1 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -2
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/metrics-conventions.md +8 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/sli-slo-design.md +2 -2
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/canary-and-rollout-strategy.md +16 -2
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/promotion-gate-and-review.md +9 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +12 -12
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/code-review-checklist.md +4 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +1 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +2 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +2 -2
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +1 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +8 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/multi-tenant-isolation.md +1 -1
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/state-machine-task-patterns.md +2 -0
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/SKILL.md +1 -1
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +12 -15
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +13 -0
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +37 -30
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +81 -0
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +12 -0
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
  44. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
  45. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
  46. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
  47. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
  48. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
  49. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
  50. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
  51. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
  52. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
  53. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
  54. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
  55. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
  56. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
  57. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
  58. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
  59. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
  60. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
  61. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/classical-test-design-techniques.md +1 -1
  62. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc-review-and-prioritization.md +1 -1
  63. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/update-lifecycle.md +2 -0
  64. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +9 -9
  65. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +5 -1
  66. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/e2e-real-flow-testing.md +2 -2
  67. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/integration-contract-testing.md +10 -0
  68. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-code-authoring-patterns.md +2 -2
  69. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-topology-and-commands.md +1 -1
  70. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +2 -1
  71. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/annotation-driven-revision.md +9 -0
  72. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/figure-and-table-craft.md +8 -2
  73. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -0
  74. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/react-architecture.md +3 -0
  75. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-quality-release.md +37 -4
  76. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-ui-quality.md +10 -1
  77. package/dist/assets/release.json +127 -67
  78. package/package.json +1 -1
@@ -16,13 +16,23 @@
16
16
  # Usage:
17
17
  # eval-routing-bank.rb <repo-root> [--bank <path>] [--model <name>] [--limit N]
18
18
  # [--dry-run] [--json <path>] [--baseline <path>] [--timeout S]
19
- # [--desc-budget-chars N] [--with-bootstrap]
19
+ # [--desc-budget-chars N] [--with-bootstrap] [--replicas N]
20
20
  #
21
21
  # --desc-budget-chars N simulates a consumer that truncates each skill description
22
22
  # to its first N characters before routing (e.g. Codex compresses the skill listing
23
23
  # under a ~2%-of-context budget, 8000 chars when the window is unknown, shortening
24
24
  # descriptions first). Run the bank once plain and once with a budget to see which
25
25
  # routes only survive on the description tail — those triggers need front-loading.
26
+ #
27
+ # expected_skill "none" marks negative controls (out-of-library utterances) and
28
+ # coverage-gap probes (in-domain utterances no skill owns): the correct outcome
29
+ # is rejection, and a catalog skill claiming them is labeled "absorbed".
30
+ # --replicas N grades each task N times: task status is the conservative
31
+ # consensus (every observed replica must PASS), replica top1 agreement is
32
+ # reported, and disagreeing replicas label the task "ownership_split". clarify
33
+ # and low-confidence counts are first-class report fields either way.
34
+ # Reports from different (bank, replicas) configurations are different rulers —
35
+ # do not diff them as a regression signal.
26
36
  # Exit: 0 = ran (advisory); 2 = usage error; 3 = grader entirely unavailable.
27
37
 
28
38
  require "yaml"
@@ -39,7 +49,7 @@ end
39
49
 
40
50
  root = ARGV[0]
41
51
  if root.nil? || root.start_with?("-")
42
- warn "usage: eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--timeout S] [--desc-budget-chars N]"
52
+ warn "usage: eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--timeout S] [--desc-budget-chars N] [--replicas N]"
43
53
  exit 2
44
54
  end
45
55
  bank_path = arg("--bank", File.join(root, "eval", "routing-tasks.jsonl"))
@@ -65,6 +75,17 @@ if ARGV.include?("--desc-budget-chars")
65
75
  desc_budget = b.to_i
66
76
  end
67
77
  timeout_s = (t = arg("--timeout")) ? t.to_i : 60
78
+ replicas = 1
79
+ if ARGV.include?("--replicas")
80
+ r = arg("--replicas")
81
+ # Same strictness as --desc-budget-chars: a silent to_i coercion would turn a
82
+ # typo into "1 replica" and the agreement metric would quietly measure nothing.
83
+ unless r && r =~ /\A[1-9]\d*\z/
84
+ warn "--replicas requires a positive integer value, got #{r.inspect}"
85
+ exit 2
86
+ end
87
+ replicas = r.to_i
88
+ end
68
89
 
69
90
  unless File.file?(bank_path)
70
91
  warn "eval_bank_missing: #{bank_path}"
@@ -89,16 +110,57 @@ end
89
110
  tasks = tasks.first(limit) if limit
90
111
 
91
112
  # Schema validation: a task must be answerable, and must not be self-contradictory.
113
+ # expected_skill "none" is the negative-control / coverage-gap sentinel: the
114
+ # correct routing outcome is that NO catalog skill claims the utterance.
115
+ all_outcomes = Dir[File.join(root, "skills", "*", "SKILL.md")]
116
+ .map { |p| File.basename(File.dirname(p)) } + ["none"]
92
117
  tasks.each do |t|
93
118
  id = t["id"] || "(no id)"
94
119
  if t["utterance"].to_s.strip.empty? || t["expected_skill"].to_s.strip.empty?
95
120
  warn "eval_bank_invalid_task: #{id}: utterance and expected_skill are required"
96
121
  exit 2
97
122
  end
123
+ # Type-check the ORIGINAL values FIRST: a present-but-non-list field (e.g. ""
124
+ # or a bare string or a number) must fail as a usage error before any
125
+ # membership check touches it — Ruby strings are truthy and respond to
126
+ # include? (substring semantics), and non-strings would raise a bare
127
+ # NoMethodError instead of the documented invalid-bank diagnostic.
128
+ %w[must_not_route_to acceptable].each do |f|
129
+ next unless t.key?(f)
130
+ unless t[f].is_a?(Array)
131
+ warn "eval_bank_invalid_task: #{id}: #{f} must be a list, got #{t[f].inspect}"
132
+ exit 2
133
+ end
134
+ end
98
135
  if (t["must_not_route_to"] || []).include?(t["expected_skill"])
99
136
  warn "eval_bank_invalid_task: #{id}: expected_skill is also in must_not_route_to (impossible)"
100
137
  exit 2
101
138
  end
139
+ if (t["must_not_route_to"] || []).include?("none")
140
+ warn "eval_bank_invalid_task: #{id}: \"none\" is a sentinel outcome, not a routable target for must_not_route_to"
141
+ exit 2
142
+ end
143
+ # Optional acceptable[] names defensible alternate outcomes (a skill name or
144
+ # "none") for utterances with more than one correct route — e.g. a coverage-gap
145
+ # ask where both coordinator intake and rejection are right. It must not
146
+ # restate expected_skill or contradict must_not_route_to.
147
+ acc = t["acceptable"] || []
148
+ if acc.include?(t["expected_skill"])
149
+ warn "eval_bank_invalid_task: #{id}: acceptable restates expected_skill"
150
+ exit 2
151
+ end
152
+ unless (acc & (t["must_not_route_to"] || [])).empty?
153
+ warn "eval_bank_invalid_task: #{id}: acceptable and must_not_route_to overlap (contradictory)"
154
+ exit 2
155
+ end
156
+ # Anti-gaming: expected + acceptable must leave at least one outcome that
157
+ # would FAIL the row. A row covering the complete catalog-and-none outcome
158
+ # space passes on every valid grader selection — a vacuous fixture that fakes
159
+ # green regardless of description behavior.
160
+ if (all_outcomes - ([t["expected_skill"]] + acc)).empty?
161
+ warn "eval_bank_invalid_task: #{id}: expected_skill plus acceptable cover every possible outcome (vacuous row)"
162
+ exit 2
163
+ end
102
164
  end
103
165
 
104
166
  # --- anti-game ---------------------------------------------------------------
@@ -188,9 +250,12 @@ def build_prompt(catalog, utterance, bootstrap_layer = nil)
188
250
  user utterance and the catalog of skills (name + description), pick the SINGLE
189
251
  best skill that should handle it. Use only the catalog; route by the
190
252
  descriptions' Use-when / Proactively / Skip rules.
253
+ If NO skill's description covers the utterance, answer "none" — do not
254
+ force-fit the nearest neighbor. Set "clarify" to true only when you would
255
+ need to ask the user a clarifying question before committing to a route.
191
256
 
192
257
  Output ONLY a JSON object on one line, no other text:
193
- {"selected_skill": "<exact skill name from the catalog>", "confidence": <0.0-1.0>, "rationale_short": "<one short clause>"}
258
+ {"selected_skill": "<exact skill name from the catalog, or none>", "clarify": <true|false>, "confidence": <0.0-1.0>, "rationale_short": "<one short clause>"}
194
259
 
195
260
  #{bootstrap_layer ? "== ALWAYS-ON ENTRY ROUTING LAYER (injected into every session; takes precedence for entry routing) ==\n#{bootstrap_layer}\n" : ""}
196
261
  == SKILL CATALOG ==
@@ -309,39 +374,77 @@ grader_available = true
309
374
  tasks.each do |t|
310
375
  exp = t["expected_skill"]
311
376
  must_not = t["must_not_route_to"] || []
377
+ acceptable = t["acceptable"] || []
312
378
  frozen_ref = t["frozen_at_sha"] == "root" ? `git -C #{Shellwords.escape(root)} rev-list --max-parents=0 HEAD`.lines.first.to_s.strip : t["frozen_at_sha"]
313
379
  frozen_ok = ancestor?(root, frozen_ref)
314
380
  prompt = build_prompt(catalog, t["utterance"], bootstrap_layer)
315
- parsed, error = grade(model, timeout_s, prompt)
316
- # A task that produced no observation contributes nothing to the totals, and an
317
- # unmeasured task is indistinguishable from a routing failure in the numbers
318
- # this bank reports. Both recoverable causes are sampling accidents rather than
319
- # verdicts — a hard timeout is usually machine load, and unparseable output is
320
- # usually one stray quote in a generated rationale — so each gets one retry.
321
- # Repairing malformed output instead of re-asking for it was tried and removed:
322
- # interpreting text that is by definition malformed has no natural boundary,
323
- # and three review rounds each found a different shape that a repair would read
324
- # as a verdict. Re-asking needs no such interpretation. Nothing else is retried:
325
- # an auth failure and a missing CLI do reproduce on a second call.
326
- if error&.start_with?("grader_timeout_") || error&.start_with?("no_json_in_output")
327
- parsed, retry_error = grade(model, timeout_s, prompt)
328
- error = parsed ? nil : retry_error
329
- end
330
- if error == "claude_not_found"
331
- grader_available = false
332
- break
381
+ verdicts = []
382
+ replicas.times do |ri|
383
+ parsed, error = grade(model, timeout_s, prompt)
384
+ # A verdict that produced no observation contributes nothing to the totals,
385
+ # and an unmeasured task is indistinguishable from a routing failure in the
386
+ # numbers this bank reports. Both recoverable causes are sampling accidents
387
+ # rather than verdicts — a hard timeout is usually machine load, and
388
+ # unparseable output is usually one stray quote in a generated rationale —
389
+ # so each gets one retry.
390
+ # Repairing malformed output instead of re-asking for it was tried and removed:
391
+ # interpreting text that is by definition malformed has no natural boundary,
392
+ # and three review rounds each found a different shape that a repair would read
393
+ # as a verdict. Re-asking needs no such interpretation. Nothing else is retried:
394
+ # an auth failure and a missing CLI do reproduce on a second call.
395
+ if error&.start_with?("grader_timeout_") || error&.start_with?("no_json_in_output")
396
+ parsed, retry_error = grade(model, timeout_s, prompt)
397
+ error = parsed ? nil : retry_error
398
+ end
399
+ if error == "claude_not_found"
400
+ grader_available = false
401
+ break
402
+ end
403
+ selected = parsed && parsed["selected_skill"]
404
+ clarify = parsed && parsed["clarify"] == true
405
+ confidence = parsed && parsed["confidence"]
406
+ v_status =
407
+ if error then "ERROR"
408
+ elsif (selected == exp || acceptable.include?(selected)) && !must_not.include?(selected) then "PASS"
409
+ else "FAIL"
410
+ end
411
+ verdicts << { replica: ri + 1, selected: selected, clarify: clarify,
412
+ confidence: confidence, status: v_status, error: error }
333
413
  end
334
- selected = parsed && parsed["selected_skill"]
335
- confidence = parsed && parsed["confidence"]
414
+ break unless grader_available
415
+ # Task-level consensus over replicas (replicas=1 reproduces the old per-task
416
+ # semantics exactly). Conservative: one failing replica fails the task —
417
+ # a route that only sometimes lands is not a stable route.
418
+ observed = verdicts.reject { |v| v[:status] == "ERROR" }
336
419
  status =
337
- if error then "ERROR"
338
- elsif selected == exp && !must_not.include?(selected) then "PASS"
420
+ if observed.empty? then "ERROR"
421
+ elsif observed.all? { |v| v[:status] == "PASS" } then "PASS"
339
422
  else "FAIL"
340
423
  end
424
+ selections = observed.map { |v| v[:selected] }.uniq
425
+ # Failure-mode labels (vocabulary in references/eval-routing.md):
426
+ # absorbed — an utterance that should be rejected (expected "none")
427
+ # or kept away from named bait neighbors (must_not_route_to)
428
+ # was claimed by such a skill anyway
429
+ # ownership_split — replicas disagreed on the top pick (unstable ownership)
430
+ labels = []
431
+ absorbed = (exp == "none" && observed.any? { |v| v[:selected] && v[:selected] != "none" }) ||
432
+ observed.any? { |v| must_not.include?(v[:selected]) }
433
+ labels << "absorbed" if absorbed
434
+ labels << "ownership_split" if selections.size > 1
435
+ # A task where some replicas erred but others graded is PARTIALLY measured:
436
+ # the consensus above sees only the observed verdicts, so without this label
437
+ # a PASS+ERROR pair would read as a clean PASS and the run as fully sampled.
438
+ labels << "partial_error" if verdicts.any? { |v| v[:status] == "ERROR" } && !observed.empty?
341
439
  results << {
342
- id: t["id"], utterance: t["utterance"], expected: exp, selected: selected,
343
- confidence: confidence, must_not_route_to: must_not, status: status,
344
- error: error, frozen_at_sha_is_ancestor: frozen_ok
440
+ id: t["id"], utterance: t["utterance"], expected: exp, acceptable: acceptable,
441
+ selected: observed.first && observed.first[:selected],
442
+ confidence: observed.first && observed.first[:confidence],
443
+ clarify: observed.any? { |v| v[:clarify] },
444
+ acceptable_hit: observed.any? { |v| acceptable.include?(v[:selected]) },
445
+ must_not_route_to: must_not, status: status, labels: labels,
446
+ verdicts: verdicts, error: (verdicts.find { |v| v[:error] } || {})[:error],
447
+ frozen_at_sha_is_ancestor: frozen_ok
345
448
  }
346
449
  end
347
450
 
@@ -355,43 +458,114 @@ fails = results.select { |r| r[:status] == "FAIL" }
355
458
  errors = results.select { |r| r[:status] == "ERROR" }
356
459
  drift = results.reject { |r| r[:frozen_at_sha_is_ancestor] }
357
460
 
461
+ # First-class routing-quality metrics beyond pass/fail (counted over observed,
462
+ # non-ERROR verdicts): clarify rate, low-confidence rate, and — when replicas
463
+ # >= 2 — how often all replicas of one task picked the same top skill.
464
+ all_observed = results.flat_map { |r| r[:verdicts].reject { |v| v[:status] == "ERROR" } }
465
+ clarify_count = all_observed.count { |v| v[:clarify] }
466
+ low_conf_count = all_observed.count { |v| v[:confidence].is_a?(Numeric) && v[:confidence] < 0.5 }
467
+ # Replica-level error accounting: task-level `error` counts only fully
468
+ # unmeasured tasks, so a PASS+ERROR pair would otherwise report zero
469
+ # grader-errors while a replica silently went missing.
470
+ error_verdicts = results.sum { |r| r[:verdicts].count { |v| v[:status] == "ERROR" } }
471
+ partial_error_ids = results.select { |r| r[:labels].include?("partial_error") }.map { |r| r[:id] }
472
+ agreement_measured = 0
473
+ agreement_agree = 0
474
+ if replicas >= 2
475
+ results.each do |r|
476
+ obs = r[:verdicts].reject { |v| v[:status] == "ERROR" }
477
+ next if obs.size < 2
478
+ agreement_measured += 1
479
+ agreement_agree += 1 if obs.map { |v| v[:selected] }.uniq.size == 1
480
+ end
481
+ end
482
+
358
483
  # baseline diff (newly failed / newly passed). The baseline is a prior report
359
- # (a hash with a "results" array) or a bare results array.
484
+ # (a hash with a "results" array) or a bare results array. Reports from a
485
+ # different bank content or replica count are DIFFERENT RULERS (documented
486
+ # above): comparing them emits false regressions/improvements, so a
487
+ # demonstrated mismatch suppresses the diff instead of computing it. A bare
488
+ # results array carries no fingerprint — its comparability is unverifiable and
489
+ # is flagged as such rather than silently trusted.
360
490
  newly_failed = []
361
491
  newly_passed = []
492
+ baseline_comparable = nil
493
+ baseline_incomparable_reason = nil
362
494
  if baseline_path && File.file?(baseline_path)
363
495
  base_json = JSON.parse(File.read(baseline_path))
364
- base_results = base_json.is_a?(Hash) ? (base_json["results"] || []) : base_json
365
- base_status = base_results.to_h { |r| [r["id"], r["status"]] }
366
- results.each do |r|
367
- was = base_status[r[:id]]
368
- newly_failed << r[:id] if was == "PASS" && r[:status] == "FAIL"
369
- newly_passed << r[:id] if was == "FAIL" && r[:status] == "PASS"
496
+ if base_json.is_a?(Hash)
497
+ base_bank_sha = base_json.dig("routing_surface", "bank_sha256")
498
+ base_replicas = base_json["replicas"] || 1 # legacy reports predate --replicas
499
+ if base_bank_sha && base_bank_sha != routing_surface[:bank_sha256]
500
+ baseline_comparable = false
501
+ baseline_incomparable_reason = "bank content differs (baseline #{base_bank_sha[0, 12]}… vs current #{routing_surface[:bank_sha256][0, 12]}…)"
502
+ elsif base_replicas != replicas
503
+ baseline_comparable = false
504
+ baseline_incomparable_reason = "replica count differs (baseline #{base_replicas} vs current #{replicas})"
505
+ else
506
+ baseline_comparable = base_bank_sha ? true : "unverified (baseline carries no bank fingerprint)"
507
+ end
508
+ else
509
+ baseline_comparable = "unverified (bare results array carries no fingerprint)"
510
+ end
511
+ unless baseline_comparable == false
512
+ base_results = base_json.is_a?(Hash) ? (base_json["results"] || []) : base_json
513
+ base_status = base_results.to_h { |r| [r["id"], r["status"]] }
514
+ results.each do |r|
515
+ was = base_status[r[:id]]
516
+ newly_failed << r[:id] if was == "PASS" && r[:status] == "FAIL"
517
+ newly_passed << r[:id] if was == "FAIL" && r[:status] == "PASS"
518
+ end
370
519
  end
371
520
  end
372
521
 
373
522
  report = {
374
523
  model: model, tasks: results.size, pass: passes, fail: fails.size, error: errors.size,
524
+ replicas: replicas, verdicts: all_observed.size,
525
+ error_verdicts: error_verdicts, partial_error_tasks: partial_error_ids,
526
+ clarify_count: clarify_count, low_confidence_count: low_conf_count,
527
+ replica_agreement: (replicas >= 2 ? { agree: agreement_agree, measured: agreement_measured } : nil),
375
528
  desc_budget_chars: desc_budget, routing_surface: routing_surface,
376
529
  co_change_bank_and_descriptions: co_change, co_change_check_available: co_change_check_ok,
377
530
  frozen_drift: drift.map { |r| r[:id] },
531
+ baseline_comparable: baseline_comparable,
532
+ baseline_incomparable_reason: baseline_incomparable_reason,
378
533
  newly_failed: newly_failed, newly_passed: newly_passed, results: results
379
534
  }
380
535
  File.write(json_path, JSON.pretty_generate(report)) if json_path
381
536
 
382
537
  puts "eval-routing-bank (#{model}): #{passes}/#{results.size} pass, #{fails.size} fail, #{errors.size} grader-error"
383
538
  puts " arm: desc-budget-chars=#{desc_budget}" if desc_budget
539
+ unless all_observed.empty?
540
+ line = " clarify: #{clarify_count}/#{all_observed.size} verdicts, low-confidence(<0.5): #{low_conf_count}/#{all_observed.size}"
541
+ line += ", replica top1 agreement: #{agreement_agree}/#{agreement_measured}" if replicas >= 2
542
+ puts line
543
+ end
544
+ if error_verdicts.positive?
545
+ puts " ⚠ grader-error verdicts: #{error_verdicts}#{partial_error_ids.empty? ? '' : " (partially measured tasks: #{partial_error_ids.join(', ')})"}"
546
+ end
384
547
  puts " ⚠ co-change check unavailable: base ref #{base.inspect} not resolvable — could not verify bank/description co-change" unless co_change_check_ok
385
548
  puts " ⚠ co-change: this change touches BOTH the bank and a SKILL.md description (verify the bank was not edited to pass)" if co_change
386
549
  puts " ⚠ frozen-drift (not ancestor of HEAD, excluded from regression judgment): #{drift.map { |r| r[:id] }.join(', ')}" unless drift.empty?
387
550
  unless fails.empty?
388
551
  puts " FAIL:"
389
- fails.each { |r| puts " - #{r[:id]}: expected #{r[:expected]} got #{r[:selected].inspect} (conf #{r[:confidence]})" }
552
+ fails.each do |r|
553
+ tag = r[:labels].empty? ? "" : " [#{r[:labels].join(',')}]"
554
+ obs = r[:verdicts].reject { |v| v[:status] == "ERROR" }
555
+ got = obs.map { |v| v[:selected].inspect }.uniq.join(" | ")
556
+ conf = obs.map { |v| v[:confidence] }.join(",")
557
+ puts " - #{r[:id]}: expected #{r[:expected]} got #{got} (conf #{conf})#{tag}"
558
+ end
390
559
  end
391
560
  unless errors.empty?
392
561
  puts " GRADER-ERROR:"
393
562
  errors.each { |r| puts " - #{r[:id]}: #{r[:error]}" }
394
563
  end
564
+ if baseline_comparable == false
565
+ puts " ⚠ baseline not compared — different ruler: #{baseline_incomparable_reason}"
566
+ elsif baseline_comparable.is_a?(String)
567
+ puts " ⚠ baseline comparability #{baseline_comparable}"
568
+ end
395
569
  unless newly_failed.empty? && newly_passed.empty?
396
570
  puts " vs baseline: newly_failed=#{newly_failed.join(',')} newly_passed=#{newly_passed.join(',')}"
397
571
  end
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env bash
2
- # Extraction-owned autonomous review wrapper: one review plus two challenges.
2
+ # Extraction-owned autonomous review wrapper: one review plus one challenge.
3
3
  set -euo pipefail
4
4
 
5
5
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
@@ -8,7 +8,7 @@ CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
8
8
  for arg in "$@"; do
9
9
  case "$arg" in
10
10
  --challenge-b*)
11
- echo "extraction_review_gate_error: challenge budget is fixed at 2" >&2
11
+ echo "extraction_review_gate_error: challenge budget is fixed at 1" >&2
12
12
  exit 2
13
13
  ;;
14
14
  esac
@@ -19,4 +19,4 @@ if [[ ! -x "$CONTROLLER" ]]; then
19
19
  exit 2
20
20
  fi
21
21
 
22
- exec bash "$CONTROLLER" --challenge-budget 2 "$@"
22
+ exec bash "$CONTROLLER" --challenge-budget 1 "$@"