@ccoalm/ccl-skills 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/feature-risk-router/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/architecture-playbook.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/multi-tenant-isolation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/state-machine-task-patterns.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/inference-capacity-operations.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/llm-client-gateway.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +4 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/contracts-and-state.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/metrics-conventions.md +8 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/sli-slo-design.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/canary-and-rollout-strategy.md +16 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/promotion-gate-and-review.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +12 -12
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/code-review-checklist.md +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/multi-tenant-isolation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/state-machine-task-patterns.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +12 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +37 -30
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +81 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/classical-test-design-techniques.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc-review-and-prioritization.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/update-lifecycle.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +9 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +5 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/e2e-real-flow-testing.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/integration-contract-testing.md +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-code-authoring-patterns.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-topology-and-commands.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +2 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/annotation-driven-revision.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/figure-and-table-craft.md +8 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/react-architecture.md +3 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-quality-release.md +37 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-ui-quality.md +10 -1
- package/dist/assets/release.json +127 -67
- package/package.json +1 -1
|
@@ -16,13 +16,23 @@
|
|
|
16
16
|
# Usage:
|
|
17
17
|
# eval-routing-bank.rb <repo-root> [--bank <path>] [--model <name>] [--limit N]
|
|
18
18
|
# [--dry-run] [--json <path>] [--baseline <path>] [--timeout S]
|
|
19
|
-
# [--desc-budget-chars N] [--with-bootstrap]
|
|
19
|
+
# [--desc-budget-chars N] [--with-bootstrap] [--replicas N]
|
|
20
20
|
#
|
|
21
21
|
# --desc-budget-chars N simulates a consumer that truncates each skill description
|
|
22
22
|
# to its first N characters before routing (e.g. Codex compresses the skill listing
|
|
23
23
|
# under a ~2%-of-context budget, 8000 chars when the window is unknown, shortening
|
|
24
24
|
# descriptions first). Run the bank once plain and once with a budget to see which
|
|
25
25
|
# routes only survive on the description tail — those triggers need front-loading.
|
|
26
|
+
#
|
|
27
|
+
# expected_skill "none" marks negative controls (out-of-library utterances) and
|
|
28
|
+
# coverage-gap probes (in-domain utterances no skill owns): the correct outcome
|
|
29
|
+
# is rejection, and a catalog skill claiming them is labeled "absorbed".
|
|
30
|
+
# --replicas N grades each task N times: task status is the conservative
|
|
31
|
+
# consensus (every observed replica must PASS), replica top1 agreement is
|
|
32
|
+
# reported, and disagreeing replicas label the task "ownership_split". clarify
|
|
33
|
+
# and low-confidence counts are first-class report fields either way.
|
|
34
|
+
# Reports from different (bank, replicas) configurations are different rulers —
|
|
35
|
+
# do not diff them as a regression signal.
|
|
26
36
|
# Exit: 0 = ran (advisory); 2 = usage error; 3 = grader entirely unavailable.
|
|
27
37
|
|
|
28
38
|
require "yaml"
|
|
@@ -39,7 +49,7 @@ end
|
|
|
39
49
|
|
|
40
50
|
root = ARGV[0]
|
|
41
51
|
if root.nil? || root.start_with?("-")
|
|
42
|
-
warn "usage: eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--timeout S] [--desc-budget-chars N]"
|
|
52
|
+
warn "usage: eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--timeout S] [--desc-budget-chars N] [--replicas N]"
|
|
43
53
|
exit 2
|
|
44
54
|
end
|
|
45
55
|
bank_path = arg("--bank", File.join(root, "eval", "routing-tasks.jsonl"))
|
|
@@ -65,6 +75,17 @@ if ARGV.include?("--desc-budget-chars")
|
|
|
65
75
|
desc_budget = b.to_i
|
|
66
76
|
end
|
|
67
77
|
timeout_s = (t = arg("--timeout")) ? t.to_i : 60
|
|
78
|
+
replicas = 1
|
|
79
|
+
if ARGV.include?("--replicas")
|
|
80
|
+
r = arg("--replicas")
|
|
81
|
+
# Same strictness as --desc-budget-chars: a silent to_i coercion would turn a
|
|
82
|
+
# typo into "1 replica" and the agreement metric would quietly measure nothing.
|
|
83
|
+
unless r && r =~ /\A[1-9]\d*\z/
|
|
84
|
+
warn "--replicas requires a positive integer value, got #{r.inspect}"
|
|
85
|
+
exit 2
|
|
86
|
+
end
|
|
87
|
+
replicas = r.to_i
|
|
88
|
+
end
|
|
68
89
|
|
|
69
90
|
unless File.file?(bank_path)
|
|
70
91
|
warn "eval_bank_missing: #{bank_path}"
|
|
@@ -89,16 +110,57 @@ end
|
|
|
89
110
|
tasks = tasks.first(limit) if limit
|
|
90
111
|
|
|
91
112
|
# Schema validation: a task must be answerable, and must not be self-contradictory.
|
|
113
|
+
# expected_skill "none" is the negative-control / coverage-gap sentinel: the
|
|
114
|
+
# correct routing outcome is that NO catalog skill claims the utterance.
|
|
115
|
+
all_outcomes = Dir[File.join(root, "skills", "*", "SKILL.md")]
|
|
116
|
+
.map { |p| File.basename(File.dirname(p)) } + ["none"]
|
|
92
117
|
tasks.each do |t|
|
|
93
118
|
id = t["id"] || "(no id)"
|
|
94
119
|
if t["utterance"].to_s.strip.empty? || t["expected_skill"].to_s.strip.empty?
|
|
95
120
|
warn "eval_bank_invalid_task: #{id}: utterance and expected_skill are required"
|
|
96
121
|
exit 2
|
|
97
122
|
end
|
|
123
|
+
# Type-check the ORIGINAL values FIRST: a present-but-non-list field (e.g. ""
|
|
124
|
+
# or a bare string or a number) must fail as a usage error before any
|
|
125
|
+
# membership check touches it — Ruby strings are truthy and respond to
|
|
126
|
+
# include? (substring semantics), and non-strings would raise a bare
|
|
127
|
+
# NoMethodError instead of the documented invalid-bank diagnostic.
|
|
128
|
+
%w[must_not_route_to acceptable].each do |f|
|
|
129
|
+
next unless t.key?(f)
|
|
130
|
+
unless t[f].is_a?(Array)
|
|
131
|
+
warn "eval_bank_invalid_task: #{id}: #{f} must be a list, got #{t[f].inspect}"
|
|
132
|
+
exit 2
|
|
133
|
+
end
|
|
134
|
+
end
|
|
98
135
|
if (t["must_not_route_to"] || []).include?(t["expected_skill"])
|
|
99
136
|
warn "eval_bank_invalid_task: #{id}: expected_skill is also in must_not_route_to (impossible)"
|
|
100
137
|
exit 2
|
|
101
138
|
end
|
|
139
|
+
if (t["must_not_route_to"] || []).include?("none")
|
|
140
|
+
warn "eval_bank_invalid_task: #{id}: \"none\" is a sentinel outcome, not a routable target for must_not_route_to"
|
|
141
|
+
exit 2
|
|
142
|
+
end
|
|
143
|
+
# Optional acceptable[] names defensible alternate outcomes (a skill name or
|
|
144
|
+
# "none") for utterances with more than one correct route — e.g. a coverage-gap
|
|
145
|
+
# ask where both coordinator intake and rejection are right. It must not
|
|
146
|
+
# restate expected_skill or contradict must_not_route_to.
|
|
147
|
+
acc = t["acceptable"] || []
|
|
148
|
+
if acc.include?(t["expected_skill"])
|
|
149
|
+
warn "eval_bank_invalid_task: #{id}: acceptable restates expected_skill"
|
|
150
|
+
exit 2
|
|
151
|
+
end
|
|
152
|
+
unless (acc & (t["must_not_route_to"] || [])).empty?
|
|
153
|
+
warn "eval_bank_invalid_task: #{id}: acceptable and must_not_route_to overlap (contradictory)"
|
|
154
|
+
exit 2
|
|
155
|
+
end
|
|
156
|
+
# Anti-gaming: expected + acceptable must leave at least one outcome that
|
|
157
|
+
# would FAIL the row. A row covering the complete catalog-and-none outcome
|
|
158
|
+
# space passes on every valid grader selection — a vacuous fixture that fakes
|
|
159
|
+
# green regardless of description behavior.
|
|
160
|
+
if (all_outcomes - ([t["expected_skill"]] + acc)).empty?
|
|
161
|
+
warn "eval_bank_invalid_task: #{id}: expected_skill plus acceptable cover every possible outcome (vacuous row)"
|
|
162
|
+
exit 2
|
|
163
|
+
end
|
|
102
164
|
end
|
|
103
165
|
|
|
104
166
|
# --- anti-game ---------------------------------------------------------------
|
|
@@ -188,9 +250,12 @@ def build_prompt(catalog, utterance, bootstrap_layer = nil)
|
|
|
188
250
|
user utterance and the catalog of skills (name + description), pick the SINGLE
|
|
189
251
|
best skill that should handle it. Use only the catalog; route by the
|
|
190
252
|
descriptions' Use-when / Proactively / Skip rules.
|
|
253
|
+
If NO skill's description covers the utterance, answer "none" — do not
|
|
254
|
+
force-fit the nearest neighbor. Set "clarify" to true only when you would
|
|
255
|
+
need to ask the user a clarifying question before committing to a route.
|
|
191
256
|
|
|
192
257
|
Output ONLY a JSON object on one line, no other text:
|
|
193
|
-
{"selected_skill": "<exact skill name from the catalog>", "confidence": <0.0-1.0>, "rationale_short": "<one short clause>"}
|
|
258
|
+
{"selected_skill": "<exact skill name from the catalog, or none>", "clarify": <true|false>, "confidence": <0.0-1.0>, "rationale_short": "<one short clause>"}
|
|
194
259
|
|
|
195
260
|
#{bootstrap_layer ? "== ALWAYS-ON ENTRY ROUTING LAYER (injected into every session; takes precedence for entry routing) ==\n#{bootstrap_layer}\n" : ""}
|
|
196
261
|
== SKILL CATALOG ==
|
|
@@ -309,39 +374,77 @@ grader_available = true
|
|
|
309
374
|
tasks.each do |t|
|
|
310
375
|
exp = t["expected_skill"]
|
|
311
376
|
must_not = t["must_not_route_to"] || []
|
|
377
|
+
acceptable = t["acceptable"] || []
|
|
312
378
|
frozen_ref = t["frozen_at_sha"] == "root" ? `git -C #{Shellwords.escape(root)} rev-list --max-parents=0 HEAD`.lines.first.to_s.strip : t["frozen_at_sha"]
|
|
313
379
|
frozen_ok = ancestor?(root, frozen_ref)
|
|
314
380
|
prompt = build_prompt(catalog, t["utterance"], bootstrap_layer)
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
381
|
+
verdicts = []
|
|
382
|
+
replicas.times do |ri|
|
|
383
|
+
parsed, error = grade(model, timeout_s, prompt)
|
|
384
|
+
# A verdict that produced no observation contributes nothing to the totals,
|
|
385
|
+
# and an unmeasured task is indistinguishable from a routing failure in the
|
|
386
|
+
# numbers this bank reports. Both recoverable causes are sampling accidents
|
|
387
|
+
# rather than verdicts — a hard timeout is usually machine load, and
|
|
388
|
+
# unparseable output is usually one stray quote in a generated rationale —
|
|
389
|
+
# so each gets one retry.
|
|
390
|
+
# Repairing malformed output instead of re-asking for it was tried and removed:
|
|
391
|
+
# interpreting text that is by definition malformed has no natural boundary,
|
|
392
|
+
# and three review rounds each found a different shape that a repair would read
|
|
393
|
+
# as a verdict. Re-asking needs no such interpretation. Nothing else is retried:
|
|
394
|
+
# an auth failure and a missing CLI do reproduce on a second call.
|
|
395
|
+
if error&.start_with?("grader_timeout_") || error&.start_with?("no_json_in_output")
|
|
396
|
+
parsed, retry_error = grade(model, timeout_s, prompt)
|
|
397
|
+
error = parsed ? nil : retry_error
|
|
398
|
+
end
|
|
399
|
+
if error == "claude_not_found"
|
|
400
|
+
grader_available = false
|
|
401
|
+
break
|
|
402
|
+
end
|
|
403
|
+
selected = parsed && parsed["selected_skill"]
|
|
404
|
+
clarify = parsed && parsed["clarify"] == true
|
|
405
|
+
confidence = parsed && parsed["confidence"]
|
|
406
|
+
v_status =
|
|
407
|
+
if error then "ERROR"
|
|
408
|
+
elsif (selected == exp || acceptable.include?(selected)) && !must_not.include?(selected) then "PASS"
|
|
409
|
+
else "FAIL"
|
|
410
|
+
end
|
|
411
|
+
verdicts << { replica: ri + 1, selected: selected, clarify: clarify,
|
|
412
|
+
confidence: confidence, status: v_status, error: error }
|
|
333
413
|
end
|
|
334
|
-
|
|
335
|
-
|
|
414
|
+
break unless grader_available
|
|
415
|
+
# Task-level consensus over replicas (replicas=1 reproduces the old per-task
|
|
416
|
+
# semantics exactly). Conservative: one failing replica fails the task —
|
|
417
|
+
# a route that only sometimes lands is not a stable route.
|
|
418
|
+
observed = verdicts.reject { |v| v[:status] == "ERROR" }
|
|
336
419
|
status =
|
|
337
|
-
if
|
|
338
|
-
elsif
|
|
420
|
+
if observed.empty? then "ERROR"
|
|
421
|
+
elsif observed.all? { |v| v[:status] == "PASS" } then "PASS"
|
|
339
422
|
else "FAIL"
|
|
340
423
|
end
|
|
424
|
+
selections = observed.map { |v| v[:selected] }.uniq
|
|
425
|
+
# Failure-mode labels (vocabulary in references/eval-routing.md):
|
|
426
|
+
# absorbed — an utterance that should be rejected (expected "none")
|
|
427
|
+
# or kept away from named bait neighbors (must_not_route_to)
|
|
428
|
+
# was claimed by such a skill anyway
|
|
429
|
+
# ownership_split — replicas disagreed on the top pick (unstable ownership)
|
|
430
|
+
labels = []
|
|
431
|
+
absorbed = (exp == "none" && observed.any? { |v| v[:selected] && v[:selected] != "none" }) ||
|
|
432
|
+
observed.any? { |v| must_not.include?(v[:selected]) }
|
|
433
|
+
labels << "absorbed" if absorbed
|
|
434
|
+
labels << "ownership_split" if selections.size > 1
|
|
435
|
+
# A task where some replicas erred but others graded is PARTIALLY measured:
|
|
436
|
+
# the consensus above sees only the observed verdicts, so without this label
|
|
437
|
+
# a PASS+ERROR pair would read as a clean PASS and the run as fully sampled.
|
|
438
|
+
labels << "partial_error" if verdicts.any? { |v| v[:status] == "ERROR" } && !observed.empty?
|
|
341
439
|
results << {
|
|
342
|
-
id: t["id"], utterance: t["utterance"], expected: exp,
|
|
343
|
-
|
|
344
|
-
|
|
440
|
+
id: t["id"], utterance: t["utterance"], expected: exp, acceptable: acceptable,
|
|
441
|
+
selected: observed.first && observed.first[:selected],
|
|
442
|
+
confidence: observed.first && observed.first[:confidence],
|
|
443
|
+
clarify: observed.any? { |v| v[:clarify] },
|
|
444
|
+
acceptable_hit: observed.any? { |v| acceptable.include?(v[:selected]) },
|
|
445
|
+
must_not_route_to: must_not, status: status, labels: labels,
|
|
446
|
+
verdicts: verdicts, error: (verdicts.find { |v| v[:error] } || {})[:error],
|
|
447
|
+
frozen_at_sha_is_ancestor: frozen_ok
|
|
345
448
|
}
|
|
346
449
|
end
|
|
347
450
|
|
|
@@ -355,43 +458,114 @@ fails = results.select { |r| r[:status] == "FAIL" }
|
|
|
355
458
|
errors = results.select { |r| r[:status] == "ERROR" }
|
|
356
459
|
drift = results.reject { |r| r[:frozen_at_sha_is_ancestor] }
|
|
357
460
|
|
|
461
|
+
# First-class routing-quality metrics beyond pass/fail (counted over observed,
|
|
462
|
+
# non-ERROR verdicts): clarify rate, low-confidence rate, and — when replicas
|
|
463
|
+
# >= 2 — how often all replicas of one task picked the same top skill.
|
|
464
|
+
all_observed = results.flat_map { |r| r[:verdicts].reject { |v| v[:status] == "ERROR" } }
|
|
465
|
+
clarify_count = all_observed.count { |v| v[:clarify] }
|
|
466
|
+
low_conf_count = all_observed.count { |v| v[:confidence].is_a?(Numeric) && v[:confidence] < 0.5 }
|
|
467
|
+
# Replica-level error accounting: task-level `error` counts only fully
|
|
468
|
+
# unmeasured tasks, so a PASS+ERROR pair would otherwise report zero
|
|
469
|
+
# grader-errors while a replica silently went missing.
|
|
470
|
+
error_verdicts = results.sum { |r| r[:verdicts].count { |v| v[:status] == "ERROR" } }
|
|
471
|
+
partial_error_ids = results.select { |r| r[:labels].include?("partial_error") }.map { |r| r[:id] }
|
|
472
|
+
agreement_measured = 0
|
|
473
|
+
agreement_agree = 0
|
|
474
|
+
if replicas >= 2
|
|
475
|
+
results.each do |r|
|
|
476
|
+
obs = r[:verdicts].reject { |v| v[:status] == "ERROR" }
|
|
477
|
+
next if obs.size < 2
|
|
478
|
+
agreement_measured += 1
|
|
479
|
+
agreement_agree += 1 if obs.map { |v| v[:selected] }.uniq.size == 1
|
|
480
|
+
end
|
|
481
|
+
end
|
|
482
|
+
|
|
358
483
|
# baseline diff (newly failed / newly passed). The baseline is a prior report
|
|
359
|
-
# (a hash with a "results" array) or a bare results array.
|
|
484
|
+
# (a hash with a "results" array) or a bare results array. Reports from a
|
|
485
|
+
# different bank content or replica count are DIFFERENT RULERS (documented
|
|
486
|
+
# above): comparing them emits false regressions/improvements, so a
|
|
487
|
+
# demonstrated mismatch suppresses the diff instead of computing it. A bare
|
|
488
|
+
# results array carries no fingerprint — its comparability is unverifiable and
|
|
489
|
+
# is flagged as such rather than silently trusted.
|
|
360
490
|
newly_failed = []
|
|
361
491
|
newly_passed = []
|
|
492
|
+
baseline_comparable = nil
|
|
493
|
+
baseline_incomparable_reason = nil
|
|
362
494
|
if baseline_path && File.file?(baseline_path)
|
|
363
495
|
base_json = JSON.parse(File.read(baseline_path))
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
496
|
+
if base_json.is_a?(Hash)
|
|
497
|
+
base_bank_sha = base_json.dig("routing_surface", "bank_sha256")
|
|
498
|
+
base_replicas = base_json["replicas"] || 1 # legacy reports predate --replicas
|
|
499
|
+
if base_bank_sha && base_bank_sha != routing_surface[:bank_sha256]
|
|
500
|
+
baseline_comparable = false
|
|
501
|
+
baseline_incomparable_reason = "bank content differs (baseline #{base_bank_sha[0, 12]}… vs current #{routing_surface[:bank_sha256][0, 12]}…)"
|
|
502
|
+
elsif base_replicas != replicas
|
|
503
|
+
baseline_comparable = false
|
|
504
|
+
baseline_incomparable_reason = "replica count differs (baseline #{base_replicas} vs current #{replicas})"
|
|
505
|
+
else
|
|
506
|
+
baseline_comparable = base_bank_sha ? true : "unverified (baseline carries no bank fingerprint)"
|
|
507
|
+
end
|
|
508
|
+
else
|
|
509
|
+
baseline_comparable = "unverified (bare results array carries no fingerprint)"
|
|
510
|
+
end
|
|
511
|
+
unless baseline_comparable == false
|
|
512
|
+
base_results = base_json.is_a?(Hash) ? (base_json["results"] || []) : base_json
|
|
513
|
+
base_status = base_results.to_h { |r| [r["id"], r["status"]] }
|
|
514
|
+
results.each do |r|
|
|
515
|
+
was = base_status[r[:id]]
|
|
516
|
+
newly_failed << r[:id] if was == "PASS" && r[:status] == "FAIL"
|
|
517
|
+
newly_passed << r[:id] if was == "FAIL" && r[:status] == "PASS"
|
|
518
|
+
end
|
|
370
519
|
end
|
|
371
520
|
end
|
|
372
521
|
|
|
373
522
|
report = {
|
|
374
523
|
model: model, tasks: results.size, pass: passes, fail: fails.size, error: errors.size,
|
|
524
|
+
replicas: replicas, verdicts: all_observed.size,
|
|
525
|
+
error_verdicts: error_verdicts, partial_error_tasks: partial_error_ids,
|
|
526
|
+
clarify_count: clarify_count, low_confidence_count: low_conf_count,
|
|
527
|
+
replica_agreement: (replicas >= 2 ? { agree: agreement_agree, measured: agreement_measured } : nil),
|
|
375
528
|
desc_budget_chars: desc_budget, routing_surface: routing_surface,
|
|
376
529
|
co_change_bank_and_descriptions: co_change, co_change_check_available: co_change_check_ok,
|
|
377
530
|
frozen_drift: drift.map { |r| r[:id] },
|
|
531
|
+
baseline_comparable: baseline_comparable,
|
|
532
|
+
baseline_incomparable_reason: baseline_incomparable_reason,
|
|
378
533
|
newly_failed: newly_failed, newly_passed: newly_passed, results: results
|
|
379
534
|
}
|
|
380
535
|
File.write(json_path, JSON.pretty_generate(report)) if json_path
|
|
381
536
|
|
|
382
537
|
puts "eval-routing-bank (#{model}): #{passes}/#{results.size} pass, #{fails.size} fail, #{errors.size} grader-error"
|
|
383
538
|
puts " arm: desc-budget-chars=#{desc_budget}" if desc_budget
|
|
539
|
+
unless all_observed.empty?
|
|
540
|
+
line = " clarify: #{clarify_count}/#{all_observed.size} verdicts, low-confidence(<0.5): #{low_conf_count}/#{all_observed.size}"
|
|
541
|
+
line += ", replica top1 agreement: #{agreement_agree}/#{agreement_measured}" if replicas >= 2
|
|
542
|
+
puts line
|
|
543
|
+
end
|
|
544
|
+
if error_verdicts.positive?
|
|
545
|
+
puts " ⚠ grader-error verdicts: #{error_verdicts}#{partial_error_ids.empty? ? '' : " (partially measured tasks: #{partial_error_ids.join(', ')})"}"
|
|
546
|
+
end
|
|
384
547
|
puts " ⚠ co-change check unavailable: base ref #{base.inspect} not resolvable — could not verify bank/description co-change" unless co_change_check_ok
|
|
385
548
|
puts " ⚠ co-change: this change touches BOTH the bank and a SKILL.md description (verify the bank was not edited to pass)" if co_change
|
|
386
549
|
puts " ⚠ frozen-drift (not ancestor of HEAD, excluded from regression judgment): #{drift.map { |r| r[:id] }.join(', ')}" unless drift.empty?
|
|
387
550
|
unless fails.empty?
|
|
388
551
|
puts " FAIL:"
|
|
389
|
-
fails.each
|
|
552
|
+
fails.each do |r|
|
|
553
|
+
tag = r[:labels].empty? ? "" : " [#{r[:labels].join(',')}]"
|
|
554
|
+
obs = r[:verdicts].reject { |v| v[:status] == "ERROR" }
|
|
555
|
+
got = obs.map { |v| v[:selected].inspect }.uniq.join(" | ")
|
|
556
|
+
conf = obs.map { |v| v[:confidence] }.join(",")
|
|
557
|
+
puts " - #{r[:id]}: expected #{r[:expected]} got #{got} (conf #{conf})#{tag}"
|
|
558
|
+
end
|
|
390
559
|
end
|
|
391
560
|
unless errors.empty?
|
|
392
561
|
puts " GRADER-ERROR:"
|
|
393
562
|
errors.each { |r| puts " - #{r[:id]}: #{r[:error]}" }
|
|
394
563
|
end
|
|
564
|
+
if baseline_comparable == false
|
|
565
|
+
puts " ⚠ baseline not compared — different ruler: #{baseline_incomparable_reason}"
|
|
566
|
+
elsif baseline_comparable.is_a?(String)
|
|
567
|
+
puts " ⚠ baseline comparability #{baseline_comparable}"
|
|
568
|
+
end
|
|
395
569
|
unless newly_failed.empty? && newly_passed.empty?
|
|
396
570
|
puts " vs baseline: newly_failed=#{newly_failed.join(',')} newly_passed=#{newly_passed.join(',')}"
|
|
397
571
|
end
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
|
-
# Extraction-owned autonomous review wrapper: one review plus
|
|
2
|
+
# Extraction-owned autonomous review wrapper: one review plus one challenge.
|
|
3
3
|
set -euo pipefail
|
|
4
4
|
|
|
5
5
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
@@ -8,7 +8,7 @@ CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
|
|
|
8
8
|
for arg in "$@"; do
|
|
9
9
|
case "$arg" in
|
|
10
10
|
--challenge-b*)
|
|
11
|
-
echo "extraction_review_gate_error: challenge budget is fixed at
|
|
11
|
+
echo "extraction_review_gate_error: challenge budget is fixed at 1" >&2
|
|
12
12
|
exit 2
|
|
13
13
|
;;
|
|
14
14
|
esac
|
|
@@ -19,4 +19,4 @@ if [[ ! -x "$CONTROLLER" ]]; then
|
|
|
19
19
|
exit 2
|
|
20
20
|
fi
|
|
21
21
|
|
|
22
|
-
exec bash "$CONTROLLER" --challenge-budget
|
|
22
|
+
exec bash "$CONTROLLER" --challenge-budget 1 "$@"
|