@ccoalm/ccl-skills 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +8 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +29 -31
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +32 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
- package/dist/assets/release.json +79 -24
- package/package.json +1 -1
|
@@ -71,20 +71,52 @@ with open(bank, encoding="utf-8") as fh:
|
|
|
71
71
|
elif rid:
|
|
72
72
|
seen[rid] = lineno
|
|
73
73
|
|
|
74
|
+
# "none" is the runner's negative-control / coverage-gap sentinel
|
|
75
|
+
# (eval-routing-bank.rb): the expected outcome is that no catalog skill
|
|
76
|
+
# claims the utterance. It is an outcome, never a routable target.
|
|
74
77
|
expected = row.get("expected_skill")
|
|
75
|
-
if expected and expected not in installed:
|
|
78
|
+
if expected and expected != "none" and expected not in installed:
|
|
76
79
|
bad(f"line {lineno} ({rid}): expected_skill '{expected}' is not a skill in skills/")
|
|
77
80
|
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
+
# Type-check the ORIGINAL value: `x or []` would first convert an
|
|
82
|
+
# explicitly invalid "" (or 0/false) into a clean empty list, so the
|
|
83
|
+
# deterministic lane would pass a row the Ruby runner rejects.
|
|
84
|
+
must_not = row.get("must_not_route_to")
|
|
85
|
+
if must_not is None:
|
|
86
|
+
must_not = []
|
|
87
|
+
elif not isinstance(must_not, list):
|
|
88
|
+
bad(f"line {lineno} ({rid}): must_not_route_to must be a list, got {must_not!r}")
|
|
81
89
|
must_not = []
|
|
82
90
|
for target in must_not:
|
|
83
|
-
if target
|
|
91
|
+
if target == "none":
|
|
92
|
+
bad(f"line {lineno} ({rid}): must_not_route_to names the sentinel 'none' (an outcome, not a routable target)")
|
|
93
|
+
elif target not in installed:
|
|
84
94
|
bad(f"line {lineno} ({rid}): must_not_route_to names '{target}', not a skill in skills/")
|
|
85
95
|
if target == expected:
|
|
86
96
|
bad(f"line {lineno} ({rid}): '{target}' is both expected_skill and must_not_route_to")
|
|
87
97
|
|
|
98
|
+
acceptable = row.get("acceptable")
|
|
99
|
+
if acceptable is None:
|
|
100
|
+
acceptable = []
|
|
101
|
+
elif not isinstance(acceptable, list):
|
|
102
|
+
bad(f"line {lineno} ({rid}): acceptable must be a list, got {acceptable!r}")
|
|
103
|
+
acceptable = []
|
|
104
|
+
for target in acceptable:
|
|
105
|
+
if target != "none" and target not in installed:
|
|
106
|
+
bad(f"line {lineno} ({rid}): acceptable names '{target}', not a skill in skills/ (nor the 'none' sentinel)")
|
|
107
|
+
if target == expected:
|
|
108
|
+
bad(f"line {lineno} ({rid}): '{target}' is both expected_skill and acceptable (redundant)")
|
|
109
|
+
if target in must_not:
|
|
110
|
+
bad(f"line {lineno} ({rid}): '{target}' is both acceptable and must_not_route_to (contradictory)")
|
|
111
|
+
|
|
112
|
+
# Anti-gaming mirror of the runner's vacuous-row rule: expected +
|
|
113
|
+
# acceptable must leave at least one outcome that would fail the row,
|
|
114
|
+
# or every valid grader selection passes and the fixture fakes green.
|
|
115
|
+
if isinstance(acceptable, list) and expected:
|
|
116
|
+
outcome_space = set(installed) | {"none"}
|
|
117
|
+
if not outcome_space - ({expected} | set(acceptable)):
|
|
118
|
+
bad(f"line {lineno} ({rid}): expected_skill plus acceptable cover every possible outcome (vacuous row)")
|
|
119
|
+
|
|
88
120
|
# A truncated or emptied bank must fail rather than vacuously pass.
|
|
89
121
|
MIN_ROWS = 100
|
|
90
122
|
if rows < MIN_ROWS:
|
|
@@ -203,3 +235,52 @@ print(f"routing_bank_provenance_info: source present on {with_source}/{rows} row
|
|
|
203
235
|
f"(docs/f4-skill-effectiveness-harness.md lists it as required; not enforced here)")
|
|
204
236
|
print(f"routing_bank_integrity_ok rows={rows} (structure only — routing NOT evaluated)")
|
|
205
237
|
PY
|
|
238
|
+
main_rc=$?
|
|
239
|
+
[ "$main_rc" -eq 0 ] || exit "$main_rc"
|
|
240
|
+
|
|
241
|
+
# --- validator self-proof (oracle-can-fail) ----------------------------------
|
|
242
|
+
# A green main run proves nothing about the new field predicates unless the
|
|
243
|
+
# validator is also seen to RED on applied mutants. Run this same script once
|
|
244
|
+
# against a synthetic root carrying one mutant per predicate, and require both
|
|
245
|
+
# the non-zero exit and each predicate's own FAIL message. Guarded against
|
|
246
|
+
# recursion: the inner invocation skips this section.
|
|
247
|
+
if [ -z "${BANK_INTEGRITY_SELFPROOF:-}" ]; then
|
|
248
|
+
SP_ROOT="$(mktemp -d "${TMPDIR:-/tmp}/bank-integrity-selfproof.XXXXXX")"
|
|
249
|
+
trap 'rm -rf "$SP_ROOT"' EXIT
|
|
250
|
+
mkdir -p "$SP_ROOT/skills/testing-strategy" "$SP_ROOT/skills/tighten-doc" "$SP_ROOT/eval"
|
|
251
|
+
printf -- '---\ndescription: stub\n---\n' > "$SP_ROOT/skills/testing-strategy/SKILL.md"
|
|
252
|
+
printf -- '---\ndescription: stub\n---\n' > "$SP_ROOT/skills/tighten-doc/SKILL.md"
|
|
253
|
+
cat > "$SP_ROOT/eval/routing-tasks.jsonl" <<'MUTANTS'
|
|
254
|
+
{"id": "bad-mustnot-none", "utterance": "x", "expected_skill": "testing-strategy", "why_expected": "y", "frozen_at_sha": "root", "must_not_route_to": ["none"]}
|
|
255
|
+
{"id": "bad-acc-restate", "utterance": "x", "expected_skill": "testing-strategy", "why_expected": "y", "frozen_at_sha": "root", "acceptable": ["testing-strategy"]}
|
|
256
|
+
{"id": "bad-acc-unknown", "utterance": "x", "expected_skill": "testing-strategy", "why_expected": "y", "frozen_at_sha": "root", "acceptable": ["not-a-skill"]}
|
|
257
|
+
{"id": "bad-acc-contradict", "utterance": "x", "expected_skill": "testing-strategy", "why_expected": "y", "frozen_at_sha": "root", "acceptable": ["tighten-doc"], "must_not_route_to": ["tighten-doc"]}
|
|
258
|
+
{"id": "bad-acc-empty-string", "utterance": "x", "expected_skill": "testing-strategy", "why_expected": "y", "frozen_at_sha": "root", "acceptable": ""}
|
|
259
|
+
{"id": "bad-mustnot-empty-string", "utterance": "x", "expected_skill": "testing-strategy", "why_expected": "y", "frozen_at_sha": "root", "must_not_route_to": ""}
|
|
260
|
+
{"id": "bad-none-typo", "utterance": "x", "expected_skill": "not-a-skill", "why_expected": "y", "frozen_at_sha": "root"}
|
|
261
|
+
{"id": "bad-universal-pass", "utterance": "x", "expected_skill": "testing-strategy", "why_expected": "y", "frozen_at_sha": "root", "acceptable": ["tighten-doc", "none"]}
|
|
262
|
+
MUTANTS
|
|
263
|
+
sp_out="$(BANK_INTEGRITY_SELFPROOF=1 BANK_ROOT="$SP_ROOT" bash "$0" 2>&1)"
|
|
264
|
+
sp_rc=$?
|
|
265
|
+
if [ "$sp_rc" -eq 0 ]; then
|
|
266
|
+
echo "FAIL: validator self-proof — the mutant bank exited 0 (the oracle cannot fail)" >&2
|
|
267
|
+
exit 1
|
|
268
|
+
fi
|
|
269
|
+
for expect in \
|
|
270
|
+
"bad-mustnot-none): must_not_route_to names the sentinel 'none'" \
|
|
271
|
+
"bad-acc-restate): 'testing-strategy' is both expected_skill and acceptable" \
|
|
272
|
+
"bad-acc-unknown): acceptable names 'not-a-skill'" \
|
|
273
|
+
"bad-acc-contradict): 'tighten-doc' is both acceptable and must_not_route_to" \
|
|
274
|
+
"bad-acc-empty-string): acceptable must be a list, got ''" \
|
|
275
|
+
"bad-mustnot-empty-string): must_not_route_to must be a list, got ''" \
|
|
276
|
+
"bad-none-typo): expected_skill 'not-a-skill' is not a skill in skills/" \
|
|
277
|
+
"bad-universal-pass): expected_skill plus acceptable cover every possible outcome (vacuous row)"; do
|
|
278
|
+
case "$sp_out" in
|
|
279
|
+
*"$expect"*) : ;;
|
|
280
|
+
*) echo "FAIL: validator self-proof — expected mutant message not found: $expect" >&2
|
|
281
|
+
echo "$sp_out" >&2
|
|
282
|
+
exit 1 ;;
|
|
283
|
+
esac
|
|
284
|
+
done
|
|
285
|
+
echo "routing_bank_validator_selfproof_ok: 8 applied mutants red on their own assertions"
|
|
286
|
+
fi
|
|
@@ -140,7 +140,7 @@ def make_fixture(
|
|
|
140
140
|
receipt_mutators=None,
|
|
141
141
|
completion=True,
|
|
142
142
|
completion_mutator=None,
|
|
143
|
-
budget=
|
|
143
|
+
budget=1,
|
|
144
144
|
base_shas=None,
|
|
145
145
|
closeout=None,
|
|
146
146
|
unmatched=0,
|
|
@@ -168,9 +168,9 @@ def make_fixture(
|
|
|
168
168
|
for index, findings in enumerate(receipt_findings, start=1):
|
|
169
169
|
mode = "review" if index == 1 else "challenge"
|
|
170
170
|
status = "findings" if findings else "passed"
|
|
171
|
-
remaining =
|
|
171
|
+
remaining = 2 - index
|
|
172
172
|
if status == "findings":
|
|
173
|
-
state = "post_review_budget" if index ==
|
|
173
|
+
state = "post_review_budget" if index == 2 else "findings_pending"
|
|
174
174
|
else:
|
|
175
175
|
state = "reviewed"
|
|
176
176
|
receipt = {
|
|
@@ -181,7 +181,7 @@ def make_fixture(
|
|
|
181
181
|
"review_chain_tracked": True,
|
|
182
182
|
"review_chain_id": "extraction-chain",
|
|
183
183
|
"autonomous_review_index": index,
|
|
184
|
-
"autonomous_review_budget":
|
|
184
|
+
"autonomous_review_budget": 2,
|
|
185
185
|
"autonomous_reviews_remaining": remaining,
|
|
186
186
|
"autonomous_review_allowed": remaining > 0,
|
|
187
187
|
"challenge_index": 0 if index == 1 else index - 1,
|
|
@@ -221,7 +221,7 @@ def make_fixture(
|
|
|
221
221
|
"review_chain_tracked": True,
|
|
222
222
|
"review_chain_id": final["review_chain_id"],
|
|
223
223
|
"autonomous_review_index": final["autonomous_review_index"],
|
|
224
|
-
"autonomous_review_budget":
|
|
224
|
+
"autonomous_review_budget": 2,
|
|
225
225
|
"autonomous_reviews_remaining": final["autonomous_reviews_remaining"],
|
|
226
226
|
"autonomous_review_allowed": False,
|
|
227
227
|
"challenge_budget": budget,
|
|
@@ -411,9 +411,9 @@ run("float-schema", float_schema["ledger"], 1, "schema_version must be 3")
|
|
|
411
411
|
|
|
412
412
|
float_budget = make_fixture(
|
|
413
413
|
"float-budget",
|
|
414
|
-
receipt_mutators={2: lambda row: row.update(challenge_budget=
|
|
414
|
+
receipt_mutators={2: lambda row: row.update(challenge_budget=1.0)},
|
|
415
415
|
)
|
|
416
|
-
run("float-budget", float_budget["ledger"], 1, "challenge_budget must be
|
|
416
|
+
run("float-budget", float_budget["ledger"], 1, "challenge_budget must be 1")
|
|
417
417
|
|
|
418
418
|
nan_schema = make_fixture("nan-schema")
|
|
419
419
|
nan_schema["ledger"]["schema_version"] = float("nan")
|
|
@@ -493,11 +493,17 @@ scope_mismatch = make_fixture("scope-mismatch", receipt_mutators={2: drift_scope
|
|
|
493
493
|
run("scope-mismatch", scope_mismatch["ledger"], 1, "review scope changed")
|
|
494
494
|
|
|
495
495
|
budget_four = make_fixture("budget-four", budget=4)
|
|
496
|
-
run("budget-four", budget_four["ledger"], 1, "challenge_budget must be
|
|
496
|
+
run("budget-four", budget_four["ledger"], 1, "challenge_budget must be 1")
|
|
497
497
|
|
|
498
498
|
review_only = make_fixture("review-only", receipt_findings=[[]])
|
|
499
499
|
run("review-only", review_only["ledger"], 1, "ready requires at least one tracked challenge")
|
|
500
500
|
|
|
501
|
+
# A caller must not out-run the wrapper budget by supplying extra receipts: a
|
|
502
|
+
# third receipt under budget 1 claims a negative remaining count and would
|
|
503
|
+
# otherwise reach completion validation as a ready-state budget bypass.
|
|
504
|
+
over_budget = make_fixture("over-budget", receipt_findings=[[], [], []])
|
|
505
|
+
run("over-budget", over_budget["ledger"], 1, "exceeds the wrapper budget")
|
|
506
|
+
|
|
501
507
|
missing_complete = make_fixture("missing-complete")
|
|
502
508
|
missing_complete["ledger"]["completion_receipt"] = None
|
|
503
509
|
run("missing-complete", missing_complete["ledger"], 1, "requires a completion receipt")
|
|
@@ -537,7 +543,7 @@ run(
|
|
|
537
543
|
|
|
538
544
|
continuation = make_fixture(
|
|
539
545
|
"continuation",
|
|
540
|
-
receipt_findings=[[finding(1)], [
|
|
546
|
+
receipt_findings=[[finding(1)], [finding(2)]],
|
|
541
547
|
completion=False,
|
|
542
548
|
open_last=True,
|
|
543
549
|
)
|
|
@@ -550,10 +556,10 @@ run(
|
|
|
550
556
|
|
|
551
557
|
continuation_unconsumed_first_drift = make_fixture(
|
|
552
558
|
"continuation-unconsumed-first-drift",
|
|
553
|
-
receipt_findings=[[finding(1)], [
|
|
559
|
+
receipt_findings=[[finding(1)], [finding(2)]],
|
|
554
560
|
completion=False,
|
|
555
561
|
open_last=True,
|
|
556
|
-
base_shas=[A, A,
|
|
562
|
+
base_shas=[A, A, B],
|
|
557
563
|
)
|
|
558
564
|
run(
|
|
559
565
|
"continuation-unconsumed-first-drift",
|
|
@@ -564,7 +570,7 @@ run(
|
|
|
564
570
|
|
|
565
571
|
continuation_candidate = make_fixture(
|
|
566
572
|
"continuation-candidate",
|
|
567
|
-
receipt_findings=[[finding(1)], [
|
|
573
|
+
receipt_findings=[[finding(1)], [finding(2)]],
|
|
568
574
|
completion=False,
|
|
569
575
|
open_last=True,
|
|
570
576
|
)
|
|
@@ -595,8 +601,8 @@ run(
|
|
|
595
601
|
|
|
596
602
|
unknown_state = make_fixture(
|
|
597
603
|
"unknown-state",
|
|
598
|
-
receipt_findings=[[finding(1)], [
|
|
599
|
-
receipt_mutators={
|
|
604
|
+
receipt_findings=[[finding(1)], [finding(2)]],
|
|
605
|
+
receipt_mutators={2: lambda row: row.update(review_state="potato")},
|
|
600
606
|
completion=False,
|
|
601
607
|
open_last=True,
|
|
602
608
|
)
|
|
@@ -605,10 +611,10 @@ run("unknown-state", unknown_state["ledger"], 1, "unknown controller review_stat
|
|
|
605
611
|
|
|
606
612
|
passed_continuation = make_fixture(
|
|
607
613
|
"passed-continuation",
|
|
608
|
-
receipt_findings=[[finding(1)], []
|
|
614
|
+
receipt_findings=[[finding(1)], []],
|
|
609
615
|
completion=False,
|
|
610
616
|
)
|
|
611
|
-
run("passed-continuation", passed_continuation["ledger"], 1, "
|
|
617
|
+
run("passed-continuation", passed_continuation["ledger"], 1, "findings in post_review_budget")
|
|
612
618
|
|
|
613
619
|
bad_finding_hash = make_fixture("bad-finding-hash")
|
|
614
620
|
bad_finding_hash["ledger"]["finding_classes"][0]["occurrences"][0][
|
|
@@ -744,7 +750,7 @@ run(
|
|
|
744
750
|
|
|
745
751
|
historical_open = make_fixture(
|
|
746
752
|
"historical-open",
|
|
747
|
-
receipt_findings=[[finding(1)
|
|
753
|
+
receipt_findings=[[finding(1), finding(2)], []],
|
|
748
754
|
)
|
|
749
755
|
historical_open["ledger"]["finding_classes"][0]["occurrences"][0][
|
|
750
756
|
"disposition"
|
|
@@ -756,7 +762,7 @@ run("historical-open", historical_open["ledger"], 1, "unresolved finding occurre
|
|
|
756
762
|
|
|
757
763
|
historical_open_linked = make_fixture(
|
|
758
764
|
"historical-open-linked",
|
|
759
|
-
receipt_findings=[[finding(1)
|
|
765
|
+
receipt_findings=[[finding(1), finding(2)], []],
|
|
760
766
|
)
|
|
761
767
|
historical_open_linked_occurrences = historical_open_linked["ledger"][
|
|
762
768
|
"finding_classes"
|
|
@@ -825,7 +831,7 @@ run("needs-human", needs_human["ledger"], 1, "unresolved finding occurrence")
|
|
|
825
831
|
|
|
826
832
|
third_occurrence = make_fixture(
|
|
827
833
|
"third-occurrence",
|
|
828
|
-
receipt_findings=[[finding(1)
|
|
834
|
+
receipt_findings=[[finding(1), finding(2)], [finding(3)]],
|
|
829
835
|
completion=False,
|
|
830
836
|
unmatched=1,
|
|
831
837
|
open_last=True,
|
|
@@ -1045,7 +1051,7 @@ for closing_disposition in sorted(CLOSED_DISPOSITIONS):
|
|
|
1045
1051
|
case_name = f"historical-needs-human-{closing_disposition.replace('_', '-')}"
|
|
1046
1052
|
historical_needs_human_linked = make_fixture(
|
|
1047
1053
|
case_name,
|
|
1048
|
-
receipt_findings=[[finding(1)
|
|
1054
|
+
receipt_findings=[[finding(1), finding(2)], []],
|
|
1049
1055
|
)
|
|
1050
1056
|
historical_needs_human_occurrences = historical_needs_human_linked["ledger"][
|
|
1051
1057
|
"finding_classes"
|
|
@@ -1128,7 +1134,7 @@ duplicate_completion_result = invoke(
|
|
|
1128
1134
|
|
|
1129
1135
|
duplicate_sweep = make_fixture(
|
|
1130
1136
|
"duplicate-sweep",
|
|
1131
|
-
receipt_findings=[[finding(1)
|
|
1137
|
+
receipt_findings=[[finding(1), finding(2)], [finding(3)]],
|
|
1132
1138
|
completion=False,
|
|
1133
1139
|
open_last=True,
|
|
1134
1140
|
)
|
|
@@ -30,6 +30,11 @@ TERMINAL_STATES = {
|
|
|
30
30
|
}
|
|
31
31
|
EXTERNAL_REVIEW_STATES = {"reviewed", "findings_pending", "post_review_budget"}
|
|
32
32
|
KNOWN_REVIEW_STATES = EXTERNAL_REVIEW_STATES | {"self_reviewed"}
|
|
33
|
+
# The extraction wrapper fixes the autonomous lane at one review plus one
|
|
34
|
+
# challenge; every numeric bound below derives from these two constants so a
|
|
35
|
+
# future budget change lands in exactly one place.
|
|
36
|
+
WRAPPER_CHALLENGE_BUDGET = 1
|
|
37
|
+
WRAPPER_AUTONOMOUS_ROUNDS = WRAPPER_CHALLENGE_BUDGET + 1
|
|
33
38
|
DISPOSITIONS = {
|
|
34
39
|
"fixed",
|
|
35
40
|
"source_refuted",
|
|
@@ -272,8 +277,8 @@ def validate_scope(receipt: dict[str, Any], label: str) -> str:
|
|
|
272
277
|
]
|
|
273
278
|
if len(normalized_risks) != len(set(normalized_risks)):
|
|
274
279
|
fail(f"{label}.review_scope.risk_tags contains duplicates")
|
|
275
|
-
if scope["challenge_budget"] !=
|
|
276
|
-
fail(f"{label}.review_scope.challenge_budget must be
|
|
280
|
+
if scope["challenge_budget"] != WRAPPER_CHALLENGE_BUDGET or type(scope["challenge_budget"]) is not int:
|
|
281
|
+
fail(f"{label}.review_scope.challenge_budget must be {WRAPPER_CHALLENGE_BUDGET}")
|
|
277
282
|
if (
|
|
278
283
|
scope["wording_only_proof_sha256"] is not None
|
|
279
284
|
or scope["wording_only_scope_sha256"] is not None
|
|
@@ -313,6 +318,11 @@ def validate_controller_receipts(
|
|
|
313
318
|
payload["candidate_sha256"], "ledger.candidate_sha256"
|
|
314
319
|
)
|
|
315
320
|
for expected_index, value in enumerate(refs, start=1):
|
|
321
|
+
# A caller supplying more receipts than the wrapper can mint would
|
|
322
|
+
# otherwise claim a negative remaining count and reach completion
|
|
323
|
+
# validation as a ready-state budget bypass.
|
|
324
|
+
if expected_index > WRAPPER_AUTONOMOUS_ROUNDS:
|
|
325
|
+
fail("controller chain exceeds the wrapper budget")
|
|
316
326
|
ref = exact_object(
|
|
317
327
|
value, {"sequence", "file", "sha256"}, f"controller_receipts[{expected_index - 1}]"
|
|
318
328
|
)
|
|
@@ -352,13 +362,13 @@ def validate_controller_receipts(
|
|
|
352
362
|
receipt.get("challenge_index")
|
|
353
363
|
) is not int:
|
|
354
364
|
fail(f"controller receipt {expected_index} challenge_index is invalid")
|
|
355
|
-
if receipt.get("challenge_budget") !=
|
|
356
|
-
fail(f"controller receipt {expected_index} challenge_budget must be
|
|
357
|
-
if receipt.get("autonomous_review_budget") !=
|
|
365
|
+
if receipt.get("challenge_budget") != WRAPPER_CHALLENGE_BUDGET or type(receipt.get("challenge_budget")) is not int:
|
|
366
|
+
fail(f"controller receipt {expected_index} challenge_budget must be {WRAPPER_CHALLENGE_BUDGET}")
|
|
367
|
+
if receipt.get("autonomous_review_budget") != WRAPPER_AUTONOMOUS_ROUNDS or type(
|
|
358
368
|
receipt.get("autonomous_review_budget")
|
|
359
369
|
) is not int:
|
|
360
|
-
fail(f"controller receipt {expected_index} autonomous_review_budget must be
|
|
361
|
-
expected_remaining =
|
|
370
|
+
fail(f"controller receipt {expected_index} autonomous_review_budget must be {WRAPPER_AUTONOMOUS_ROUNDS}")
|
|
371
|
+
expected_remaining = WRAPPER_AUTONOMOUS_ROUNDS - expected_index
|
|
362
372
|
if receipt.get("autonomous_reviews_remaining") != expected_remaining or type(
|
|
363
373
|
receipt.get("autonomous_reviews_remaining")
|
|
364
374
|
) is not int:
|
|
@@ -415,7 +425,7 @@ def validate_controller_receipts(
|
|
|
415
425
|
fail(f"controller receipt {expected_index} has unknown controller review_state {state}")
|
|
416
426
|
expected_state = (
|
|
417
427
|
"post_review_budget"
|
|
418
|
-
if receipt["status"] == "findings" and expected_index ==
|
|
428
|
+
if receipt["status"] == "findings" and expected_index == WRAPPER_AUTONOMOUS_ROUNDS
|
|
419
429
|
else "findings_pending"
|
|
420
430
|
if receipt["status"] == "findings"
|
|
421
431
|
else "reviewed"
|
|
@@ -469,17 +479,17 @@ def validate_completion_receipt(
|
|
|
469
479
|
fail("completion receipt cannot close a final external receipt with findings")
|
|
470
480
|
if receipt.get("review_chain_tracked") is not True or receipt.get("review_chain_id") != chain_id:
|
|
471
481
|
fail("completion receipt review_chain_id does not match the controller chain")
|
|
472
|
-
if receipt.get("challenge_budget") !=
|
|
473
|
-
fail("completion receipt challenge_budget must be
|
|
474
|
-
if receipt.get("autonomous_review_budget") !=
|
|
482
|
+
if receipt.get("challenge_budget") != WRAPPER_CHALLENGE_BUDGET or type(receipt.get("challenge_budget")) is not int:
|
|
483
|
+
fail(f"completion receipt challenge_budget must be {WRAPPER_CHALLENGE_BUDGET}")
|
|
484
|
+
if receipt.get("autonomous_review_budget") != WRAPPER_AUTONOMOUS_ROUNDS or type(
|
|
475
485
|
receipt.get("autonomous_review_budget")
|
|
476
486
|
) is not int:
|
|
477
|
-
fail("completion receipt autonomous_review_budget must be
|
|
487
|
+
fail(f"completion receipt autonomous_review_budget must be {WRAPPER_AUTONOMOUS_ROUNDS}")
|
|
478
488
|
if receipt.get("autonomous_review_index") != len(receipts) or type(
|
|
479
489
|
receipt.get("autonomous_review_index")
|
|
480
490
|
) is not int:
|
|
481
491
|
fail("completion receipt autonomous_review_index does not match the final round")
|
|
482
|
-
expected_remaining =
|
|
492
|
+
expected_remaining = WRAPPER_AUTONOMOUS_ROUNDS - len(receipts)
|
|
483
493
|
if receipt.get("autonomous_reviews_remaining") != expected_remaining or type(
|
|
484
494
|
receipt.get("autonomous_reviews_remaining")
|
|
485
495
|
) is not int:
|
|
@@ -941,12 +951,12 @@ def validate(payload: dict[str, Any], ledger_dir: Path) -> tuple[str, int, int]:
|
|
|
941
951
|
elif closeout == "continuation_authorization_required":
|
|
942
952
|
final = receipts[-1]
|
|
943
953
|
if (
|
|
944
|
-
len(receipts) !=
|
|
954
|
+
len(receipts) != WRAPPER_AUTONOMOUS_ROUNDS
|
|
945
955
|
or final.get("status") != "findings"
|
|
946
956
|
or final.get("review_state") != "post_review_budget"
|
|
947
957
|
or final.get("human_decision_required") is not True
|
|
948
958
|
):
|
|
949
|
-
fail("continuation_authorization_required requires round
|
|
959
|
+
fail(f"continuation_authorization_required requires final-round (round {WRAPPER_AUTONOMOUS_ROUNDS}) findings in post_review_budget")
|
|
950
960
|
elif closeout == "baseline_race" and not delta:
|
|
951
961
|
fail("baseline_race requires a non-empty unreviewed_delta")
|
|
952
962
|
|