@ccoalm/ccl-skills 0.17.0 → 0.18.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +8 -7
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-review-covers-head.sh +128 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-untracked-background.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_review_covers_head.sh +104 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_untracked_background.sh +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +9 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +13 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +401 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +8 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +230 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +6 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +48 -119
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +9 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check_review_evidence_present.py +122 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +0 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +37 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +102 -20
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +16 -27
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +3 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_review_evidence_present.sh +81 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +130 -310
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +49 -3
- package/dist/assets/release.json +55 -45
- package/package.json +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +0 -1242
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +0 -978
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +0 -1477
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +0 -1183
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Refuse a pull request that changes shared skill behavior with no recorded review.
|
|
3
|
+
|
|
4
|
+
The extraction review lane owes one review and one challenge per non-wording
|
|
5
|
+
change (references/dual-track-review-gate.md). The merge-side candidate binding
|
|
6
|
+
that used to sit here was retired: it tied every pass to one tree hash, so every
|
|
7
|
+
fix voided the passes and restarted the sequence. What it also did — refuse a
|
|
8
|
+
landing with no review at all — is the half worth keeping, and it needs no hash.
|
|
9
|
+
|
|
10
|
+
This gate asks one question: does a pull request that changes `skills/` or
|
|
11
|
+
`hooks/` carry at least one conclusive review result added or modified under a
|
|
12
|
+
specs/<round>/evidence/ directory?
|
|
13
|
+
|
|
14
|
+
It deliberately does not require a challenge result. An earlier version did,
|
|
15
|
+
with a waiver for wording-only changes; independent review broke that waiver in
|
|
16
|
+
four successive forms (a global waiver, a second skill, a later edit to the same
|
|
17
|
+
file, a rewritten frontmatter delimiter), because deciding "still wording-only"
|
|
18
|
+
means re-parsing diffs this gate does not own. Same-class recurrence is the cue
|
|
19
|
+
to delete the capability, so the challenge obligation stays in the review-lane
|
|
20
|
+
rule and this gate only catches a round that recorded no external pass at all.
|
|
21
|
+
|
|
22
|
+
A result is conclusive when it is a schema-3 controller envelope whose status is
|
|
23
|
+
`passed` or `findings` and which names the client that ran it. It does not check
|
|
24
|
+
which candidate a result reviewed, and it cannot tell a genuine result from a
|
|
25
|
+
hand-written one: like the repository's other author-declared gates it catches a
|
|
26
|
+
pass that was never recorded, not a forged one.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import argparse
|
|
32
|
+
import json
|
|
33
|
+
import subprocess
|
|
34
|
+
import sys
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
|
|
37
|
+
SUBJECT_PREFIXES = ("skills/", "hooks/")
|
|
38
|
+
CONCLUSIVE = ("passed", "findings")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def changed_paths(root: Path, base: str) -> list[str]:
|
|
42
|
+
result = subprocess.run(
|
|
43
|
+
# No --diff-filter: every change class counts, including a type change
|
|
44
|
+
# (a regular file replaced by a symlink), which a filter list omits.
|
|
45
|
+
["git", "-C", str(root), "diff", "--name-only", "--no-renames", base, "HEAD"],
|
|
46
|
+
capture_output=True, text=True,
|
|
47
|
+
)
|
|
48
|
+
if result.returncode != 0:
|
|
49
|
+
raise RuntimeError(result.stderr.strip() or "git diff failed")
|
|
50
|
+
return [line for line in result.stdout.splitlines() if line]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def is_evidence_json(path: str) -> bool:
|
|
54
|
+
parts = path.split("/")
|
|
55
|
+
return (
|
|
56
|
+
len(parts) >= 4
|
|
57
|
+
and parts[0] == "specs"
|
|
58
|
+
and "evidence" in parts[2:-1]
|
|
59
|
+
and parts[-1].endswith(".json")
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def load_result(root: Path, path: str) -> dict | None:
|
|
64
|
+
target = root / path
|
|
65
|
+
if not target.is_file():
|
|
66
|
+
return None
|
|
67
|
+
try:
|
|
68
|
+
value = json.loads(target.read_text(encoding="utf-8"))
|
|
69
|
+
except (OSError, UnicodeDecodeError, json.JSONDecodeError):
|
|
70
|
+
return None
|
|
71
|
+
if not isinstance(value, dict) or value.get("schema_version") != 3:
|
|
72
|
+
return None
|
|
73
|
+
if value.get("mode") not in ("review", "challenge"):
|
|
74
|
+
return None
|
|
75
|
+
if value.get("status") not in CONCLUSIVE or not value.get("selected_client"):
|
|
76
|
+
return None
|
|
77
|
+
return value
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def main(argv: list[str] | None = None) -> int:
|
|
81
|
+
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
82
|
+
parser.add_argument("--repo-root", default=".")
|
|
83
|
+
parser.add_argument("--base", required=True)
|
|
84
|
+
args = parser.parse_args(argv)
|
|
85
|
+
root = Path(args.repo_root).resolve()
|
|
86
|
+
|
|
87
|
+
try:
|
|
88
|
+
paths = changed_paths(root, args.base)
|
|
89
|
+
except RuntimeError as exc:
|
|
90
|
+
print(f"review_evidence_unevaluated: {exc}")
|
|
91
|
+
return 2
|
|
92
|
+
|
|
93
|
+
subjects = [p for p in paths if p.startswith(SUBJECT_PREFIXES)]
|
|
94
|
+
if not subjects:
|
|
95
|
+
print("review_evidence_not_required: no change under skills/ or hooks/")
|
|
96
|
+
return 0
|
|
97
|
+
|
|
98
|
+
reviews: list[str] = []
|
|
99
|
+
challenges: list[str] = []
|
|
100
|
+
for path in paths:
|
|
101
|
+
if not is_evidence_json(path):
|
|
102
|
+
continue
|
|
103
|
+
result = load_result(root, path)
|
|
104
|
+
if result is None:
|
|
105
|
+
continue
|
|
106
|
+
(reviews if result["mode"] == "review" else challenges).append(path)
|
|
107
|
+
|
|
108
|
+
if not reviews:
|
|
109
|
+
print(
|
|
110
|
+
"review_evidence_missing: this pull request changes "
|
|
111
|
+
f"{len(subjects)} path(s) under skills/ or hooks/ but carries no conclusive "
|
|
112
|
+
"review result under specs/<round>/evidence/"
|
|
113
|
+
)
|
|
114
|
+
print(" fix: commit the controller result JSON of each owed pass "
|
|
115
|
+
"(references/dual-track-review-gate.md, Recording the passes)")
|
|
116
|
+
return 1
|
|
117
|
+
print(f"review_evidence_present_ok: {len(reviews)} review, {len(challenges)} challenge")
|
|
118
|
+
return 0
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
if __name__ == "__main__":
|
|
122
|
+
sys.exit(main())
|
|
@@ -13,7 +13,6 @@ burn-rate-page-row-reference skills/platform-observability/references/sli-slo-de
|
|
|
13
13
|
coverage-tier-provenance skills/testing-strategy/references/test-code-authoring-patterns.md 60% acceptable / 75% commendable / 90% exemplary 分档 071-chainC-r1f4: externally verified coverage tiers (specs/071 source-verification)
|
|
14
14
|
pairwise-trigger-range skills/test-artifact-management/references/classical-test-design-techniques.md 2-way 累计触发 53–97% 071-chainC-r1f4: NIST SP 800-142 empirical range, externally verified (specs/071 source-verification)
|
|
15
15
|
bva-two-vs-three-value skills/test-artifact-management/references/classical-test-design-techniques.md 2-value(边界 + 下一格)和 3-value(边界 + 两侧) 071-chainC-r1f4: ISTQB v4 BVA variant definitions, externally verified (specs/071 source-verification)
|
|
16
|
-
merge-side-ledger-binding-failclosed skills/skill-extraction-workflow/scripts/review_ledger_binding.py return 0 if args.allow_unevaluated else 2 084-r4f1: the no-base fail-closed branch; flipping it to an unconditional 0 restores a gate that passes having checked nothing
|
|
17
16
|
lane-budget-default-range skills/code-review/references/timeout-auth-and-capabilities.md defaults to 2400 seconds and accepts 5 to 3600 125-r1f1: the entrypoint now points here for the cumulative lane budget instead of restating it; dropping or drifting the default/range would silently strip the bound from both surfaces
|
|
18
17
|
lane-budget-mode-minimums skills/code-review/references/timeout-auth-and-capabilities.md 21 total seconds for review and 16 125-r1f1: the per-mode fail-closed minimums the entrypoint delegates here; without the pin the fail-closed limit can vanish with no suite failing
|
|
19
18
|
lane-budget-reserved-seconds skills/code-review/references/timeout-auth-and-capabilities.md while reserving ten controller 125-r1f1: the reserved controller seconds in the per-invocation division the entrypoint delegates here
|
|
@@ -1,22 +1,51 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
|
-
# Extraction-owned
|
|
2
|
+
# Extraction-owned review wrapper: every call is one single-shot pass.
|
|
3
|
+
#
|
|
4
|
+
# A non-wording extraction owes one review and one challenge, plus a delta pass
|
|
5
|
+
# per fixed P0/P1 (references/dual-track-review-gate.md). None of them is bound
|
|
6
|
+
# to another pass or to one candidate hash, so this wrapper fixes the controller
|
|
7
|
+
# options that make a pass single-shot and refuses the options that would open a
|
|
8
|
+
# tracked review chain. The generic controller keeps its chain mode for other
|
|
9
|
+
# callers; this lane does not use it.
|
|
3
10
|
set -euo pipefail
|
|
4
11
|
|
|
5
12
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
6
13
|
CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
|
|
7
14
|
|
|
15
|
+
fail() {
|
|
16
|
+
echo "extraction_review_gate_error: $*" >&2
|
|
17
|
+
exit 2
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
mode=""
|
|
21
|
+
expect_mode=0
|
|
8
22
|
for arg in "$@"; do
|
|
23
|
+
if [[ "$expect_mode" == 1 ]]; then
|
|
24
|
+
mode="$arg"
|
|
25
|
+
expect_mode=0
|
|
26
|
+
continue
|
|
27
|
+
fi
|
|
9
28
|
case "$arg" in
|
|
10
|
-
--
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
29
|
+
--mode) expect_mode=1 ;;
|
|
30
|
+
--mode=*) mode="${arg#--mode=}" ;;
|
|
31
|
+
# Prefixes cover the controller's unambiguous abbreviations as well as the
|
|
32
|
+
# full spellings, so a shortened flag cannot reopen a chain.
|
|
33
|
+
--challenge-b*|--challenge-i*)
|
|
34
|
+
fail "the extraction lane fixes the challenge budget and index; do not pass $arg" ;;
|
|
35
|
+
--review-c*|--au*|--prio*|--pre*|--com*)
|
|
36
|
+
fail "the extraction lane is single-shot; review-chain option $arg is not accepted" ;;
|
|
14
37
|
esac
|
|
15
38
|
done
|
|
16
39
|
|
|
40
|
+
case "$mode" in
|
|
41
|
+
review) fixed=(--challenge-budget 0) ;;
|
|
42
|
+
challenge) fixed=(--challenge-budget 1 --challenge-index 1) ;;
|
|
43
|
+
"") fail "pass --mode review or --mode challenge" ;;
|
|
44
|
+
*) fail "the extraction lane runs only --mode review or --mode challenge, not $mode" ;;
|
|
45
|
+
esac
|
|
46
|
+
|
|
17
47
|
if [[ ! -x "$CONTROLLER" ]]; then
|
|
18
|
-
|
|
19
|
-
exit 2
|
|
48
|
+
fail "code-review controller is unavailable"
|
|
20
49
|
fi
|
|
21
50
|
|
|
22
|
-
exec bash "$CONTROLLER"
|
|
51
|
+
exec bash "$CONTROLLER" "${fixed[@]}" "$@"
|
|
@@ -83,16 +83,31 @@ unless File.file?(register_path)
|
|
|
83
83
|
exit 1
|
|
84
84
|
end
|
|
85
85
|
|
|
86
|
-
# Reviewed waivers cover
|
|
86
|
+
# Reviewed waivers cover the immutable historical rows whose locator was retired
|
|
87
87
|
# by an explicit superseding round. The digest table below binds each waiver to
|
|
88
|
-
#
|
|
88
|
+
# those exact rows (one digest, or an array when several rows cited the retired
|
|
89
|
+
# locator); a new row cannot inherit it by reusing the locator.
|
|
89
90
|
EXEMPT = {
|
|
90
91
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#Disabled semantics are real, not painted" =>
|
|
91
92
|
"065 replaced the combined platform walkthrough with an authority-classed claim ledger and executable delivery contract",
|
|
92
93
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#predictive-back geometry routes to" =>
|
|
93
94
|
"065 moved platform mechanics to the canonical client-owner return while keeping platform guidance scoped",
|
|
94
95
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#Interaction-state matrix is complete" =>
|
|
95
|
-
"065 replaced walkthrough-level proof with criterion IDs, test-layer selection, runtime evidence, and a candidate-bound verdict"
|
|
96
|
+
"065 replaced walkthrough-level proof with criterion IDs, test-layer selection, runtime evidence, and a candidate-bound verdict",
|
|
97
|
+
"command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh" =>
|
|
98
|
+
"127 retired the closeout ledger and its validator; the extraction lane records single-shot passes instead",
|
|
99
|
+
"command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh" =>
|
|
100
|
+
"127 retired the merge-side candidate binding; the post-review delta and the human merge replace it",
|
|
101
|
+
"command:skills/skill-extraction-workflow/scripts/review_ledger_binding.py" =>
|
|
102
|
+
"127 retired the merge-side candidate binding; the post-review delta and the human merge replace it",
|
|
103
|
+
"file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#Sum spent rounds across all chains before opening one more" =>
|
|
104
|
+
"127 replaced the round budget with one review, one challenge and bounded delta passes",
|
|
105
|
+
"file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen" =>
|
|
106
|
+
"127 lets the challenge run after the review's fixes; nothing is bound to one candidate",
|
|
107
|
+
"file:skills/skill-extraction-workflow/references/extraction-quickstart.md#still owes the two-round chain before it can land" =>
|
|
108
|
+
"127 retired the merge-side binding that imposed the chain on wording-only changes",
|
|
109
|
+
"file:skills/skill-extraction-workflow/references/extraction-quickstart.md#must exclude every evidence JSON the round has already added" =>
|
|
110
|
+
"127 retired the merge-side binding whose evidence exclusion this rule mirrored"
|
|
96
111
|
}.freeze
|
|
97
112
|
|
|
98
113
|
# `\p{Word}` rather than `[a-z0-9]`: GitHub keeps non-ASCII characters in a slug,
|
|
@@ -471,7 +486,45 @@ EXEMPT_ROW_DIGESTS = {
|
|
|
471
486
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#predictive-back geometry routes to" =>
|
|
472
487
|
"6847e68f062c2fe65a32f34bc743a6162a4e245a3a5d858c020faa841868f929",
|
|
473
488
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#Interaction-state matrix is complete" =>
|
|
474
|
-
"9671789a4bac36787dc26965ca03c98c7ab1c10de3312c6792587cb34e7c548e"
|
|
489
|
+
"9671789a4bac36787dc26965ca03c98c7ab1c10de3312c6792587cb34e7c548e",
|
|
490
|
+
"command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh" => [
|
|
491
|
+
"da20ca2ada11f70a55bbfbffde84a80d0a8a8a46c9e4010a96a9614ca854fd34",
|
|
492
|
+
"b8eede70a62910f0a7f2d11cfd854537c3450889bc294cfc933b37f2a13c55b8",
|
|
493
|
+
"1c575fdad17f3fca00569832d487dcf0ac0f28fe5f8fb4bfa21e784d9111a243",
|
|
494
|
+
"b40a21a554cbc8bb3484c345160403a69f1cfb0b41702559fc2442572f52afce",
|
|
495
|
+
"0579ef1411fc866c0e612f9475634f073cfc0e0be429482ee9e8b4fd7abafe10"
|
|
496
|
+
],
|
|
497
|
+
"command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh" => [
|
|
498
|
+
"0579ef1411fc866c0e612f9475634f073cfc0e0be429482ee9e8b4fd7abafe10",
|
|
499
|
+
"fdb08300002289dc5b3d0bcd279589ec8eca7dd5d3b14f0b7b079d34372b41f7",
|
|
500
|
+
"4a593029ba36b128519adeb8a3b9cbee32053caec7953c884ad587efaf88a3aa",
|
|
501
|
+
"f56d544525a45d0b26e8752eec4f2ff516239c3ca866df600506ea3ccb39f4b2",
|
|
502
|
+
"6f3b518a04ea8c20adc1ec3cf377a7eda2ef775903354cf54bed59268a8332b4",
|
|
503
|
+
"dfa47eac33add5ce0a7a73ce237b8bf4a1589bd325eb34b4531e25f9adbb574d",
|
|
504
|
+
"c6e6f74ca2407db2d98124e0517e6d70fae6f9df669f190bb7336ab76f13eb56",
|
|
505
|
+
"16f862e9df7fea0ea817e392e78e561690b0120f97125c2db5f7c7b987154385",
|
|
506
|
+
"c450336b8031f4ea9b3a614db3908b1f21c37559098b4e24672947172bb42de4",
|
|
507
|
+
"aaa569c4934575aae78c2a3901ef44696501ee56d0cb700bfa2559ac79677d6c",
|
|
508
|
+
"a9a2484c102a9392ffa00edc684b8feddea9ccc6a881d35433c72594e72959c5",
|
|
509
|
+
"f9cfc29b561d55da7838d371f1307bf015822bdea2961cfb6dd0b6da6232eede",
|
|
510
|
+
"1e769efadec0325d53de5d0c1023eea675499d9d67b086bcfc4c507ca0f25dac",
|
|
511
|
+
"749ef2d63c4dda83e3885b7c2dfaedc9f2970dc09ae310082ad5367bc322cebc",
|
|
512
|
+
"2402bf5b6b7a00aba5e6200c47277bdf6bb65e4d255f8126b8615eb9d1168545",
|
|
513
|
+
"7f68e4204587602d53d135152d9a0b4fa7e8da1cc0c43f02813debb8c3a99528",
|
|
514
|
+
"0d7e1a6de3ced156e54b9a50737835f49c424846e86adf28f1b19a1aa46f0f5a",
|
|
515
|
+
"2c894914dd47b1764c63bd525891246c8adcb14b0f00c446e84e28320451a213",
|
|
516
|
+
"0ab74921c2222979701499bc12d53c5a0942f51a7b1dd3a090ef58e83cea42f9"
|
|
517
|
+
],
|
|
518
|
+
"command:skills/skill-extraction-workflow/scripts/review_ledger_binding.py" =>
|
|
519
|
+
"6216104a650a7c30f78d0163d0a7cde37c9d809eb5967f291a30ebd516b1e337",
|
|
520
|
+
"file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#Sum spent rounds across all chains before opening one more" =>
|
|
521
|
+
"988cd8128ee6f93b4ea5a0cc8ada170401a1790831b2cd7b95a79642a76f1112",
|
|
522
|
+
"file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen" =>
|
|
523
|
+
"d999fadd8e3778bfdb9d54d34d3cabf965547ed39448a6ffba5e8310ca151612",
|
|
524
|
+
"file:skills/skill-extraction-workflow/references/extraction-quickstart.md#still owes the two-round chain before it can land" =>
|
|
525
|
+
"bf1fa382021b079b200b6d8d367e407bd0adbbac1d862e22d78062d83906c05e",
|
|
526
|
+
"file:skills/skill-extraction-workflow/references/extraction-quickstart.md#must exclude every evidence JSON the round has already added" =>
|
|
527
|
+
"d762cbab094279d26461843dfe12c5dd28ae41a1e9a40b7ae9f7f806fee51c36"
|
|
475
528
|
}.freeze
|
|
476
529
|
# TRUST BOUNDARY. The count answers "one row"; it cannot answer "WHICH row", so
|
|
477
530
|
# it is paired with the digest of the citing row in EXEMPT_ROW_DIGESTS above.
|
|
@@ -535,6 +588,11 @@ File.foreach(register_path).with_index(1) do |line, lineno|
|
|
|
535
588
|
target = File.join(root, rel)
|
|
536
589
|
if !syntactically_contained?(rel)
|
|
537
590
|
unresolved << [lineno, locator, "path escapes the repository"]
|
|
591
|
+
elsif !File.file?(target) && anchor_waived
|
|
592
|
+
# A waived command locator names an executable a superseding round
|
|
593
|
+
# deliberately retired; its absence is the recorded retirement, and the
|
|
594
|
+
# digest check below still pins which historical rows may cite it.
|
|
595
|
+
next
|
|
538
596
|
elsif !File.file?(target)
|
|
539
597
|
unresolved << [lineno, locator, "executable not found"]
|
|
540
598
|
elsif !resolves_inside?(root, rel)
|
|
@@ -701,20 +759,29 @@ end
|
|
|
701
759
|
|
|
702
760
|
exempt_uses.each do |locator, linenos|
|
|
703
761
|
rows = linenos.uniq.sort
|
|
704
|
-
|
|
705
|
-
|
|
762
|
+
# A waiver binds either one row (a digest string) or a fixed set of rows (an
|
|
763
|
+
# array of digests): a retired script or rule that several historical rows
|
|
764
|
+
# cited is still one retirement, and each of those rows is named by its digest.
|
|
765
|
+
expected = exempt_row_digests[locator]
|
|
766
|
+
expected_digests = expected.is_a?(Array) ? expected : [expected].compact
|
|
767
|
+
# How many rows a waiver covers is a property of the waiver, so it is read from
|
|
768
|
+
# the built-in table even when a test injects its own identity table.
|
|
769
|
+
builtin = EXEMPT_ROW_DIGESTS[locator]
|
|
770
|
+
builtin_rows = builtin.is_a?(Array) ? builtin.length : (builtin ? 1 : 0)
|
|
771
|
+
allowance = [expected_digests.length, builtin_rows, EXEMPT_USE_ALLOWANCE].max
|
|
772
|
+
if rows.length > allowance
|
|
773
|
+
rows.drop(allowance).each do |lineno|
|
|
706
774
|
unresolved << [lineno, locator,
|
|
707
|
-
"EXEMPT locator cited by #{rows.length} rows (allowance #{
|
|
708
|
-
"a waiver covers the
|
|
775
|
+
"EXEMPT locator cited by #{rows.length} rows (allowance #{allowance}); " \
|
|
776
|
+
"a waiver covers the recorded historical rows starting at line #{rows.first}, " \
|
|
709
777
|
"not a new row quoting the same retired locator"]
|
|
710
778
|
end
|
|
711
779
|
next
|
|
712
780
|
end
|
|
713
|
-
# The count says
|
|
714
|
-
# historical row and writing a different claim that cites the same
|
|
715
|
-
# the count
|
|
716
|
-
|
|
717
|
-
unless expected
|
|
781
|
+
# The count says how many rows; the digests say WHICH rows. Without them,
|
|
782
|
+
# deleting a historical row and writing a different claim that cites the same
|
|
783
|
+
# locator keeps the count and silently inherits the waiver.
|
|
784
|
+
if expected_digests.empty?
|
|
718
785
|
# A waiver with no recorded row identity keeps only the use-count layer,
|
|
719
786
|
# which cannot tell a rewritten or repurposed row from the one that was
|
|
720
787
|
# waived. On the BUILT-IN table that is a silent downgrade, so a new EXEMPT
|
|
@@ -729,13 +796,28 @@ exempt_uses.each do |locator, linenos|
|
|
|
729
796
|
end
|
|
730
797
|
next
|
|
731
798
|
end
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
799
|
+
remaining = expected_digests.tally
|
|
800
|
+
mismatched = 0
|
|
801
|
+
rows.each do |lineno|
|
|
802
|
+
actual = Digest::SHA256.hexdigest(register_lines[lineno - 1].to_s.rstrip)
|
|
803
|
+
if remaining[actual].to_i.positive?
|
|
804
|
+
remaining[actual] -= 1
|
|
805
|
+
next
|
|
806
|
+
end
|
|
807
|
+
mismatched += 1
|
|
808
|
+
unresolved << [lineno, locator,
|
|
809
|
+
"EXEMPT citing row does not match the waived row (digest #{actual[0, 12]} not recorded); " \
|
|
810
|
+
"a waiver covers specific unrepairable historical rows, so a rewritten or replaced row " \
|
|
811
|
+
"does not inherit it — restore the row, or land a new waiver entry with its own digest and reason"]
|
|
812
|
+
end
|
|
813
|
+
# A rewritten row is already reported above; only rows that are gone entirely
|
|
814
|
+
# are left to name here.
|
|
815
|
+
missing = remaining.values.sum - mismatched
|
|
816
|
+
next unless missing.positive?
|
|
817
|
+
unresolved << [rows.first, locator,
|
|
818
|
+
"EXEMPT entry has no citing row in the ledger for #{missing} of its #{expected_digests.length} " \
|
|
819
|
+
"recorded rows; a waived historical row was deleted or its locator was edited — restore the row, " \
|
|
820
|
+
"or retire its digest in the same change"]
|
|
739
821
|
end
|
|
740
822
|
|
|
741
823
|
# Both groups print before exiting. Bailing out on `malformed` alone would hide
|
|
@@ -736,7 +736,8 @@ DUAL_TRACK_REF="$REPO_ROOT/skills/skill-extraction-workflow/references/dual-trac
|
|
|
736
736
|
LEDGER_REF="$REPO_ROOT/skills/skill-extraction-workflow/references/source-register.md"
|
|
737
737
|
WALK_REF="$REPO_ROOT/skills/testing-strategy/references/run-killing-mutation-walk.md"
|
|
738
738
|
DT_SELF_AUDIT_SECTION='### The self-adversary enumeration — method detail (relocated from `SKILL.md`)'
|
|
739
|
-
DT_AUTHORITY_SECTION='### Findings
|
|
739
|
+
DT_AUTHORITY_SECTION='### Findings and dispositions'
|
|
740
|
+
DT_LANE_SECTION='### The extraction review lane: one review, one challenge'
|
|
740
741
|
TS_CORE_RULES_SECTION='## Core Rules'
|
|
741
742
|
WALK_PROBE_SECTION='## Encoded Probe For Destructive Artifacts'
|
|
742
743
|
LEDGER_RULE_PARAGRAPH='Round-consolidation rule (append-once)'
|
|
@@ -749,8 +750,8 @@ assert_in_section "$DUAL_TRACK_REF" "$DT_SELF_AUDIT_SECTION" '**Re-owe after fix
|
|
|
749
750
|
"process controls (re-owe rule anchored in the self-audit section)"
|
|
750
751
|
assert_in_section "$DUAL_TRACK_REF" "$DT_AUTHORITY_SECTION" 'A convergence or closure declaration must be written falsifiably.' \
|
|
751
752
|
"process controls (falsifiable closure declaration anchored in the authority section)"
|
|
752
|
-
assert_in_section "$DUAL_TRACK_REF" "$
|
|
753
|
-
"process controls (
|
|
753
|
+
assert_in_section "$DUAL_TRACK_REF" "$DT_LANE_SECTION" '**Task authority.**' \
|
|
754
|
+
"process controls (task authority anchored in the extraction review lane)"
|
|
754
755
|
# Remediation re-owes the pre-cover axes; third same-class round escalates to one full-matrix self-enumeration.
|
|
755
756
|
assert_same_line "$DUAL_TRACK_REF" 'remediation text written mid-round re-owes the draft-time axes BEFORE the candidate goes back to the reviewer' \
|
|
756
757
|
'**Re-owe after fixes.**' \
|
|
@@ -768,7 +769,7 @@ assert_same_line "$DUAL_TRACK_REF" 'states × failure points × orderings × res
|
|
|
768
769
|
'**Re-owe after fixes.**' \
|
|
769
770
|
"process controls (full-matrix axes named)"
|
|
770
771
|
# Convergence/closure declarations are falsifiable: named candidate/evidence/axes/open items, scoped adjectives.
|
|
771
|
-
assert_same_line "$DUAL_TRACK_REF" 'Name the
|
|
772
|
+
assert_same_line "$DUAL_TRACK_REF" 'Name the commit it covers' \
|
|
772
773
|
'must be written falsifiably' \
|
|
773
774
|
"process controls (declaration names the candidate identity)"
|
|
774
775
|
assert_same_line "$DUAL_TRACK_REF" "each lane's terminal evidence, the axes/dimensions the closing self-audit actually crossed, and every standing open item by name" \
|
|
@@ -780,29 +781,17 @@ assert_same_line "$DUAL_TRACK_REF" 'cannot be checked false and is inconclusive'
|
|
|
780
781
|
assert_same_line "$DUAL_TRACK_REF" 'any "full X" adjective is scoped to the named axes, never wider' \
|
|
781
782
|
'must be written falsifiably' \
|
|
782
783
|
"process controls (full-adjective scoped to named axes)"
|
|
783
|
-
#
|
|
784
|
-
assert_same_line "$DUAL_TRACK_REF" 'necessary in-scope fixes, tests and review
|
|
785
|
-
'
|
|
786
|
-
assert_same_line "$DUAL_TRACK_REF" '
|
|
787
|
-
'
|
|
788
|
-
assert_same_line "$DUAL_TRACK_REF" '
|
|
789
|
-
'
|
|
790
|
-
assert_same_line "$DUAL_TRACK_REF" '
|
|
791
|
-
'
|
|
792
|
-
assert_same_line "$DUAL_TRACK_REF"
|
|
793
|
-
'
|
|
794
|
-
assert_same_line "$DUAL_TRACK_REF" 'fresh current-candidate bindings and preserves every prior receipt, focus, finding and disposition' \
|
|
795
|
-
'`continuation_authorization`' "process controls (fresh binding does not discard history)"
|
|
796
|
-
assert_same_line "$DUAL_TRACK_REF" 'The existing per-sequence format, timeout and validation bounds remain unchanged' \
|
|
797
|
-
'`continuation_authorization`' "process controls (bounded invocation format remains enforced)"
|
|
798
|
-
assert_same_line "$DUAL_TRACK_REF" 'never relabel these calls as newly human-requested or erase earlier spending' \
|
|
799
|
-
'`continuation_authorization`' "process controls (no fabricated human request or count reset)"
|
|
800
|
-
assert_same_line "$DUAL_TRACK_REF" 'Ask only for scope or authority the original task lacks, an explicit user limit' \
|
|
801
|
-
'`continuation_authorization`' "process controls (real missing authority and user limits remain blocking)"
|
|
802
|
-
assert_same_line "$DUAL_TRACK_REF" 'Continuation waives no review, test or evidence obligation and grants no merge, publication or risk-acceptance authority' \
|
|
803
|
-
'`continuation_authorization`' "process controls (continuation is not a waiver or landing authority)"
|
|
804
|
-
assert_same_line "$DUAL_TRACK_REF" 'Never infer a lane waiver from silence or from authorization to continue' \
|
|
805
|
-
'`continuation_authorization`' "process controls (no inferred lane waiver)"
|
|
784
|
+
# Task authority: necessary passes are inherited, never relabelled, never a waiver or merge grant.
|
|
785
|
+
assert_same_line "$DUAL_TRACK_REF" 'necessary in-scope fixes, tests and review passes by default' \
|
|
786
|
+
'**Task authority.**' "process controls (necessary review inherits task authority)"
|
|
787
|
+
assert_same_line "$DUAL_TRACK_REF" 'never relabels an Agent-run pass as newly human-requested' \
|
|
788
|
+
'**Task authority.**' "process controls (no fabricated human request)"
|
|
789
|
+
assert_same_line "$DUAL_TRACK_REF" 'never infers a lane waiver from silence or from authorization to continue' \
|
|
790
|
+
'**Task authority.**' "process controls (no inferred lane waiver)"
|
|
791
|
+
assert_same_line "$DUAL_TRACK_REF" 'Ask only for scope or authority the original task lacks, or when an explicit user limit' \
|
|
792
|
+
'**Task authority.**' "process controls (real missing authority and user limits remain blocking)"
|
|
793
|
+
assert_same_line "$DUAL_TRACK_REF" 'waives no review, test or evidence obligation and grants no merge, publication or risk-acceptance authority' \
|
|
794
|
+
'**Task authority.**' "process controls (passes are not a waiver or landing authority)"
|
|
806
795
|
assert_contains "$PRODUCT_SKILL" 'Necessary fixes, tests and review inherit task authorization' \
|
|
807
796
|
"process controls (implementation entry reaches inherited authority)"
|
|
808
797
|
assert_contains "$PRE_FINAL_REF" 'continuation_basis=existing-task-scope' \
|
|
@@ -18,7 +18,7 @@
|
|
|
18
18
|
# - test_generic_r0_leak_scan.sh
|
|
19
19
|
# - test_shared_git_surface_gate.sh
|
|
20
20
|
# - test_extraction_review_gate.sh
|
|
21
|
-
# -
|
|
21
|
+
# - test_check_review_evidence_present.sh
|
|
22
22
|
# - test_check_ccl_route_drift.sh
|
|
23
23
|
# - test_check_sync_pointers.sh
|
|
24
24
|
# - test_check_ccl_register_pending_exclusion.sh
|
|
@@ -122,10 +122,8 @@ fast_tests=(
|
|
|
122
122
|
test_generic_r0_leak_scan.sh
|
|
123
123
|
test_shared_git_surface_gate.sh
|
|
124
124
|
test_extraction_review_gate.sh
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
# inspected it: own throwaway git repo, no clone, seconds.
|
|
128
|
-
test_review_ledger_binding.sh
|
|
125
|
+
# Review-evidence presence gate: own throwaway git repo, seconds.
|
|
126
|
+
test_check_review_evidence_present.sh
|
|
129
127
|
# Candidate-SHA-bound gate receipts (mint/verify): own throwaway git repo,
|
|
130
128
|
# no clone, seconds — belongs in the lane every run exercises.
|
|
131
129
|
test_gate_receipt.sh
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Regression for check_review_evidence_present.py: a pull request that changes
|
|
3
|
+
# skills/ or hooks/ must carry at least one conclusive review result; everything
|
|
4
|
+
# else passes untouched. Own throwaway git repo.
|
|
5
|
+
set -euo pipefail
|
|
6
|
+
|
|
7
|
+
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
8
|
+
GATE="$SCRIPT_DIR/check_review_evidence_present.py"
|
|
9
|
+
TMP="$(mktemp -d "${TMPDIR:-/tmp}/review-evidence.XXXXXX")"
|
|
10
|
+
trap 'rm -rf "$TMP"' EXIT
|
|
11
|
+
|
|
12
|
+
pass=0
|
|
13
|
+
fail() { echo "FAIL: $*" >&2; exit 1; }
|
|
14
|
+
ok() { pass=$((pass + 1)); echo "ok - $1"; }
|
|
15
|
+
|
|
16
|
+
REPO="$TMP/repo"
|
|
17
|
+
git init -q "$REPO"
|
|
18
|
+
git -C "$REPO" config user.email test@example.invalid
|
|
19
|
+
git -C "$REPO" config user.name "Test User"
|
|
20
|
+
mkdir -p "$REPO/skills/demo" "$REPO/docs"
|
|
21
|
+
printf -- '---\nname: demo\ndescription: demo skill\n---\n\n# demo\n\nKeep the rule, always.\n' > "$REPO/skills/demo/SKILL.md"
|
|
22
|
+
echo "# doc" > "$REPO/docs/a.md"
|
|
23
|
+
git -C "$REPO" add -A && git -C "$REPO" commit -qm base
|
|
24
|
+
BASE="$(git -C "$REPO" rev-parse HEAD)"
|
|
25
|
+
|
|
26
|
+
result() { # result <path> <mode> <status>
|
|
27
|
+
mkdir -p "$(dirname "$REPO/$1")"
|
|
28
|
+
printf '{"schema_version":3,"mode":"%s","status":"%s","selected_client":"codex"}\n' \
|
|
29
|
+
"$2" "$3" > "$REPO/$1"
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
# case <label> <expected rc> <expected token> <setup...>: reset to base, apply setup, commit, run.
|
|
33
|
+
run_case() {
|
|
34
|
+
local label="$1" want_rc="$2" want="$3"; shift 3
|
|
35
|
+
git -C "$REPO" checkout -q --detach "$BASE"
|
|
36
|
+
git -C "$REPO" clean -qfdx
|
|
37
|
+
"$@"
|
|
38
|
+
git -C "$REPO" add -A
|
|
39
|
+
git -C "$REPO" commit -qm case --allow-empty
|
|
40
|
+
set +e
|
|
41
|
+
out="$(python3 "$GATE" --repo-root "$REPO" --base "$BASE" 2>&1)"
|
|
42
|
+
rc=$?
|
|
43
|
+
set -e
|
|
44
|
+
[ "$rc" = "$want_rc" ] || fail "$label: expected rc=$want_rc got rc=$rc: $out"
|
|
45
|
+
case "$out" in *"$want"*) : ;; *) fail "$label: expected '$want': $out" ;; esac
|
|
46
|
+
ok "$label"
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
docs_only() { echo "more" >> "$REPO/docs/a.md"; }
|
|
50
|
+
skill_change() { echo "rule" >> "$REPO/skills/demo/SKILL.md"; }
|
|
51
|
+
hook_change() { mkdir -p "$REPO/hooks"; echo "#!/bin/sh" > "$REPO/hooks/h.sh"; }
|
|
52
|
+
both_passes() { result specs/r1/evidence/round1-review.json review findings; result specs/r1/evidence/round2-challenge.json challenge passed; }
|
|
53
|
+
|
|
54
|
+
run_case "no skills or hooks change needs nothing" 0 review_evidence_not_required docs_only
|
|
55
|
+
run_case "skill change with no evidence is refused" 1 "no conclusive review result" skill_change
|
|
56
|
+
run_case "hook change with no evidence is refused" 1 review_evidence_missing hook_change
|
|
57
|
+
run_case "review plus challenge passes" 0 "review_evidence_present_ok: 1 review, 1 challenge" \
|
|
58
|
+
bash -c "$(declare -f result skill_change both_passes); REPO='$REPO'; skill_change; both_passes"
|
|
59
|
+
run_case "a file replaced by a symlink still needs evidence" 1 review_evidence_missing \
|
|
60
|
+
bash -c "rm '$REPO/skills/demo/SKILL.md'; ln -s ../../docs/a.md '$REPO/skills/demo/SKILL.md'"
|
|
61
|
+
run_case "a review alone is enough for the gate" 0 "review_evidence_present_ok: 1 review, 0 challenge" \
|
|
62
|
+
bash -c "$(declare -f result skill_change); REPO='$REPO'; skill_change; result specs/r1/evidence/round1-review.json review passed"
|
|
63
|
+
run_case "a challenge alone does not satisfy the gate" 1 review_evidence_missing \
|
|
64
|
+
bash -c "$(declare -f result skill_change); REPO='$REPO'; skill_change; result specs/r1/evidence/round2-challenge.json challenge passed"
|
|
65
|
+
run_case "an inconclusive result does not count" 1 review_evidence_missing \
|
|
66
|
+
bash -c "$(declare -f result skill_change); REPO='$REPO'; skill_change; result specs/r1/evidence/round1-review.json review inconclusive"
|
|
67
|
+
run_case "a result outside an evidence directory does not count" 1 review_evidence_missing \
|
|
68
|
+
bash -c "$(declare -f result skill_change); REPO='$REPO'; skill_change; result specs/r1/round1-review.json review passed; result specs/r1/round2-challenge.json challenge passed"
|
|
69
|
+
run_case "malformed JSON does not count" 1 review_evidence_missing \
|
|
70
|
+
bash -c "mkdir -p '$REPO/specs/r1/evidence'; echo 'not json' > '$REPO/specs/r1/evidence/round1-review.json'; echo rule >> '$REPO/skills/demo/SKILL.md'"
|
|
71
|
+
|
|
72
|
+
set +e
|
|
73
|
+
out="$(python3 "$GATE" --repo-root "$REPO" --base does-not-exist 2>&1)"
|
|
74
|
+
rc=$?
|
|
75
|
+
set -e
|
|
76
|
+
[ "$rc" = 2 ] || fail "an unresolvable base must be unevaluated (rc=2), got rc=$rc"
|
|
77
|
+
case "$out" in *review_evidence_unevaluated*) : ;; *) fail "unresolvable base reason: $out" ;; esac
|
|
78
|
+
ok "an unresolvable base is unevaluated, never a pass"
|
|
79
|
+
|
|
80
|
+
[ "$pass" = 11 ] || fail "expected 11 assertions, saw $pass"
|
|
81
|
+
echo "test_check_review_evidence_present_ok ($pass assertions)"
|