@ccoalm/ccl-skills 0.16.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +8 -7
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-review-covers-head.sh +128 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-untracked-background.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_review_covers_head.sh +104 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_untracked_background.sh +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +13 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +78 -126
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/wording-only-review.md +136 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +521 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +25 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_order.sh +30 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +439 -19
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +14 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +6 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +48 -119
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +9 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check_review_evidence_present.py +122 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +4 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +37 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +102 -20
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +16 -27
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +3 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_review_evidence_present.sh +81 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +137 -279
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +41 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +49 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/closeout-reread.md +40 -0
- package/dist/assets/release.json +72 -52
- package/package.json +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +0 -1242
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +0 -978
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +0 -1477
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +0 -1183
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh
CHANGED
|
@@ -951,7 +951,8 @@ cat >"$WORK/review-plan.json" <<'JSON'
|
|
|
951
951
|
{"concern": "tests_evidence", "conclusion": "Focused deterministic contract tests cover the change.", "evidence_refs": ["e1"]},
|
|
952
952
|
{"concern": "compatibility", "conclusion": "Existing provider routing remains backward compatible.", "evidence_refs": ["e1"]},
|
|
953
953
|
{"concern": "rollout_rollback", "conclusion": "The local CLI change has a direct revert path.", "evidence_refs": ["e1"]},
|
|
954
|
-
{"concern": "observability_operations", "conclusion": "The JSON envelope exposes stage and depth for diagnosis.", "evidence_refs": ["e1"]}
|
|
954
|
+
{"concern": "observability_operations", "conclusion": "The JSON envelope exposes stage and depth for diagnosis.", "evidence_refs": ["e1"]},
|
|
955
|
+
{"concern": "claim_strength", "conclusion": "Each claim is scoped to the fixture it was observed on.", "evidence_refs": ["e1"]}
|
|
955
956
|
],
|
|
956
957
|
"evidence": [
|
|
957
958
|
{"id": "e1", "result": "Deterministic fake-wrapper contract fixture."}
|
|
@@ -1033,7 +1034,8 @@ cat >"$WORK/placeholder-plan.json" <<'JSON'
|
|
|
1033
1034
|
{"concern": "safety", "conclusion": "Packet and tool boundaries remain fail closed.", "evidence_refs": ["e1"]},
|
|
1034
1035
|
{"concern": "failure_paths", "conclusion": "Invalid and inconclusive paths remain terminal.", "evidence_refs": ["e1"]},
|
|
1035
1036
|
{"concern": "tests_evidence", "conclusion": "A focused regression proves filler is rejected.", "evidence_refs": ["e1"]},
|
|
1036
|
-
{"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]}
|
|
1037
|
+
{"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]},
|
|
1038
|
+
{"concern": "claim_strength", "conclusion": "No claim reaches past the filler-rejection fixture.", "evidence_refs": ["e1"]}
|
|
1037
1039
|
],
|
|
1038
1040
|
"evidence": [{"id": "e1", "result": "Deterministic placeholder-validation fixture."}]
|
|
1039
1041
|
}
|
|
@@ -1058,6 +1060,7 @@ cat >"$WORK/high-risk-plan.json" <<'JSON'
|
|
|
1058
1060
|
{"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]},
|
|
1059
1061
|
{"concern": "rollout_rollback", "conclusion": "The local contract change has a direct revert path.", "evidence_refs": ["e1"]},
|
|
1060
1062
|
{"concern": "observability_operations", "conclusion": "The result exposes depth and risk tags for diagnosis.", "evidence_refs": ["e1"]},
|
|
1063
|
+
{"concern": "claim_strength", "conclusion": "Each claim is scoped to the high-risk fixture it was observed on.", "evidence_refs": ["e1"]},
|
|
1061
1064
|
{"concern": "high_risk_boundary", "conclusion": "Bypass attempts cannot remove controller-required concerns.", "evidence_refs": ["e1"]}
|
|
1062
1065
|
],
|
|
1063
1066
|
"evidence": [{"id": "e1", "result": "Deterministic high-risk gate fixture."}]
|
|
@@ -1081,7 +1084,8 @@ cat >"$WORK/near-limit-plan.json" <<JSON
|
|
|
1081
1084
|
{"concern": "safety", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
1082
1085
|
{"concern": "failure_paths", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
1083
1086
|
{"concern": "tests_evidence", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
1084
|
-
{"concern": "compatibility", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]}
|
|
1087
|
+
{"concern": "compatibility", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
1088
|
+
{"concern": "claim_strength", "conclusion": "Scoped to this size fixture.", "evidence_refs": ["e1"]}
|
|
1085
1089
|
],
|
|
1086
1090
|
"evidence": [{"id": "e1", "result": "$large_evidence"}]
|
|
1087
1091
|
}
|
|
@@ -1374,6 +1378,144 @@ for malformed_plan in non-string-owner-review-plan empty-owner-review-plan; do
|
|
|
1374
1378
|
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true'
|
|
1375
1379
|
done
|
|
1376
1380
|
|
|
1381
|
+
# The claim-strength walk is a required concern, so it is owed BEFORE round 1 --
|
|
1382
|
+
# the one point in a round where correcting an overstated claim costs nothing. The
|
|
1383
|
+
# class it covers (absolutes, universals, causal and exhaustiveness claims the cited
|
|
1384
|
+
# evidence does not carry) otherwise keeps arriving as a LATE correction, after the
|
|
1385
|
+
# receipts are bound, where any candidate edit voids them.
|
|
1386
|
+
python3 - "$WORK/review-plan.json" "$WORK/no-claim-strength-review-plan.json" <<'PLAN'
|
|
1387
|
+
import json, sys
|
|
1388
|
+
from pathlib import Path
|
|
1389
|
+
plan = json.loads(Path(sys.argv[1]).read_text())
|
|
1390
|
+
plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != "claim_strength"]
|
|
1391
|
+
Path(sys.argv[2]).write_text(json.dumps(plan))
|
|
1392
|
+
PLAN
|
|
1393
|
+
reset_case passed unavailable unavailable
|
|
1394
|
+
out="$(run_gate --review-plan-file "$WORK/no-claim-strength-review-plan.json")"; rc=$?
|
|
1395
|
+
check "a plan that skips the claim-strength walk fails before any provider runs" \
|
|
1396
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true self_review_gate.required_triggers.0=before_external_review'
|
|
1397
|
+
|
|
1398
|
+
reset_case passed unavailable unavailable
|
|
1399
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1400
|
+
check "the build reviewer is asked to check claim strength" \
|
|
1401
|
+
'[ "$rc" = 0 ] && json_fields "$out" reviewed_concerns.5=claim_strength'
|
|
1402
|
+
|
|
1403
|
+
# The exported list is only worth deriving from if it IS the enforced one. Build a
|
|
1404
|
+
# plan covering exactly what the controller prints, and then drop each printed
|
|
1405
|
+
# concern in turn: acceptance proves the print covers everything the gate demands,
|
|
1406
|
+
# and every single-drop rejection proves nothing printed is decorative. Without
|
|
1407
|
+
# both directions a caller could derive from a list that had quietly diverged --
|
|
1408
|
+
# which is the drift this export exists to remove.
|
|
1409
|
+
# The exit status is asserted too. This suite runs without errexit, so a command
|
|
1410
|
+
# substitution silently discards it: a printer that emits the right concerns and then
|
|
1411
|
+
# fails would satisfy a non-empty check and report agreement it never reached.
|
|
1412
|
+
printed_rc=0
|
|
1413
|
+
printed_concerns="$("$DIR/review_gate.sh" --print-required-concerns --stage build)" || printed_rc=$?
|
|
1414
|
+
check "the controller can print the concern set a plan owes, and succeeds doing it" \
|
|
1415
|
+
'[ -n "$printed_concerns" ] && [ "$printed_rc" = 0 ]'
|
|
1416
|
+
# Proving agreement at ONE depth leaves the other branch free to diverge with every
|
|
1417
|
+
# test green -- and release/high-risk is the branch that carries the most concerns.
|
|
1418
|
+
# Assert the depth-raising branch answers what the gate itself derives for it.
|
|
1419
|
+
# Parity includes what each side REFUSES. A printer that answers for tags the enforcer
|
|
1420
|
+
# rejects reintroduces the divergence this export removes: a caller deriving from a
|
|
1421
|
+
# malformed tag would get a list where the real round fails closed.
|
|
1422
|
+
for bad_tag in "a b" "" "$(printf 'x%.0s' $(seq 81))"; do
|
|
1423
|
+
# rc captured without touching shell options: this suite runs under `set -uo pipefail`
|
|
1424
|
+
# and enabling errexit here would abort every later case at its first non-zero command.
|
|
1425
|
+
bad_tag_rc=0
|
|
1426
|
+
"$DIR/review_gate.sh" --print-required-concerns --stage build --risk-tag "$bad_tag" >/dev/null 2>&1 || bad_tag_rc=$?
|
|
1427
|
+
# Parity is a claim about TWO sides, so both are exercised: asserting only the
|
|
1428
|
+
# printer would keep these checks green if the enforcer's own rejection were
|
|
1429
|
+
# removed, which is the half this pair exists to tie together.
|
|
1430
|
+
enforcer_tag_rc=0
|
|
1431
|
+
run_gate --risk-tag "$bad_tag" >/dev/null 2>&1 || enforcer_tag_rc=$?
|
|
1432
|
+
check "printer and enforcer both refuse the same malformed risk tag (${#bad_tag} chars)" \
|
|
1433
|
+
'[ "$bad_tag_rc" != 0 ] && [ "$enforcer_tag_rc" != 0 ]'
|
|
1434
|
+
done
|
|
1435
|
+
printed_release_rc=0
|
|
1436
|
+
printed_release="$("$DIR/review_gate.sh" --print-required-concerns --stage explore --risk-tag shared-gate)" || printed_release_rc=$?
|
|
1437
|
+
check "the raised-depth print succeeds" '[ "$printed_release_rc" = 0 ]'
|
|
1438
|
+
# Hoisted for the same reason as the calls above: nested inside the comparison, this
|
|
1439
|
+
# printer call's exit status was discarded, so a regression failing only for explicit
|
|
1440
|
+
# release depth would have compared equal and passed. Every printer invocation in this
|
|
1441
|
+
# suite now has its status asserted.
|
|
1442
|
+
printed_plain_release_rc=0
|
|
1443
|
+
printed_plain_release="$("$DIR/review_gate.sh" --print-required-concerns --stage release)" || printed_plain_release_rc=$?
|
|
1444
|
+
check "the plain release print succeeds" '[ "$printed_plain_release_rc" = 0 ]'
|
|
1445
|
+
check "a high-risk tag raises the printed set to release depth and adds the boundary concern" \
|
|
1446
|
+
'[ "$(printf %s "$printed_release" | tr "\n" " ")" = "$(printf "%s\nhigh_risk_boundary" "$printed_plain_release" | tr "\n" " ")" ]'
|
|
1447
|
+
python3 - "$WORK/review-plan.json" "$WORK/printed-plan.json" $printed_concerns <<'PLAN'
|
|
1448
|
+
import json, sys
|
|
1449
|
+
from pathlib import Path
|
|
1450
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
1451
|
+
printed = sys.argv[3:]
|
|
1452
|
+
by_concern = {row["concern"]: row for row in source["self_review"]}
|
|
1453
|
+
source["self_review"] = [
|
|
1454
|
+
by_concern.get(concern, {"concern": concern,
|
|
1455
|
+
"conclusion": f"The fixture covers {concern}.",
|
|
1456
|
+
"evidence_refs": ["e1"]})
|
|
1457
|
+
for concern in printed
|
|
1458
|
+
]
|
|
1459
|
+
Path(sys.argv[2]).write_text(json.dumps(source))
|
|
1460
|
+
PLAN
|
|
1461
|
+
reset_case passed unavailable unavailable
|
|
1462
|
+
out="$(run_gate --review-plan-file "$WORK/printed-plan.json" --allow-fallback-egress)"; rc=$?
|
|
1463
|
+
check "a plan built from the printed set satisfies the gate" '[ "$rc" = 0 ]'
|
|
1464
|
+
printed_drop_failures=0
|
|
1465
|
+
for dropped in $printed_concerns; do
|
|
1466
|
+
python3 - "$WORK/printed-plan.json" "$WORK/printed-plan-minus.json" "$dropped" <<'PLAN'
|
|
1467
|
+
import json, sys
|
|
1468
|
+
from pathlib import Path
|
|
1469
|
+
plan = json.loads(Path(sys.argv[1]).read_text())
|
|
1470
|
+
plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != sys.argv[3]]
|
|
1471
|
+
Path(sys.argv[2]).write_text(json.dumps(plan))
|
|
1472
|
+
PLAN
|
|
1473
|
+
reset_case passed unavailable unavailable
|
|
1474
|
+
out="$(run_gate --review-plan-file "$WORK/printed-plan-minus.json")"; rc=$?
|
|
1475
|
+
if [ "$rc" = 2 ] && json_fields "$out" reason_code=self_review_incomplete; then
|
|
1476
|
+
printed_drop_failures=$((printed_drop_failures+1))
|
|
1477
|
+
fi
|
|
1478
|
+
done
|
|
1479
|
+
check "every printed concern is one the gate actually demands" \
|
|
1480
|
+
'[ "$printed_drop_failures" = "$(printf %s "$printed_concerns" | wc -w | tr -d " ")" ]'
|
|
1481
|
+
|
|
1482
|
+
# The same two directions at the OTHER depth. A printer that agreed with the gate at
|
|
1483
|
+
# build and diverged at release/high-risk would keep every test above green, and
|
|
1484
|
+
# release is the branch carrying the most concerns.
|
|
1485
|
+
python3 - "$WORK/high-risk-plan.json" "$WORK/printed-release-plan.json" $printed_release <<'PLAN'
|
|
1486
|
+
import json, sys
|
|
1487
|
+
from pathlib import Path
|
|
1488
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
1489
|
+
by_concern = {row["concern"]: row for row in source["self_review"]}
|
|
1490
|
+
source["self_review"] = [
|
|
1491
|
+
by_concern.get(concern, {"concern": concern,
|
|
1492
|
+
"conclusion": f"The high-risk fixture covers {concern}.",
|
|
1493
|
+
"evidence_refs": ["e1"]})
|
|
1494
|
+
for concern in sys.argv[3:]
|
|
1495
|
+
]
|
|
1496
|
+
Path(sys.argv[2]).write_text(json.dumps(source))
|
|
1497
|
+
PLAN
|
|
1498
|
+
reset_case passed unavailable unavailable
|
|
1499
|
+
out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/printed-release-plan.json" --review-chain-id printed-release --autonomous-review-index 1 --allow-fallback-egress)"; rc=$?
|
|
1500
|
+
check "a plan built from the printed release set satisfies the raised-depth gate" '[ "$rc" = 0 ]'
|
|
1501
|
+
printed_release_drop_failures=0
|
|
1502
|
+
for dropped in $printed_release; do
|
|
1503
|
+
python3 - "$WORK/printed-release-plan.json" "$WORK/printed-release-minus.json" "$dropped" <<'PLAN'
|
|
1504
|
+
import json, sys
|
|
1505
|
+
from pathlib import Path
|
|
1506
|
+
plan = json.loads(Path(sys.argv[1]).read_text())
|
|
1507
|
+
plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != sys.argv[3]]
|
|
1508
|
+
Path(sys.argv[2]).write_text(json.dumps(plan))
|
|
1509
|
+
PLAN
|
|
1510
|
+
reset_case passed unavailable unavailable
|
|
1511
|
+
out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/printed-release-minus.json" --review-chain-id printed-release-minus --autonomous-review-index 1)"; rc=$?
|
|
1512
|
+
if [ "$rc" = 2 ] && json_fields "$out" reason_code=self_review_incomplete; then
|
|
1513
|
+
printed_release_drop_failures=$((printed_release_drop_failures+1))
|
|
1514
|
+
fi
|
|
1515
|
+
done
|
|
1516
|
+
check "every printed release concern is one the raised-depth gate actually demands" \
|
|
1517
|
+
'[ "$printed_release_drop_failures" = "$(printf %s "$printed_release" | wc -w | tr -d " ")" ]'
|
|
1518
|
+
|
|
1377
1519
|
owner_lstat_classification="$(python3 - "$DIR/review_gate.py" <<'PY'
|
|
1378
1520
|
import errno
|
|
1379
1521
|
import importlib.util
|
|
@@ -1679,7 +1821,7 @@ diff_alternate = root / "alternate.patch"
|
|
|
1679
1821
|
diff_source.write_bytes(original_diff)
|
|
1680
1822
|
diff_alternate.write_bytes(alternate_diff)
|
|
1681
1823
|
with replace_after_symlink_check(diff_source, diff_alternate):
|
|
1682
|
-
packet_path, digest, _candidate, _n, _, _ = review_gate.freeze_packet(
|
|
1824
|
+
packet_path, digest, _candidate, _n, _, _, _ = review_gate.freeze_packet(
|
|
1683
1825
|
SimpleNamespace(
|
|
1684
1826
|
cwd=str(root), diff_file=str(diff_source), base=None, paths=[]
|
|
1685
1827
|
),
|
|
@@ -1776,9 +1918,9 @@ check "diff, prior, and completion inputs are read once from a bounded opened de
|
|
|
1776
1918
|
|
|
1777
1919
|
# The reviewer's packet and the landing candidate are two objects. A widened
|
|
1778
1920
|
# packet exists so a reviewer can judge a claim against code outside the diff;
|
|
1779
|
-
# the candidate exists so
|
|
1921
|
+
# the candidate exists so a caller can compare a receipt with what actually
|
|
1780
1922
|
# lands. Aliasing them made the two mutually exclusive: widening produced a
|
|
1781
|
-
# receipt
|
|
1923
|
+
# receipt whose candidate no landing could match. These assert the split and the one
|
|
1782
1924
|
# invariant that replaces the equality -- the candidate appears in the packet
|
|
1783
1925
|
# verbatim, so nothing lands that its reviewer did not read.
|
|
1784
1926
|
subject_packet_probe="$(
|
|
@@ -1853,7 +1995,7 @@ def expect_refused(label, **kwargs):
|
|
|
1853
1995
|
|
|
1854
1996
|
|
|
1855
1997
|
# The base-derived subject: exactly what the landing binder recomputes.
|
|
1856
|
-
subject_path, subject_hash, subject_candidate_hash, subject_n, subject_paths, _ = freeze()
|
|
1998
|
+
subject_path, subject_hash, subject_candidate_hash, subject_n, subject_paths, _, _ = freeze()
|
|
1857
1999
|
subject_bytes = subject_path.read_bytes()
|
|
1858
2000
|
subject_path.unlink()
|
|
1859
2001
|
assert subject_hash == digest(subject_bytes)
|
|
@@ -1862,7 +2004,7 @@ assert subject_n == len(subject_bytes)
|
|
|
1862
2004
|
assert subject_paths == ["landing.txt"], subject_paths
|
|
1863
2005
|
|
|
1864
2006
|
# A7 -- with no --diff-file the two hashes are the same value, as they are today.
|
|
1865
|
-
plain_path, plain_packet_hash, plain_candidate_hash, _plain_n, plain_paths, _ = freeze()
|
|
2007
|
+
plain_path, plain_packet_hash, plain_candidate_hash, _plain_n, plain_paths, _, _ = freeze()
|
|
1866
2008
|
plain_path.unlink()
|
|
1867
2009
|
assert plain_packet_hash == subject_hash
|
|
1868
2010
|
assert plain_candidate_hash == subject_hash
|
|
@@ -1877,7 +2019,7 @@ context = (
|
|
|
1877
2019
|
)
|
|
1878
2020
|
widened = outside / "widened.patch"
|
|
1879
2021
|
widened.write_bytes(subject_bytes + context)
|
|
1880
|
-
wide_path, wide_packet_hash, wide_candidate_hash, wide_n, wide_paths, _ = freeze(
|
|
2022
|
+
wide_path, wide_packet_hash, wide_candidate_hash, wide_n, wide_paths, _, _ = freeze(
|
|
1881
2023
|
diff_file=widened
|
|
1882
2024
|
)
|
|
1883
2025
|
try:
|
|
@@ -1934,7 +2076,7 @@ expect_refused("an unrelated packet was accepted", diff_file=unrelated)
|
|
|
1934
2076
|
|
|
1935
2077
|
# A8 -- --diff-file alone keeps today's meaning: no base, so no subject, and
|
|
1936
2078
|
# the candidate hash stays the packet's own hash.
|
|
1937
|
-
alone_path, alone_packet_hash, alone_candidate_hash, _n, _, _ = freeze(
|
|
2079
|
+
alone_path, alone_packet_hash, alone_candidate_hash, _n, _, _, _ = freeze(
|
|
1938
2080
|
diff_file=widened, base_ref=None
|
|
1939
2081
|
)
|
|
1940
2082
|
alone_path.unlink()
|
|
@@ -2073,7 +2215,7 @@ check "complete but placeholder concern conclusions cannot false-green" \
|
|
|
2073
2215
|
reset_case findings unavailable unavailable
|
|
2074
2216
|
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
2075
2217
|
check "Claude findings remain findings" \
|
|
2076
|
-
'[ "$rc" = 0 ] && json_fields "$out" status=findings selected_client=claude next_action=triage_findings_and_continue_independent_work autonomous_review_budget=1 autonomous_review_index=1 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim self_review_gate.allowed_next_actions.0=deep_self_review self_review_gate.allowed_next_actions.1=continue_implementation self_review_gate.allowed_next_actions.2=continue_independent_work'
|
|
2218
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings selected_client=claude next_action=triage_findings_and_continue_independent_work autonomous_review_budget=1 autonomous_review_index=1 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim self_review_gate.allowed_next_actions.0=deep_self_review self_review_gate.allowed_next_actions.1=continue_implementation self_review_gate.allowed_next_actions.2=continue_independent_work && ! grep -q recurring_findings_design_check <<<"$out"'
|
|
2077
2219
|
|
|
2078
2220
|
reset_case quota passed unavailable
|
|
2079
2221
|
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
@@ -2351,7 +2493,7 @@ check "an initial review with challenge capacity requires a tracked Agent chain"
|
|
|
2351
2493
|
reset_case passed unavailable unavailable
|
|
2352
2494
|
out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/high-risk-plan.json" --review-chain-id high-risk-task --autonomous-review-index 1)"; rc=$?
|
|
2353
2495
|
check "high-risk tags raise explore to release depth and default one challenge" \
|
|
2354
|
-
'[ "$rc" = 0 ] && json_fields "$out" stage=explore stage_source=caller-declared review_depth=release risk_tags_source=caller-declared challenge_budget=1 challenge_rounds_remaining=1 review_chain_tracked=true review_chain_id=high-risk-task autonomous_review_budget=2 autonomous_review_index=1 autonomous_reviews_remaining=1 autonomous_review_allowed=true next_action=run_challenge completion_gated=true risk_tags.0=shared-gate reviewed_concerns.
|
|
2496
|
+
'[ "$rc" = 0 ] && json_fields "$out" stage=explore stage_source=caller-declared review_depth=release risk_tags_source=caller-declared challenge_budget=1 challenge_rounds_remaining=1 review_chain_tracked=true review_chain_id=high-risk-task autonomous_review_budget=2 autonomous_review_index=1 autonomous_reviews_remaining=1 autonomous_review_allowed=true next_action=run_challenge completion_gated=true risk_tags.0=shared-gate reviewed_concerns.8=high_risk_boundary self_review_gate.required=false self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.satisfied_triggers.1=risk_or_scope_escalation'
|
|
2355
2497
|
|
|
2356
2498
|
# Release/high-risk normally requires a challenge. The only single-review
|
|
2357
2499
|
# exception is a candidate-bound deterministic wording-only proof whose result
|
|
@@ -3160,7 +3302,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
3160
3302
|
--wording-only-proof-file "$WORK/wording-punctuation-proof.json")"; rc=$?
|
|
3161
3303
|
punctuation_proof_hash="$(shasum -a 256 "$WORK/wording-punctuation-proof.json" | awk '{print $1}')"
|
|
3162
3304
|
check "release wording-only punctuation scope can take one proof-bound review" \
|
|
3163
|
-
'[ "$rc" = 0 ] && json_fields "$out" challenge_budget=0 review_chain_tracked=false wording_only_scope.status=passed wording_only_scope.check_kind=markdown-punctuation-only wording_only_proof_sha256="$punctuation_proof_hash" reviewed_concerns.
|
|
3305
|
+
'[ "$rc" = 0 ] && json_fields "$out" challenge_budget=0 review_chain_tracked=false wording_only_scope.status=passed wording_only_scope.check_kind=markdown-punctuation-only wording_only_proof_sha256="$punctuation_proof_hash" reviewed_concerns.8=wording_only_boundary'
|
|
3164
3306
|
|
|
3165
3307
|
reset_case passed unavailable unavailable
|
|
3166
3308
|
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
@@ -3169,7 +3311,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
3169
3311
|
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
3170
3312
|
--wording-only-proof-file "$WORK/wording-punctuation-proof.json")"; rc=$?
|
|
3171
3313
|
check "build wording-only review records the same controller-bound proof" \
|
|
3172
|
-
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-punctuation-only reviewed_concerns.
|
|
3314
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-punctuation-only reviewed_concerns.6=wording_only_boundary'
|
|
3173
3315
|
printf '%s\n' "$out" >"$WORK/wording-punctuation-review.json"
|
|
3174
3316
|
reset_case passed unavailable unavailable
|
|
3175
3317
|
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
@@ -3245,7 +3387,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
3245
3387
|
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
3246
3388
|
--wording-only-proof-file "$WORK/wording-token-proof.json")"; rc=$?
|
|
3247
3389
|
check "build exact typo replacement can take one proof-bound review" \
|
|
3248
|
-
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.old_token=teh wording_only_scope.new_token=the wording_only_scope.expected_count=1 wording_only_scope.replacement_count=1 reviewed_concerns.
|
|
3390
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.old_token=teh wording_only_scope.new_token=the wording_only_scope.expected_count=1 wording_only_scope.replacement_count=1 reviewed_concerns.6=wording_only_boundary'
|
|
3249
3391
|
|
|
3250
3392
|
reset_case passed unavailable unavailable
|
|
3251
3393
|
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
@@ -3272,7 +3414,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
3272
3414
|
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
3273
3415
|
--wording-only-proof-file "$WORK/wording-base-proof.json")"; rc=$?
|
|
3274
3416
|
check "base-mode build wording proof freezes full context from line one" \
|
|
3275
|
-
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.replacement_count=1 reviewed_concerns.
|
|
3417
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.replacement_count=1 reviewed_concerns.6=wording_only_boundary'
|
|
3276
3418
|
|
|
3277
3419
|
for rejected_scope in multi-skill-wording truncated-context-wording symlink-mode-wording frontmatter-shift-insert-wording frontmatter-shift-delete-wording no-final-newline-wording invalid-octal-wording huge-hunk-number-wording zero-width-wording bidi-control-wording emoji-symbol-wording currency-symbol-wording decomposed-boundary-wording zwj-boundary-wording; do
|
|
3278
3420
|
reset_case passed unavailable unavailable
|
|
@@ -3499,6 +3641,13 @@ printf '%s\n' "$passed_round_one" >"$WORK/passed-round-one.json"
|
|
|
3499
3641
|
check "a passed first tracked round still owes its challenge before completion" \
|
|
3500
3642
|
'[ "$passed_round_one_rc" = 0 ] && json_fields "$passed_round_one" status=passed autonomous_review_index=1 autonomous_reviews_remaining=2 autonomous_review_allowed=true next_action=run_challenge completion_gated=true'
|
|
3501
3643
|
|
|
3644
|
+
# Control leg for the recurrence trigger: same shape, first findings round. Without it a
|
|
3645
|
+
# probe that fires for an unrelated reason would read as the recurrence being detected.
|
|
3646
|
+
reset_case findings unavailable unavailable
|
|
3647
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus passed-prior-findings --review-chain-id passed-task --autonomous-review-index 2 --prior-review-result-file "$WORK/passed-round-one.json")"; rc=$?
|
|
3648
|
+
check "findings after a clean prior round stay a first findings round" \
|
|
3649
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings self_review_gate.required_triggers.0=findings_returned && ! grep -q recurring_findings_design_check <<<"$out"'
|
|
3650
|
+
|
|
3502
3651
|
# Chain succession. A fix that touches the owner package moves selected_skills_sha256
|
|
3503
3652
|
# and ends the chain by design, so the post-fix candidate can never be challenged
|
|
3504
3653
|
# inside it. Succession opens ONE new chain whose first Agent round is a challenge,
|
|
@@ -3522,7 +3671,8 @@ python3 - "$WORK/succ-round-two.json" \
|
|
|
3522
3671
|
"$WORK/succ-predecessor-owner-moved.json" \
|
|
3523
3672
|
"$WORK/succ-predecessor-forged-controller.json" \
|
|
3524
3673
|
"$WORK/succ-predecessor-foreign-scope.json" \
|
|
3525
|
-
"$WORK/succ-predecessor-forged-terminal.json"
|
|
3674
|
+
"$WORK/succ-predecessor-forged-terminal.json" \
|
|
3675
|
+
"$WORK/succ-predecessor-complete-mode.json" <<'PY'
|
|
3526
3676
|
import json
|
|
3527
3677
|
from pathlib import Path
|
|
3528
3678
|
import sys
|
|
@@ -3548,6 +3698,10 @@ forged_terminal["challenge_index"] = 0
|
|
|
3548
3698
|
forged_terminal["autonomous_reviews_remaining"] = 1
|
|
3549
3699
|
forged_terminal["autonomous_review_allowed"] = True
|
|
3550
3700
|
Path(sys.argv[5]).write_text(json.dumps(forged_terminal, separators=(",", ":")))
|
|
3701
|
+
# Neither lane: a completion checkpoint is not a round the succession may carry.
|
|
3702
|
+
complete_mode = json.loads(json.dumps(source))
|
|
3703
|
+
complete_mode["mode"] = "complete"
|
|
3704
|
+
Path(sys.argv[6]).write_text(json.dumps(complete_mode, separators=(",", ":")))
|
|
3551
3705
|
PY
|
|
3552
3706
|
|
|
3553
3707
|
reset_case passed unavailable unavailable
|
|
@@ -3560,15 +3714,53 @@ out="$(run_challenge_gate --focus owner-moved --review-chain-id succ-owner-moved
|
|
|
3560
3714
|
check "a succession accepts the owner-package hash move that ended the prior chain" \
|
|
3561
3715
|
'[ "$rc" = 0 ] && json_fields "$out" mode=challenge predecessor_chain_id=succ-phase-one'
|
|
3562
3716
|
|
|
3717
|
+
# Budget is one review plus one challenge, so a fix ends the chain and the SECOND
|
|
3718
|
+
# findings round usually lands in the SUCCEEDING chain. Counting only in-chain rounds
|
|
3719
|
+
# would therefore never see the recurrence the trigger exists for.
|
|
3720
|
+
reset_case findings unavailable unavailable
|
|
3721
|
+
out="$(run_challenge_gate --focus recurring-findings --review-chain-id succ-recurrence --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-round-two.json")"; rc=$?
|
|
3722
|
+
check "findings after a predecessor chain that also returned findings raise the design check" \
|
|
3723
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings self_review_gate.required_triggers.0=findings_returned && grep -q recurring_findings_design_check <<<"$out"'
|
|
3724
|
+
|
|
3563
3725
|
reset_case passed unavailable unavailable
|
|
3564
3726
|
out="$(run_challenge_gate --focus no-predecessor --review-chain-id succ-orphan --autonomous-review-index 1)"; rc=$?
|
|
3565
3727
|
check "a tracked challenge cannot open a chain without a predecessor receipt" \
|
|
3566
3728
|
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession"*) false;; *) true;; esac'
|
|
3567
3729
|
|
|
3730
|
+
# A fix applied straight after the REVIEW ends the chain exactly as a fix after the
|
|
3731
|
+
# challenge does -- the owner digest moves either way -- so the ended chain's terminal
|
|
3732
|
+
# receipt is its review. Requiring a challenge receipt here forced that challenge to be
|
|
3733
|
+
# spent on a candidate the author had already decided to change, and bought no evidence
|
|
3734
|
+
# about the candidate that lands: the succession challenge covers it either way. What is
|
|
3735
|
+
# exempted is exactly one class -- a challenge on a candidate that will never land.
|
|
3568
3736
|
reset_case passed unavailable unavailable
|
|
3569
3737
|
out="$(run_challenge_gate --focus review-predecessor --review-chain-id succ-review-predecessor --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-round-one.json")"; rc=$?
|
|
3570
|
-
check "a succession
|
|
3571
|
-
'[ "$rc" =
|
|
3738
|
+
check "a succession may carry a chain whose terminal receipt is its review" \
|
|
3739
|
+
'[ "$rc" = 0 ] && json_fields "$out" mode=challenge review_chain_tracked=true review_chain_id=succ-review-predecessor autonomous_review_index=1 predecessor_chain_id=succ-phase-one'
|
|
3740
|
+
|
|
3741
|
+
# The exemption is bounded by the receipt's own arithmetic. This is a FORGERY guard and
|
|
3742
|
+
# is asserted as one: the fixture below is a shape the controller never emits, because a
|
|
3743
|
+
# genuine round-1 review reads the same whether its chain later ran a challenge or not.
|
|
3744
|
+
# A caller who spent the challenge and presents only the review is accepted here -- the
|
|
3745
|
+
# stateless controller cannot see omitted history -- so no test claims otherwise.
|
|
3746
|
+
python3 - "$WORK/succ-round-one.json" "$WORK/succ-predecessor-spent-review.json" <<'PLAN'
|
|
3747
|
+
import json, sys
|
|
3748
|
+
from pathlib import Path
|
|
3749
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
3750
|
+
spent = json.loads(json.dumps(source))
|
|
3751
|
+
spent["autonomous_reviews_remaining"] = 0
|
|
3752
|
+
spent["autonomous_review_allowed"] = False
|
|
3753
|
+
Path(sys.argv[2]).write_text(json.dumps(spent, separators=(",", ":")))
|
|
3754
|
+
PLAN
|
|
3755
|
+
reset_case passed unavailable unavailable
|
|
3756
|
+
out="$(run_challenge_gate --focus spent-review --review-chain-id succ-spent-review --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-predecessor-spent-review.json")"; rc=$?
|
|
3757
|
+
check "a succession rejects a forged review receipt whose own arithmetic says its chain is spent" \
|
|
3758
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession predecessor is not its chain'"'"'s terminal round"*) true;; *) false;; esac'
|
|
3759
|
+
|
|
3760
|
+
reset_case passed unavailable unavailable
|
|
3761
|
+
out="$(run_challenge_gate --focus complete-predecessor --review-chain-id succ-complete-predecessor --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-predecessor-complete-mode.json")"; rc=$?
|
|
3762
|
+
check "a succession rejects a predecessor that is neither a review nor a challenge round" \
|
|
3763
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession predecessor is not a tracked review or challenge receipt"*) true;; *) false;; esac'
|
|
3572
3764
|
|
|
3573
3765
|
# A mid-chain challenge is a live chain, not an ended one: succeeding it would
|
|
3574
3766
|
# silently retire rounds the wrapper still owes. chain-round-two above is round 2
|
|
@@ -4167,6 +4359,11 @@ post_deadline_ok" ]'
|
|
|
4167
4359
|
check "the staged contract declares the current result schema" \
|
|
4168
4360
|
'grep -q "current result envelope is schema 3" "$DIR/../references/staged-review-contract.md"'
|
|
4169
4361
|
|
|
4362
|
+
# The merge-side ledger binder was retired; a contract that still says a script
|
|
4363
|
+
# recomputes the candidate at merge time promises a check nothing performs.
|
|
4364
|
+
check "the staged contract promises no retired merge-side binder" \
|
|
4365
|
+
'! grep -q "review_ledger_binding" "$DIR/../references/staged-review-contract.md"'
|
|
4366
|
+
|
|
4170
4367
|
reset_case passed unavailable unavailable
|
|
4171
4368
|
out="$(run_gate --timeout 600 --total-timeout 90)"; rc=$?
|
|
4172
4369
|
claude_timeout="$(cat "$WORK/state/claude_timeout" 2>/dev/null || true)"
|
|
@@ -4405,6 +4602,11 @@ reset_case findings unavailable unavailable
|
|
|
4405
4602
|
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 2 --focus final-findings --review-chain-id long-task --autonomous-review-index 3 --prior-review-result-file "$WORK/chain-round-one.json" --prior-review-result-file "$WORK/chain-round-two.json")"; rc=$?
|
|
4406
4603
|
check "findings in the last tracked Agent round return to a post-budget checkpoint" \
|
|
4407
4604
|
'[ "$rc" = 0 ] && json_fields "$out" status=findings review_chain_tracked=true autonomous_review_index=3 autonomous_reviews_remaining=0 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true next_action=triage_findings_and_continue_independent_work self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.blocks.0=external_review self_review_gate.allowed_next_actions.2=continue_independent_work'
|
|
4605
|
+
# Round 1 of this chain also returned findings, so this is the second one: the rule the
|
|
4606
|
+
# trigger carries is about the RECURRENCE, and the agent reads it here rather than in a
|
|
4607
|
+
# skill it never loads while inside the chain.
|
|
4608
|
+
check "a second findings round in one chain raises the design check" \
|
|
4609
|
+
'[ "$rc" = 0 ] && json_fields "$out" self_review_gate.required_triggers.2=recurring_findings_design_check && grep -q decide_keep_delete_narrow_replace <<<"$out"'
|
|
4408
4610
|
|
|
4409
4611
|
reset_case passed unavailable unavailable
|
|
4410
4612
|
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus missing-history --review-chain-id long-task --autonomous-review-index 2)"; rc=$?
|
|
@@ -4540,6 +4742,224 @@ out="$(run_base_gate --allow-fallback-egress)"; rc=$?
|
|
|
4540
4742
|
check "base packet freezes tracked and untracked changes once" \
|
|
4541
4743
|
'[ "$rc" = 0 ] && grep -q tracked.txt "$WORK/state/claude_packet" && grep -q untracked.txt "$WORK/state/claude_packet" && grep -q "Untracked files" "$WORK/state/claude_packet"'
|
|
4542
4744
|
|
|
4745
|
+
# A review quotes the repository's own contract files after the candidate: the
|
|
4746
|
+
# tracked AGENTS.override.md-or-AGENTS.md, CLAUDE.md and .claude/CLAUDE.md of
|
|
4747
|
+
# every directory from the root down to a changed path, root first. Untracked
|
|
4748
|
+
# and CLAUDE.local.md files are personal and never quoted; a file off every
|
|
4749
|
+
# changed path is not quoted; a linked file is omitted, not followed.
|
|
4750
|
+
contract_repo="$WORK/contract-repo"
|
|
4751
|
+
rm -rf "$contract_repo"
|
|
4752
|
+
mkdir -p "$contract_repo/sub/deep" "$contract_repo/other" "$contract_repo/.claude" \
|
|
4753
|
+
"$contract_repo/linked" "$contract_repo/big"
|
|
4754
|
+
(
|
|
4755
|
+
cd "$contract_repo"
|
|
4756
|
+
git init -q
|
|
4757
|
+
git config user.email test@example.invalid
|
|
4758
|
+
git config user.name 'Test User'
|
|
4759
|
+
printf 'root agents rule\n' >AGENTS.md
|
|
4760
|
+
printf 'root claude rule\n' >.claude/CLAUDE.md
|
|
4761
|
+
printf 'personal note must not egress\n' >CLAUDE.local.md
|
|
4762
|
+
printf 'sub override rule\n' >sub/AGENTS.override.md
|
|
4763
|
+
printf 'sub shadowed rule\n' >sub/AGENTS.md
|
|
4764
|
+
printf 'sub claude rule\n' >sub/CLAUDE.md
|
|
4765
|
+
printf 'other rule\n' >other/AGENTS.md
|
|
4766
|
+
ln -s ../AGENTS.md linked/AGENTS.md
|
|
4767
|
+
python3 -c 'print("x" * 40000)' >big/AGENTS.md
|
|
4768
|
+
for dir in sub/deep other linked big; do printf 'before\n' >"$dir/code.txt"; done
|
|
4769
|
+
printf 'before\n' >.claude/notes.txt
|
|
4770
|
+
git add -A
|
|
4771
|
+
git commit -q -m initial
|
|
4772
|
+
printf 'untracked rule\n' >sub/deep/CLAUDE.md
|
|
4773
|
+
printf 'after\n' >sub/deep/code.txt
|
|
4774
|
+
)
|
|
4775
|
+
run_contract_gate() {
|
|
4776
|
+
REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
4777
|
+
--cwd "$contract_repo" --base HEAD --implementer-family openai \
|
|
4778
|
+
--review-plan-file "$WORK/review-plan.json" --allow-fallback-egress "$@"
|
|
4779
|
+
}
|
|
4780
|
+
contract_packet_check() { # <result json> <expected quoted paths, comma-joined> <coverage>
|
|
4781
|
+
JSON_PAYLOAD="$1" python3 - "$WORK/state/claude_packet" "$WORK/state/claude_profile" "$2" "$3" <<'PY' 2>&1
|
|
4782
|
+
import hashlib, json, os, re, sys
|
|
4783
|
+
result = json.loads(os.environ["JSON_PAYLOAD"])
|
|
4784
|
+
packet = open(sys.argv[1], "rb").read()
|
|
4785
|
+
profile = json.load(open(sys.argv[2], encoding="utf-8"))
|
|
4786
|
+
expected = [item for item in sys.argv[3].split(",") if item]
|
|
4787
|
+
quoted = re.findall(rb"^CCL_REPOSITORY_CONTRACT_[0-9a-f]{24}_BEGIN (.+)$", packet, re.M)
|
|
4788
|
+
assert [item.decode() for item in quoted] == expected, quoted
|
|
4789
|
+
assert b"personal note must not egress" not in packet
|
|
4790
|
+
assert hashlib.sha256(packet[: profile["candidate_bytes"]]).hexdigest() == result["candidate_sha256"]
|
|
4791
|
+
assert result["packet_sha256"] != result["candidate_sha256"]
|
|
4792
|
+
assert b"CCL_REPOSITORY_CONTRACT_" not in packet[: profile["candidate_bytes"]]
|
|
4793
|
+
concerns = [item["id"] for item in profile["required_concerns"]]
|
|
4794
|
+
assert "repository_contract" in concerns, concerns
|
|
4795
|
+
contract = result["repository_contract"]
|
|
4796
|
+
assert contract["coverage"] == sys.argv[4], contract
|
|
4797
|
+
assert [item["path"] for item in contract["files"]] == expected, contract
|
|
4798
|
+
print("contract_packet_ok")
|
|
4799
|
+
PY
|
|
4800
|
+
}
|
|
4801
|
+
reset_case passed unavailable unavailable
|
|
4802
|
+
out="$(run_contract_gate --mode review)"; rc=$?
|
|
4803
|
+
check "review quotes the tracked contract files governing the changed path, root first" \
|
|
4804
|
+
'[ "$rc" = 0 ] && contract_packet_check "$out" "AGENTS.md,.claude/CLAUDE.md,sub/AGENTS.override.md,sub/CLAUDE.md" complete | grep -qx contract_packet_ok'
|
|
4805
|
+
|
|
4806
|
+
printf 'after\n' >"$contract_repo/linked/code.txt"
|
|
4807
|
+
printf 'after\n' >"$contract_repo/big/code.txt"
|
|
4808
|
+
# A change under .claude/ reaches .claude/CLAUDE.md twice (root and .claude/);
|
|
4809
|
+
# the expected list below admits it once.
|
|
4810
|
+
printf 'after\n' >"$contract_repo/.claude/notes.txt"
|
|
4811
|
+
reset_case passed unavailable unavailable
|
|
4812
|
+
out="$(run_contract_gate --mode review)"; rc=$?
|
|
4813
|
+
check "a linked or oversized contract file is omitted and reported, never followed or failed on" \
|
|
4814
|
+
'[ "$rc" = 0 ] && contract_packet_check "$out" "AGENTS.md,.claude/CLAUDE.md,sub/AGENTS.override.md,sub/CLAUDE.md" partial | grep -qx contract_packet_ok && json_fields "$out" repository_contract.omitted.0.path=big/AGENTS.md repository_contract.omitted.0.reason=size_budget repository_contract.omitted.1.path=linked/AGENTS.md repository_contract.omitted.1.reason=unreadable'
|
|
4815
|
+
|
|
4816
|
+
reset_case passed unavailable unavailable
|
|
4817
|
+
out="$(run_contract_gate --mode challenge --challenge-budget 1 --challenge-index 1 --focus "contract rules")"; rc=$?
|
|
4818
|
+
check "a challenge keeps its focus: no contract section and no contract concern" \
|
|
4819
|
+
'[ "$rc" = 0 ] && ! grep -q CCL_REPOSITORY_CONTRACT_ "$WORK/state/claude_packet" && python3 -c "import json,sys; p=json.load(open(sys.argv[1])); assert p[\"repository_contract\"] is None and all(c[\"id\"] != \"repository_contract\" for c in p[\"required_concerns\"])" "$WORK/state/claude_profile" && json_fields "$out" repository_contract=None'
|
|
4820
|
+
|
|
4821
|
+
receipt_file="$contract_repo/.git/ccl-code-review/last-review.json"
|
|
4822
|
+
check "a conclusive run leaves a local receipt of the HEAD it reviewed, outside the tree" \
|
|
4823
|
+
'[ "$rc" = 0 ] && [ -f "$receipt_file" ] && [ "$(jq -r .head "$receipt_file")" = "$(git -C "$contract_repo" rev-parse HEAD)" ] && [ "$(jq -r .worktree_clean "$receipt_file")" = false ] && [ "$(jq -r .mode "$receipt_file")" = challenge ]'
|
|
4824
|
+
|
|
4825
|
+
cp "$receipt_file" "$WORK/receipt-before"
|
|
4826
|
+
reset_case passed unavailable unavailable
|
|
4827
|
+
out="$(run_contract_gate --mode challenge --challenge-budget 1 --challenge-index 1)"; rc=$?
|
|
4828
|
+
check "an inconclusive run leaves the last receipt untouched" \
|
|
4829
|
+
'[ "$rc" = 2 ] && cmp -s "$receipt_file" "$WORK/receipt-before"'
|
|
4830
|
+
|
|
4831
|
+
# The receipt directory swapped for a link: the write is refused, never followed.
|
|
4832
|
+
mkdir -p "$WORK/receipt-target"
|
|
4833
|
+
rm -rf "$contract_repo/.git/ccl-code-review"
|
|
4834
|
+
ln -s "$WORK/receipt-target" "$contract_repo/.git/ccl-code-review"
|
|
4835
|
+
reset_case passed unavailable unavailable
|
|
4836
|
+
out="$(run_contract_gate --mode challenge --challenge-budget 1 --challenge-index 1 --focus "contract rules")"; rc=$?
|
|
4837
|
+
check "a linked receipt directory is never written through" \
|
|
4838
|
+
'[ "$rc" = 0 ] && [ -L "$contract_repo/.git/ccl-code-review" ] && [ -z "$(ls -A "$WORK/receipt-target")" ]'
|
|
4839
|
+
rm -f "$contract_repo/.git/ccl-code-review"
|
|
4840
|
+
|
|
4841
|
+
anchor_out="$(PYTHONPATH="$WORK/harness/scripts" python3 - <<'PY' 2>&1
|
|
4842
|
+
import review_gate
|
|
4843
|
+
a = {"git_dir": "/g", "head": "a" * 40, "worktree_clean": True}
|
|
4844
|
+
b = dict(a, head="b" * 40)
|
|
4845
|
+
c = dict(a, worktree_clean=False)
|
|
4846
|
+
assert review_gate.stable_review_anchor(a, dict(a)) == a
|
|
4847
|
+
assert review_gate.stable_review_anchor(a, b) is None
|
|
4848
|
+
assert review_gate.stable_review_anchor(a, c) is None
|
|
4849
|
+
assert review_gate.stable_review_anchor(None, a) is None
|
|
4850
|
+
print("anchor_stability_ok")
|
|
4851
|
+
PY
|
|
4852
|
+
)"
|
|
4853
|
+
check "a freeze that straddles a HEAD or cleanliness change records nothing" \
|
|
4854
|
+
'[ "$anchor_out" = anchor_stability_ok ]'
|
|
4855
|
+
|
|
4856
|
+
# A bare --diff-file packet may cover less than HEAD, so it records no receipt.
|
|
4857
|
+
git -C "$contract_repo" diff HEAD >"$WORK/contract.patch"
|
|
4858
|
+
rm -rf "$contract_repo/.git/ccl-code-review"
|
|
4859
|
+
reset_case passed unavailable unavailable
|
|
4860
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
4861
|
+
--mode review --cwd "$contract_repo" --diff-file "$WORK/contract.patch" --implementer-family openai \
|
|
4862
|
+
--review-plan-file "$WORK/review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
4863
|
+
check "a bare --diff-file review records no receipt" \
|
|
4864
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=passed && [ ! -e "$contract_repo/.git/ccl-code-review" ]'
|
|
4865
|
+
|
|
4866
|
+
swap_out="$(PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY' 2>&1
|
|
4867
|
+
import os, sys, review_gate
|
|
4868
|
+
root = os.path.realpath(os.path.join(sys.argv[1], "gitdir-swap"))
|
|
4869
|
+
os.makedirs(root)
|
|
4870
|
+
def anchor_for(path):
|
|
4871
|
+
st = os.stat(path)
|
|
4872
|
+
return {"git_dir": path, "git_dir_identity": [st.st_dev, st.st_ino], "head": "a" * 40, "worktree_clean": True}
|
|
4873
|
+
def receipt_under(path):
|
|
4874
|
+
return os.path.exists(os.path.join(path, "ccl-code-review", "last-review.json"))
|
|
4875
|
+
result = {"mode": "review", "status": "passed"}
|
|
4876
|
+
# A different directory swapped in under the anchored path.
|
|
4877
|
+
git_dir, other = os.path.join(root, "gitdir"), os.path.join(root, "other")
|
|
4878
|
+
os.mkdir(git_dir)
|
|
4879
|
+
anchor = anchor_for(git_dir)
|
|
4880
|
+
os.rename(git_dir, os.path.join(root, "moved"))
|
|
4881
|
+
os.mkdir(other)
|
|
4882
|
+
os.symlink(other, git_dir)
|
|
4883
|
+
review_gate.record_local_review(anchor, result)
|
|
4884
|
+
assert not receipt_under(other), "wrote through a swapped git directory"
|
|
4885
|
+
# The same directory relocated into a tree, its old path now a link to it.
|
|
4886
|
+
original = os.path.join(root, "original")
|
|
4887
|
+
os.mkdir(original)
|
|
4888
|
+
anchor = anchor_for(original)
|
|
4889
|
+
tree = os.path.join(root, "worktree")
|
|
4890
|
+
os.mkdir(tree)
|
|
4891
|
+
os.rename(original, os.path.join(tree, "relocated"))
|
|
4892
|
+
os.symlink(os.path.join(tree, "relocated"), original)
|
|
4893
|
+
review_gate.record_local_review(anchor, result)
|
|
4894
|
+
assert not receipt_under(os.path.join(tree, "relocated")), "followed a link to the relocated git directory"
|
|
4895
|
+
# A parent component swapped for a link while the directory keeps its inode.
|
|
4896
|
+
parent = os.path.join(root, "parent")
|
|
4897
|
+
os.makedirs(os.path.join(parent, "gd"))
|
|
4898
|
+
anchor = anchor_for(os.path.join(parent, "gd"))
|
|
4899
|
+
os.rename(parent, os.path.join(root, "parent-moved"))
|
|
4900
|
+
os.symlink(os.path.join(root, "parent-moved"), parent)
|
|
4901
|
+
review_gate.record_local_review(anchor, result)
|
|
4902
|
+
assert not receipt_under(os.path.join(root, "parent-moved", "gd")), "followed a linked parent"
|
|
4903
|
+
# Control: an unchanged directory is written.
|
|
4904
|
+
control = os.path.join(root, "control")
|
|
4905
|
+
os.mkdir(control)
|
|
4906
|
+
review_gate.record_local_review(anchor_for(control), result)
|
|
4907
|
+
assert receipt_under(control), "control write missing"
|
|
4908
|
+
print("gitdir_swap_ok")
|
|
4909
|
+
PY
|
|
4910
|
+
)"
|
|
4911
|
+
check "a git directory swapped during the review is never written into" \
|
|
4912
|
+
'[ "$swap_out" = gitdir_swap_ok ]'
|
|
4913
|
+
|
|
4914
|
+
# Repeated failed rooted reads (a linked ancestor, a missing file) must not
|
|
4915
|
+
# leave directory descriptors behind: every omitted contract file is one.
|
|
4916
|
+
fd_out="$(PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY' 2>&1
|
|
4917
|
+
import os, sys, review_gate
|
|
4918
|
+
root = os.path.realpath(os.path.join(sys.argv[1], "fd-leak"))
|
|
4919
|
+
os.makedirs(os.path.join(root, "a", "b"))
|
|
4920
|
+
os.makedirs(os.path.join(root, "target"))
|
|
4921
|
+
os.symlink(os.path.join(root, "target"), os.path.join(root, "a", "linked"))
|
|
4922
|
+
def attempt(relative):
|
|
4923
|
+
try:
|
|
4924
|
+
review_gate.read_bounded_regular_file(relative, root=root, label="probe", maximum=10,
|
|
4925
|
+
regular_error="irregular", oversized_error="big")
|
|
4926
|
+
except review_gate.GateError:
|
|
4927
|
+
pass
|
|
4928
|
+
before = len(os.listdir("/dev/fd"))
|
|
4929
|
+
for _ in range(50):
|
|
4930
|
+
attempt("a/b/missing.md")
|
|
4931
|
+
attempt("a/linked/AGENTS.md")
|
|
4932
|
+
after = len(os.listdir("/dev/fd"))
|
|
4933
|
+
assert after == before, (before, after)
|
|
4934
|
+
print("fd_release_ok")
|
|
4935
|
+
PY
|
|
4936
|
+
)"
|
|
4937
|
+
check "failed rooted reads release every directory descriptor" \
|
|
4938
|
+
'[ "$fd_out" = fd_release_ok ]'
|
|
4939
|
+
|
|
4940
|
+
# A tracked contract file whose ancestor became a link is omitted, and the
|
|
4941
|
+
# linked target's text never reaches the packet.
|
|
4942
|
+
(
|
|
4943
|
+
cd "$contract_repo"
|
|
4944
|
+
git add -A && git commit -q -m "settle earlier cases"
|
|
4945
|
+
mkdir anc hl && printf 'anc rule\n' >anc/AGENTS.md && printf 'before\n' >anc/code.txt
|
|
4946
|
+
printf 'hl rule\n' >hl/AGENTS.md && printf 'before\n' >hl/code.txt
|
|
4947
|
+
git add -A && git commit -q -m "add anc"
|
|
4948
|
+
ln hl/AGENTS.md "$WORK/hl-second-name.md" && printf 'after\n' >hl/code.txt
|
|
4949
|
+
mkdir -p "$WORK/outside-anc" && printf 'outside secret rule\n' >"$WORK/outside-anc/AGENTS.md"
|
|
4950
|
+
printf 'after\n' >"$WORK/outside-anc/code.txt"
|
|
4951
|
+
rm -rf anc && ln -s "$WORK/outside-anc" anc
|
|
4952
|
+
)
|
|
4953
|
+
# Base mode refuses an untracked link as a candidate outright, so the transient
|
|
4954
|
+
# state is reached through a --diff-file packet that names the anc/ paths.
|
|
4955
|
+
git -C "$contract_repo" diff HEAD -- anc hl >"$WORK/anc.patch"
|
|
4956
|
+
reset_case passed unavailable unavailable
|
|
4957
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
4958
|
+
--mode review --cwd "$contract_repo" --diff-file "$WORK/anc.patch" --implementer-family openai \
|
|
4959
|
+
--review-plan-file "$WORK/review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
4960
|
+
check "a contract file behind a linked ancestor or with a second hard link is omitted and never quoted" \
|
|
4961
|
+
'[ "$rc" = 0 ] && ! grep -q "outside secret rule" "$WORK/state/claude_packet" && printf "%s" "$out" | python3 -c "import json,sys; d=json.load(sys.stdin)[\"repository_contract\"]; assert {\"path\": \"anc/AGENTS.md\", \"reason\": \"unreadable\"} in d[\"omitted\"] and {\"path\": \"hl/AGENTS.md\", \"reason\": \"unreadable\"} in d[\"omitted\"], d" && ! grep -q "hl rule" "$WORK/state/claude_packet"'
|
|
4962
|
+
|
|
4543
4963
|
# The explicit cwd is the only repository identity. Ambient Git variables must
|
|
4544
4964
|
# not redirect discovery, objects, refs, index, or worktree to a clean decoy.
|
|
4545
4965
|
git clone -q "$WORK/repo" "$WORK/git-env-decoy"
|