@ccoalm/ccl-skills 0.15.5 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +21 -19
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +113 -123
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/wording-only-review.md +136 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +99 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +297 -100
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +162 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +17 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_order.sh +30 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +445 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +14 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +56 -18
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +41 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +68 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/closeout-reread.md +40 -0
- package/dist/assets/release.json +31 -21
- package/package.json +1 -1
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh
CHANGED
|
@@ -951,7 +951,8 @@ cat >"$WORK/review-plan.json" <<'JSON'
|
|
|
951
951
|
{"concern": "tests_evidence", "conclusion": "Focused deterministic contract tests cover the change.", "evidence_refs": ["e1"]},
|
|
952
952
|
{"concern": "compatibility", "conclusion": "Existing provider routing remains backward compatible.", "evidence_refs": ["e1"]},
|
|
953
953
|
{"concern": "rollout_rollback", "conclusion": "The local CLI change has a direct revert path.", "evidence_refs": ["e1"]},
|
|
954
|
-
{"concern": "observability_operations", "conclusion": "The JSON envelope exposes stage and depth for diagnosis.", "evidence_refs": ["e1"]}
|
|
954
|
+
{"concern": "observability_operations", "conclusion": "The JSON envelope exposes stage and depth for diagnosis.", "evidence_refs": ["e1"]},
|
|
955
|
+
{"concern": "claim_strength", "conclusion": "Each claim is scoped to the fixture it was observed on.", "evidence_refs": ["e1"]}
|
|
955
956
|
],
|
|
956
957
|
"evidence": [
|
|
957
958
|
{"id": "e1", "result": "Deterministic fake-wrapper contract fixture."}
|
|
@@ -1033,7 +1034,8 @@ cat >"$WORK/placeholder-plan.json" <<'JSON'
|
|
|
1033
1034
|
{"concern": "safety", "conclusion": "Packet and tool boundaries remain fail closed.", "evidence_refs": ["e1"]},
|
|
1034
1035
|
{"concern": "failure_paths", "conclusion": "Invalid and inconclusive paths remain terminal.", "evidence_refs": ["e1"]},
|
|
1035
1036
|
{"concern": "tests_evidence", "conclusion": "A focused regression proves filler is rejected.", "evidence_refs": ["e1"]},
|
|
1036
|
-
{"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]}
|
|
1037
|
+
{"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]},
|
|
1038
|
+
{"concern": "claim_strength", "conclusion": "No claim reaches past the filler-rejection fixture.", "evidence_refs": ["e1"]}
|
|
1037
1039
|
],
|
|
1038
1040
|
"evidence": [{"id": "e1", "result": "Deterministic placeholder-validation fixture."}]
|
|
1039
1041
|
}
|
|
@@ -1058,6 +1060,7 @@ cat >"$WORK/high-risk-plan.json" <<'JSON'
|
|
|
1058
1060
|
{"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]},
|
|
1059
1061
|
{"concern": "rollout_rollback", "conclusion": "The local contract change has a direct revert path.", "evidence_refs": ["e1"]},
|
|
1060
1062
|
{"concern": "observability_operations", "conclusion": "The result exposes depth and risk tags for diagnosis.", "evidence_refs": ["e1"]},
|
|
1063
|
+
{"concern": "claim_strength", "conclusion": "Each claim is scoped to the high-risk fixture it was observed on.", "evidence_refs": ["e1"]},
|
|
1061
1064
|
{"concern": "high_risk_boundary", "conclusion": "Bypass attempts cannot remove controller-required concerns.", "evidence_refs": ["e1"]}
|
|
1062
1065
|
],
|
|
1063
1066
|
"evidence": [{"id": "e1", "result": "Deterministic high-risk gate fixture."}]
|
|
@@ -1081,7 +1084,8 @@ cat >"$WORK/near-limit-plan.json" <<JSON
|
|
|
1081
1084
|
{"concern": "safety", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
1082
1085
|
{"concern": "failure_paths", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
1083
1086
|
{"concern": "tests_evidence", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
1084
|
-
{"concern": "compatibility", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]}
|
|
1087
|
+
{"concern": "compatibility", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
1088
|
+
{"concern": "claim_strength", "conclusion": "Scoped to this size fixture.", "evidence_refs": ["e1"]}
|
|
1085
1089
|
],
|
|
1086
1090
|
"evidence": [{"id": "e1", "result": "$large_evidence"}]
|
|
1087
1091
|
}
|
|
@@ -1374,6 +1378,144 @@ for malformed_plan in non-string-owner-review-plan empty-owner-review-plan; do
|
|
|
1374
1378
|
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true'
|
|
1375
1379
|
done
|
|
1376
1380
|
|
|
1381
|
+
# The claim-strength walk is a required concern, so it is owed BEFORE round 1 --
|
|
1382
|
+
# the one point in a round where correcting an overstated claim costs nothing. The
|
|
1383
|
+
# class it covers (absolutes, universals, causal and exhaustiveness claims the cited
|
|
1384
|
+
# evidence does not carry) otherwise keeps arriving as a LATE correction, after the
|
|
1385
|
+
# receipts are bound, where any candidate edit voids them.
|
|
1386
|
+
python3 - "$WORK/review-plan.json" "$WORK/no-claim-strength-review-plan.json" <<'PLAN'
|
|
1387
|
+
import json, sys
|
|
1388
|
+
from pathlib import Path
|
|
1389
|
+
plan = json.loads(Path(sys.argv[1]).read_text())
|
|
1390
|
+
plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != "claim_strength"]
|
|
1391
|
+
Path(sys.argv[2]).write_text(json.dumps(plan))
|
|
1392
|
+
PLAN
|
|
1393
|
+
reset_case passed unavailable unavailable
|
|
1394
|
+
out="$(run_gate --review-plan-file "$WORK/no-claim-strength-review-plan.json")"; rc=$?
|
|
1395
|
+
check "a plan that skips the claim-strength walk fails before any provider runs" \
|
|
1396
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true self_review_gate.required_triggers.0=before_external_review'
|
|
1397
|
+
|
|
1398
|
+
reset_case passed unavailable unavailable
|
|
1399
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1400
|
+
check "the build reviewer is asked to check claim strength" \
|
|
1401
|
+
'[ "$rc" = 0 ] && json_fields "$out" reviewed_concerns.5=claim_strength'
|
|
1402
|
+
|
|
1403
|
+
# The exported list is only worth deriving from if it IS the enforced one. Build a
|
|
1404
|
+
# plan covering exactly what the controller prints, and then drop each printed
|
|
1405
|
+
# concern in turn: acceptance proves the print covers everything the gate demands,
|
|
1406
|
+
# and every single-drop rejection proves nothing printed is decorative. Without
|
|
1407
|
+
# both directions a caller could derive from a list that had quietly diverged --
|
|
1408
|
+
# which is the drift this export exists to remove.
|
|
1409
|
+
# The exit status is asserted too. This suite runs without errexit, so a command
|
|
1410
|
+
# substitution silently discards it: a printer that emits the right concerns and then
|
|
1411
|
+
# fails would satisfy a non-empty check and report agreement it never reached.
|
|
1412
|
+
printed_rc=0
|
|
1413
|
+
printed_concerns="$("$DIR/review_gate.sh" --print-required-concerns --stage build)" || printed_rc=$?
|
|
1414
|
+
check "the controller can print the concern set a plan owes, and succeeds doing it" \
|
|
1415
|
+
'[ -n "$printed_concerns" ] && [ "$printed_rc" = 0 ]'
|
|
1416
|
+
# Proving agreement at ONE depth leaves the other branch free to diverge with every
|
|
1417
|
+
# test green -- and release/high-risk is the branch that carries the most concerns.
|
|
1418
|
+
# Assert the depth-raising branch answers what the gate itself derives for it.
|
|
1419
|
+
# Parity includes what each side REFUSES. A printer that answers for tags the enforcer
|
|
1420
|
+
# rejects reintroduces the divergence this export removes: a caller deriving from a
|
|
1421
|
+
# malformed tag would get a list where the real round fails closed.
|
|
1422
|
+
for bad_tag in "a b" "" "$(printf 'x%.0s' $(seq 81))"; do
|
|
1423
|
+
# rc captured without touching shell options: this suite runs under `set -uo pipefail`
|
|
1424
|
+
# and enabling errexit here would abort every later case at its first non-zero command.
|
|
1425
|
+
bad_tag_rc=0
|
|
1426
|
+
"$DIR/review_gate.sh" --print-required-concerns --stage build --risk-tag "$bad_tag" >/dev/null 2>&1 || bad_tag_rc=$?
|
|
1427
|
+
# Parity is a claim about TWO sides, so both are exercised: asserting only the
|
|
1428
|
+
# printer would keep these checks green if the enforcer's own rejection were
|
|
1429
|
+
# removed, which is the half this pair exists to tie together.
|
|
1430
|
+
enforcer_tag_rc=0
|
|
1431
|
+
run_gate --risk-tag "$bad_tag" >/dev/null 2>&1 || enforcer_tag_rc=$?
|
|
1432
|
+
check "printer and enforcer both refuse the same malformed risk tag (${#bad_tag} chars)" \
|
|
1433
|
+
'[ "$bad_tag_rc" != 0 ] && [ "$enforcer_tag_rc" != 0 ]'
|
|
1434
|
+
done
|
|
1435
|
+
printed_release_rc=0
|
|
1436
|
+
printed_release="$("$DIR/review_gate.sh" --print-required-concerns --stage explore --risk-tag shared-gate)" || printed_release_rc=$?
|
|
1437
|
+
check "the raised-depth print succeeds" '[ "$printed_release_rc" = 0 ]'
|
|
1438
|
+
# Hoisted for the same reason as the calls above: nested inside the comparison, this
|
|
1439
|
+
# printer call's exit status was discarded, so a regression failing only for explicit
|
|
1440
|
+
# release depth would have compared equal and passed. Every printer invocation in this
|
|
1441
|
+
# suite now has its status asserted.
|
|
1442
|
+
printed_plain_release_rc=0
|
|
1443
|
+
printed_plain_release="$("$DIR/review_gate.sh" --print-required-concerns --stage release)" || printed_plain_release_rc=$?
|
|
1444
|
+
check "the plain release print succeeds" '[ "$printed_plain_release_rc" = 0 ]'
|
|
1445
|
+
check "a high-risk tag raises the printed set to release depth and adds the boundary concern" \
|
|
1446
|
+
'[ "$(printf %s "$printed_release" | tr "\n" " ")" = "$(printf "%s\nhigh_risk_boundary" "$printed_plain_release" | tr "\n" " ")" ]'
|
|
1447
|
+
python3 - "$WORK/review-plan.json" "$WORK/printed-plan.json" $printed_concerns <<'PLAN'
|
|
1448
|
+
import json, sys
|
|
1449
|
+
from pathlib import Path
|
|
1450
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
1451
|
+
printed = sys.argv[3:]
|
|
1452
|
+
by_concern = {row["concern"]: row for row in source["self_review"]}
|
|
1453
|
+
source["self_review"] = [
|
|
1454
|
+
by_concern.get(concern, {"concern": concern,
|
|
1455
|
+
"conclusion": f"The fixture covers {concern}.",
|
|
1456
|
+
"evidence_refs": ["e1"]})
|
|
1457
|
+
for concern in printed
|
|
1458
|
+
]
|
|
1459
|
+
Path(sys.argv[2]).write_text(json.dumps(source))
|
|
1460
|
+
PLAN
|
|
1461
|
+
reset_case passed unavailable unavailable
|
|
1462
|
+
out="$(run_gate --review-plan-file "$WORK/printed-plan.json" --allow-fallback-egress)"; rc=$?
|
|
1463
|
+
check "a plan built from the printed set satisfies the gate" '[ "$rc" = 0 ]'
|
|
1464
|
+
printed_drop_failures=0
|
|
1465
|
+
for dropped in $printed_concerns; do
|
|
1466
|
+
python3 - "$WORK/printed-plan.json" "$WORK/printed-plan-minus.json" "$dropped" <<'PLAN'
|
|
1467
|
+
import json, sys
|
|
1468
|
+
from pathlib import Path
|
|
1469
|
+
plan = json.loads(Path(sys.argv[1]).read_text())
|
|
1470
|
+
plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != sys.argv[3]]
|
|
1471
|
+
Path(sys.argv[2]).write_text(json.dumps(plan))
|
|
1472
|
+
PLAN
|
|
1473
|
+
reset_case passed unavailable unavailable
|
|
1474
|
+
out="$(run_gate --review-plan-file "$WORK/printed-plan-minus.json")"; rc=$?
|
|
1475
|
+
if [ "$rc" = 2 ] && json_fields "$out" reason_code=self_review_incomplete; then
|
|
1476
|
+
printed_drop_failures=$((printed_drop_failures+1))
|
|
1477
|
+
fi
|
|
1478
|
+
done
|
|
1479
|
+
check "every printed concern is one the gate actually demands" \
|
|
1480
|
+
'[ "$printed_drop_failures" = "$(printf %s "$printed_concerns" | wc -w | tr -d " ")" ]'
|
|
1481
|
+
|
|
1482
|
+
# The same two directions at the OTHER depth. A printer that agreed with the gate at
|
|
1483
|
+
# build and diverged at release/high-risk would keep every test above green, and
|
|
1484
|
+
# release is the branch carrying the most concerns.
|
|
1485
|
+
python3 - "$WORK/high-risk-plan.json" "$WORK/printed-release-plan.json" $printed_release <<'PLAN'
|
|
1486
|
+
import json, sys
|
|
1487
|
+
from pathlib import Path
|
|
1488
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
1489
|
+
by_concern = {row["concern"]: row for row in source["self_review"]}
|
|
1490
|
+
source["self_review"] = [
|
|
1491
|
+
by_concern.get(concern, {"concern": concern,
|
|
1492
|
+
"conclusion": f"The high-risk fixture covers {concern}.",
|
|
1493
|
+
"evidence_refs": ["e1"]})
|
|
1494
|
+
for concern in sys.argv[3:]
|
|
1495
|
+
]
|
|
1496
|
+
Path(sys.argv[2]).write_text(json.dumps(source))
|
|
1497
|
+
PLAN
|
|
1498
|
+
reset_case passed unavailable unavailable
|
|
1499
|
+
out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/printed-release-plan.json" --review-chain-id printed-release --autonomous-review-index 1 --allow-fallback-egress)"; rc=$?
|
|
1500
|
+
check "a plan built from the printed release set satisfies the raised-depth gate" '[ "$rc" = 0 ]'
|
|
1501
|
+
printed_release_drop_failures=0
|
|
1502
|
+
for dropped in $printed_release; do
|
|
1503
|
+
python3 - "$WORK/printed-release-plan.json" "$WORK/printed-release-minus.json" "$dropped" <<'PLAN'
|
|
1504
|
+
import json, sys
|
|
1505
|
+
from pathlib import Path
|
|
1506
|
+
plan = json.loads(Path(sys.argv[1]).read_text())
|
|
1507
|
+
plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != sys.argv[3]]
|
|
1508
|
+
Path(sys.argv[2]).write_text(json.dumps(plan))
|
|
1509
|
+
PLAN
|
|
1510
|
+
reset_case passed unavailable unavailable
|
|
1511
|
+
out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/printed-release-minus.json" --review-chain-id printed-release-minus --autonomous-review-index 1)"; rc=$?
|
|
1512
|
+
if [ "$rc" = 2 ] && json_fields "$out" reason_code=self_review_incomplete; then
|
|
1513
|
+
printed_release_drop_failures=$((printed_release_drop_failures+1))
|
|
1514
|
+
fi
|
|
1515
|
+
done
|
|
1516
|
+
check "every printed release concern is one the raised-depth gate actually demands" \
|
|
1517
|
+
'[ "$printed_release_drop_failures" = "$(printf %s "$printed_release" | wc -w | tr -d " ")" ]'
|
|
1518
|
+
|
|
1377
1519
|
owner_lstat_classification="$(python3 - "$DIR/review_gate.py" <<'PY'
|
|
1378
1520
|
import errno
|
|
1379
1521
|
import importlib.util
|
|
@@ -1679,7 +1821,7 @@ diff_alternate = root / "alternate.patch"
|
|
|
1679
1821
|
diff_source.write_bytes(original_diff)
|
|
1680
1822
|
diff_alternate.write_bytes(alternate_diff)
|
|
1681
1823
|
with replace_after_symlink_check(diff_source, diff_alternate):
|
|
1682
|
-
packet_path, digest, _, _ = review_gate.freeze_packet(
|
|
1824
|
+
packet_path, digest, _candidate, _n, _, _ = review_gate.freeze_packet(
|
|
1683
1825
|
SimpleNamespace(
|
|
1684
1826
|
cwd=str(root), diff_file=str(diff_source), base=None, paths=[]
|
|
1685
1827
|
),
|
|
@@ -1774,6 +1916,241 @@ file_input_race_rc=$?
|
|
|
1774
1916
|
check "diff, prior, and completion inputs are read once from a bounded opened descriptor" \
|
|
1775
1917
|
'[ "$file_input_race_rc" = 0 ] && [ "$file_input_race_probe" = open_once_file_inputs_ok ]'
|
|
1776
1918
|
|
|
1919
|
+
# The reviewer's packet and the landing candidate are two objects. A widened
|
|
1920
|
+
# packet exists so a reviewer can judge a claim against code outside the diff;
|
|
1921
|
+
# the candidate exists so the merge-side binder can recompute what actually
|
|
1922
|
+
# lands. Aliasing them made the two mutually exclusive: widening produced a
|
|
1923
|
+
# receipt the binder could never match. These assert the split and the one
|
|
1924
|
+
# invariant that replaces the equality -- the candidate appears in the packet
|
|
1925
|
+
# verbatim, so nothing lands that its reviewer did not read.
|
|
1926
|
+
subject_packet_probe="$(
|
|
1927
|
+
PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY'
|
|
1928
|
+
import hashlib
|
|
1929
|
+
import subprocess
|
|
1930
|
+
import sys
|
|
1931
|
+
import time
|
|
1932
|
+
from pathlib import Path
|
|
1933
|
+
from types import SimpleNamespace
|
|
1934
|
+
|
|
1935
|
+
import review_gate
|
|
1936
|
+
|
|
1937
|
+
root = Path(sys.argv[1]) / "subject-vs-packet"
|
|
1938
|
+
root.mkdir()
|
|
1939
|
+
# Packet files live OUTSIDE the repository on purpose: the base-derived
|
|
1940
|
+
# candidate includes untracked files, so a packet written into the worktree
|
|
1941
|
+
# would become part of the very candidate it has to contain.
|
|
1942
|
+
outside = Path(sys.argv[1]) / "subject-vs-packet-packets"
|
|
1943
|
+
outside.mkdir()
|
|
1944
|
+
|
|
1945
|
+
|
|
1946
|
+
def git(*args):
|
|
1947
|
+
subprocess.run(
|
|
1948
|
+
["git", "-C", str(root), *args],
|
|
1949
|
+
check=True,
|
|
1950
|
+
stdout=subprocess.DEVNULL,
|
|
1951
|
+
stderr=subprocess.DEVNULL,
|
|
1952
|
+
)
|
|
1953
|
+
|
|
1954
|
+
|
|
1955
|
+
git("init", "-q")
|
|
1956
|
+
git("config", "user.email", "fixture@example.invalid")
|
|
1957
|
+
git("config", "user.name", "fixture")
|
|
1958
|
+
(root / "landing.txt").write_text("old\n", encoding="utf-8")
|
|
1959
|
+
(root / "context.txt").write_text("context-base\n", encoding="utf-8")
|
|
1960
|
+
git("add", "landing.txt", "context.txt")
|
|
1961
|
+
git("commit", "-qm", "base")
|
|
1962
|
+
base = subprocess.run(
|
|
1963
|
+
["git", "-C", str(root), "rev-parse", "HEAD"],
|
|
1964
|
+
check=True,
|
|
1965
|
+
capture_output=True,
|
|
1966
|
+
text=True,
|
|
1967
|
+
).stdout.strip()
|
|
1968
|
+
(root / "landing.txt").write_text("new\n", encoding="utf-8")
|
|
1969
|
+
|
|
1970
|
+
|
|
1971
|
+
def freeze(*, diff_file=None, paths=(), base_ref=base, wording=None):
|
|
1972
|
+
return review_gate.freeze_packet(
|
|
1973
|
+
SimpleNamespace(
|
|
1974
|
+
cwd=str(root),
|
|
1975
|
+
diff_file=str(diff_file) if diff_file else None,
|
|
1976
|
+
base=base_ref,
|
|
1977
|
+
paths=list(paths),
|
|
1978
|
+
wording_only_proof_file=wording,
|
|
1979
|
+
),
|
|
1980
|
+
time.monotonic() + 30,
|
|
1981
|
+
)
|
|
1982
|
+
|
|
1983
|
+
|
|
1984
|
+
def digest(value: bytes) -> str:
|
|
1985
|
+
return hashlib.sha256(value).hexdigest()
|
|
1986
|
+
|
|
1987
|
+
|
|
1988
|
+
def expect_refused(label, **kwargs):
|
|
1989
|
+
try:
|
|
1990
|
+
result = freeze(**kwargs)
|
|
1991
|
+
except review_gate.GateError as exc:
|
|
1992
|
+
return exc
|
|
1993
|
+
result[0].unlink()
|
|
1994
|
+
raise AssertionError(label)
|
|
1995
|
+
|
|
1996
|
+
|
|
1997
|
+
# The base-derived subject: exactly what the landing binder recomputes.
|
|
1998
|
+
subject_path, subject_hash, subject_candidate_hash, subject_n, subject_paths, _ = freeze()
|
|
1999
|
+
subject_bytes = subject_path.read_bytes()
|
|
2000
|
+
subject_path.unlink()
|
|
2001
|
+
assert subject_hash == digest(subject_bytes)
|
|
2002
|
+
assert subject_candidate_hash == subject_hash
|
|
2003
|
+
assert subject_n == len(subject_bytes)
|
|
2004
|
+
assert subject_paths == ["landing.txt"], subject_paths
|
|
2005
|
+
|
|
2006
|
+
# A7 -- with no --diff-file the two hashes are the same value, as they are today.
|
|
2007
|
+
plain_path, plain_packet_hash, plain_candidate_hash, _plain_n, plain_paths, _ = freeze()
|
|
2008
|
+
plain_path.unlink()
|
|
2009
|
+
assert plain_packet_hash == subject_hash
|
|
2010
|
+
assert plain_candidate_hash == subject_hash
|
|
2011
|
+
assert plain_paths == ["landing.txt"], plain_paths
|
|
2012
|
+
|
|
2013
|
+
# A1/A2/A3 -- a widened packet carries the whole subject plus context the
|
|
2014
|
+
# reviewer needs. The packet hash is the widened bytes; the candidate hash is
|
|
2015
|
+
# still the base-derived subject the binder will recompute.
|
|
2016
|
+
context = (
|
|
2017
|
+
b"\n--- context: skills/code-review/SKILL.md (unchanged, for judgment) ---\n"
|
|
2018
|
+
b"the sibling clause the changed lines must not contradict\n"
|
|
2019
|
+
)
|
|
2020
|
+
widened = outside / "widened.patch"
|
|
2021
|
+
widened.write_bytes(subject_bytes + context)
|
|
2022
|
+
wide_path, wide_packet_hash, wide_candidate_hash, wide_n, wide_paths, _ = freeze(
|
|
2023
|
+
diff_file=widened
|
|
2024
|
+
)
|
|
2025
|
+
try:
|
|
2026
|
+
assert wide_packet_hash == digest(subject_bytes + context)
|
|
2027
|
+
assert wide_candidate_hash == subject_hash
|
|
2028
|
+
assert wide_packet_hash != wide_candidate_hash
|
|
2029
|
+
# The reviewer is told where the candidate ends; without that, appended
|
|
2030
|
+
# hunks that continue or appear to revert the diff are indistinguishable
|
|
2031
|
+
# from candidate content in a packet-bounded read.
|
|
2032
|
+
assert wide_n == len(subject_bytes), wide_n
|
|
2033
|
+
# A10 -- candidate paths follow the subject, not the packet, so owner
|
|
2034
|
+
# selection and the wording-only changed-file comparison stay bound to what
|
|
2035
|
+
# lands rather than to whatever context was appended.
|
|
2036
|
+
assert wide_paths == ["landing.txt"], wide_paths
|
|
2037
|
+
finally:
|
|
2038
|
+
wide_path.unlink()
|
|
2039
|
+
|
|
2040
|
+
# A5 -- a packet missing part of the candidate is refused. This is the property
|
|
2041
|
+
# the equality used to provide for free.
|
|
2042
|
+
truncated = outside / "truncated.patch"
|
|
2043
|
+
truncated.write_bytes(subject_bytes[: len(subject_bytes) // 2] + context)
|
|
2044
|
+
exc = expect_refused(
|
|
2045
|
+
"a packet missing part of the candidate was accepted", diff_file=truncated
|
|
2046
|
+
)
|
|
2047
|
+
assert "BEGIN" in str(exc), str(exc)
|
|
2048
|
+
|
|
2049
|
+
# A6 -- context appended passes; context spliced into the middle of the
|
|
2050
|
+
# candidate does not, because then the candidate is no longer in the packet
|
|
2051
|
+
# verbatim and no cheap check can tell a splice from a silent edit.
|
|
2052
|
+
split = len(subject_bytes) // 2
|
|
2053
|
+
interleaved = outside / "interleaved.patch"
|
|
2054
|
+
interleaved.write_bytes(subject_bytes[:split] + context + subject_bytes[split:])
|
|
2055
|
+
expect_refused(
|
|
2056
|
+
"a packet interleaving context inside the candidate was accepted",
|
|
2057
|
+
diff_file=interleaved,
|
|
2058
|
+
)
|
|
2059
|
+
|
|
2060
|
+
# A11 -- context BEFORE the candidate is refused even though the candidate is
|
|
2061
|
+
# present verbatim. A bare containment test accepts this, and an adversarial
|
|
2062
|
+
# round showed what it buys: a sanitized decoy diff read as the change while the
|
|
2063
|
+
# real candidate reads as trailing context.
|
|
2064
|
+
prepended = outside / "prepended.patch"
|
|
2065
|
+
prepended.write_bytes(context + subject_bytes)
|
|
2066
|
+
exc = expect_refused(
|
|
2067
|
+
"a packet preceding the candidate with other content was accepted",
|
|
2068
|
+
diff_file=prepended,
|
|
2069
|
+
)
|
|
2070
|
+
assert "BEGIN" in str(exc), str(exc)
|
|
2071
|
+
|
|
2072
|
+
# A4 -- a packet with no relation to the candidate is refused.
|
|
2073
|
+
unrelated = outside / "unrelated.patch"
|
|
2074
|
+
unrelated.write_bytes(b"diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n")
|
|
2075
|
+
expect_refused("an unrelated packet was accepted", diff_file=unrelated)
|
|
2076
|
+
|
|
2077
|
+
# A8 -- --diff-file alone keeps today's meaning: no base, so no subject, and
|
|
2078
|
+
# the candidate hash stays the packet's own hash.
|
|
2079
|
+
alone_path, alone_packet_hash, alone_candidate_hash, _n, _, _ = freeze(
|
|
2080
|
+
diff_file=widened, base_ref=None
|
|
2081
|
+
)
|
|
2082
|
+
alone_path.unlink()
|
|
2083
|
+
assert alone_packet_hash == digest(subject_bytes + context)
|
|
2084
|
+
assert alone_candidate_hash == alone_packet_hash
|
|
2085
|
+
|
|
2086
|
+
# A9 -- the wording-only proof is a machine check over a full-context
|
|
2087
|
+
# base-derived diff and has no meaning over an author-assembled packet.
|
|
2088
|
+
proof = outside / "wording-only.json"
|
|
2089
|
+
proof.write_text("{}", encoding="utf-8")
|
|
2090
|
+
expect_refused(
|
|
2091
|
+
"a wording-only proof was accepted over an author-assembled packet",
|
|
2092
|
+
diff_file=widened,
|
|
2093
|
+
wording=str(proof),
|
|
2094
|
+
)
|
|
2095
|
+
# ... in the COMBINED form. Bare --diff-file with a wording-only proof stays
|
|
2096
|
+
# accepted, which the cases above this block exercise throughout; the boundary
|
|
2097
|
+
# is where a base-derived candidate and author-assembled bytes would both be in
|
|
2098
|
+
# play with nothing saying which one the proof's scope describes.
|
|
2099
|
+
result = freeze(diff_file=widened, base_ref=None, wording=str(proof))
|
|
2100
|
+
result[0].unlink()
|
|
2101
|
+
|
|
2102
|
+
# A12 -- an empty base-derived candidate is refused rather than trivially
|
|
2103
|
+
# satisfying the prefix check, which every packet does for empty bytes.
|
|
2104
|
+
empty_repo = Path(sys.argv[1]) / "empty-candidate"
|
|
2105
|
+
empty_repo.mkdir()
|
|
2106
|
+
subprocess.run(["git", "-C", str(empty_repo), "init", "-q"], check=True)
|
|
2107
|
+
subprocess.run(
|
|
2108
|
+
["git", "-C", str(empty_repo), "config", "user.email", "fixture@example.invalid"],
|
|
2109
|
+
check=True,
|
|
2110
|
+
)
|
|
2111
|
+
subprocess.run(
|
|
2112
|
+
["git", "-C", str(empty_repo), "config", "user.name", "fixture"], check=True
|
|
2113
|
+
)
|
|
2114
|
+
(empty_repo / "kept.txt").write_text("unchanged\n", encoding="utf-8")
|
|
2115
|
+
subprocess.run(
|
|
2116
|
+
["git", "-C", str(empty_repo), "add", "kept.txt"],
|
|
2117
|
+
check=True,
|
|
2118
|
+
stdout=subprocess.DEVNULL,
|
|
2119
|
+
)
|
|
2120
|
+
subprocess.run(
|
|
2121
|
+
["git", "-C", str(empty_repo), "commit", "-qm", "base"],
|
|
2122
|
+
check=True,
|
|
2123
|
+
stdout=subprocess.DEVNULL,
|
|
2124
|
+
)
|
|
2125
|
+
empty_base = subprocess.run(
|
|
2126
|
+
["git", "-C", str(empty_repo), "rev-parse", "HEAD"],
|
|
2127
|
+
check=True,
|
|
2128
|
+
capture_output=True,
|
|
2129
|
+
text=True,
|
|
2130
|
+
).stdout.strip()
|
|
2131
|
+
try:
|
|
2132
|
+
review_gate.freeze_packet(
|
|
2133
|
+
SimpleNamespace(
|
|
2134
|
+
cwd=str(empty_repo),
|
|
2135
|
+
diff_file=str(widened),
|
|
2136
|
+
base=empty_base,
|
|
2137
|
+
paths=[],
|
|
2138
|
+
wording_only_proof_file=None,
|
|
2139
|
+
),
|
|
2140
|
+
time.monotonic() + 30,
|
|
2141
|
+
)
|
|
2142
|
+
except review_gate.GateError as exc:
|
|
2143
|
+
assert exc.reason_code == "empty_diff", exc.reason_code
|
|
2144
|
+
else:
|
|
2145
|
+
raise AssertionError("an empty base-derived candidate was accepted")
|
|
2146
|
+
|
|
2147
|
+
print("subject_packet_split_ok")
|
|
2148
|
+
PY
|
|
2149
|
+
)"
|
|
2150
|
+
subject_packet_rc=$?
|
|
2151
|
+
check "a widened packet keeps the base-derived candidate and must contain it verbatim" \
|
|
2152
|
+
'[ "$subject_packet_rc" = 0 ] && [ "$subject_packet_probe" = subject_packet_split_ok ]'
|
|
2153
|
+
|
|
1777
2154
|
reset_case missing_coverage passed unavailable
|
|
1778
2155
|
out="$(run_gate --diff-file "$WORK/secret-diff.patch")"; rc=$?
|
|
1779
2156
|
check "missing coverage cannot widen egress for a secret-bearing diff without approval" \
|
|
@@ -1838,7 +2215,7 @@ check "complete but placeholder concern conclusions cannot false-green" \
|
|
|
1838
2215
|
reset_case findings unavailable unavailable
|
|
1839
2216
|
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1840
2217
|
check "Claude findings remain findings" \
|
|
1841
|
-
'[ "$rc" = 0 ] && json_fields "$out" status=findings selected_client=claude next_action=triage_findings_and_continue_independent_work autonomous_review_budget=1 autonomous_review_index=1 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim self_review_gate.allowed_next_actions.0=deep_self_review self_review_gate.allowed_next_actions.1=continue_implementation self_review_gate.allowed_next_actions.2=continue_independent_work'
|
|
2218
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings selected_client=claude next_action=triage_findings_and_continue_independent_work autonomous_review_budget=1 autonomous_review_index=1 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim self_review_gate.allowed_next_actions.0=deep_self_review self_review_gate.allowed_next_actions.1=continue_implementation self_review_gate.allowed_next_actions.2=continue_independent_work && ! grep -q recurring_findings_design_check <<<"$out"'
|
|
1842
2219
|
|
|
1843
2220
|
reset_case quota passed unavailable
|
|
1844
2221
|
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
@@ -2116,7 +2493,7 @@ check "an initial review with challenge capacity requires a tracked Agent chain"
|
|
|
2116
2493
|
reset_case passed unavailable unavailable
|
|
2117
2494
|
out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/high-risk-plan.json" --review-chain-id high-risk-task --autonomous-review-index 1)"; rc=$?
|
|
2118
2495
|
check "high-risk tags raise explore to release depth and default one challenge" \
|
|
2119
|
-
'[ "$rc" = 0 ] && json_fields "$out" stage=explore stage_source=caller-declared review_depth=release risk_tags_source=caller-declared challenge_budget=1 challenge_rounds_remaining=1 review_chain_tracked=true review_chain_id=high-risk-task autonomous_review_budget=2 autonomous_review_index=1 autonomous_reviews_remaining=1 autonomous_review_allowed=true next_action=run_challenge completion_gated=true risk_tags.0=shared-gate reviewed_concerns.
|
|
2496
|
+
'[ "$rc" = 0 ] && json_fields "$out" stage=explore stage_source=caller-declared review_depth=release risk_tags_source=caller-declared challenge_budget=1 challenge_rounds_remaining=1 review_chain_tracked=true review_chain_id=high-risk-task autonomous_review_budget=2 autonomous_review_index=1 autonomous_reviews_remaining=1 autonomous_review_allowed=true next_action=run_challenge completion_gated=true risk_tags.0=shared-gate reviewed_concerns.8=high_risk_boundary self_review_gate.required=false self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.satisfied_triggers.1=risk_or_scope_escalation'
|
|
2120
2497
|
|
|
2121
2498
|
# Release/high-risk normally requires a challenge. The only single-review
|
|
2122
2499
|
# exception is a candidate-bound deterministic wording-only proof whose result
|
|
@@ -2925,7 +3302,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
2925
3302
|
--wording-only-proof-file "$WORK/wording-punctuation-proof.json")"; rc=$?
|
|
2926
3303
|
punctuation_proof_hash="$(shasum -a 256 "$WORK/wording-punctuation-proof.json" | awk '{print $1}')"
|
|
2927
3304
|
check "release wording-only punctuation scope can take one proof-bound review" \
|
|
2928
|
-
'[ "$rc" = 0 ] && json_fields "$out" challenge_budget=0 review_chain_tracked=false wording_only_scope.status=passed wording_only_scope.check_kind=markdown-punctuation-only wording_only_proof_sha256="$punctuation_proof_hash" reviewed_concerns.
|
|
3305
|
+
'[ "$rc" = 0 ] && json_fields "$out" challenge_budget=0 review_chain_tracked=false wording_only_scope.status=passed wording_only_scope.check_kind=markdown-punctuation-only wording_only_proof_sha256="$punctuation_proof_hash" reviewed_concerns.8=wording_only_boundary'
|
|
2929
3306
|
|
|
2930
3307
|
reset_case passed unavailable unavailable
|
|
2931
3308
|
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
@@ -2934,7 +3311,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
2934
3311
|
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
2935
3312
|
--wording-only-proof-file "$WORK/wording-punctuation-proof.json")"; rc=$?
|
|
2936
3313
|
check "build wording-only review records the same controller-bound proof" \
|
|
2937
|
-
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-punctuation-only reviewed_concerns.
|
|
3314
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-punctuation-only reviewed_concerns.6=wording_only_boundary'
|
|
2938
3315
|
printf '%s\n' "$out" >"$WORK/wording-punctuation-review.json"
|
|
2939
3316
|
reset_case passed unavailable unavailable
|
|
2940
3317
|
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
@@ -3010,7 +3387,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
3010
3387
|
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
3011
3388
|
--wording-only-proof-file "$WORK/wording-token-proof.json")"; rc=$?
|
|
3012
3389
|
check "build exact typo replacement can take one proof-bound review" \
|
|
3013
|
-
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.old_token=teh wording_only_scope.new_token=the wording_only_scope.expected_count=1 wording_only_scope.replacement_count=1 reviewed_concerns.
|
|
3390
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.old_token=teh wording_only_scope.new_token=the wording_only_scope.expected_count=1 wording_only_scope.replacement_count=1 reviewed_concerns.6=wording_only_boundary'
|
|
3014
3391
|
|
|
3015
3392
|
reset_case passed unavailable unavailable
|
|
3016
3393
|
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
@@ -3037,7 +3414,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
3037
3414
|
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
3038
3415
|
--wording-only-proof-file "$WORK/wording-base-proof.json")"; rc=$?
|
|
3039
3416
|
check "base-mode build wording proof freezes full context from line one" \
|
|
3040
|
-
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.replacement_count=1 reviewed_concerns.
|
|
3417
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_depth=build wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.replacement_count=1 reviewed_concerns.6=wording_only_boundary'
|
|
3041
3418
|
|
|
3042
3419
|
for rejected_scope in multi-skill-wording truncated-context-wording symlink-mode-wording frontmatter-shift-insert-wording frontmatter-shift-delete-wording no-final-newline-wording invalid-octal-wording huge-hunk-number-wording zero-width-wording bidi-control-wording emoji-symbol-wording currency-symbol-wording decomposed-boundary-wording zwj-boundary-wording; do
|
|
3043
3420
|
reset_case passed unavailable unavailable
|
|
@@ -3264,6 +3641,13 @@ printf '%s\n' "$passed_round_one" >"$WORK/passed-round-one.json"
|
|
|
3264
3641
|
check "a passed first tracked round still owes its challenge before completion" \
|
|
3265
3642
|
'[ "$passed_round_one_rc" = 0 ] && json_fields "$passed_round_one" status=passed autonomous_review_index=1 autonomous_reviews_remaining=2 autonomous_review_allowed=true next_action=run_challenge completion_gated=true'
|
|
3266
3643
|
|
|
3644
|
+
# Control leg for the recurrence trigger: same shape, first findings round. Without it a
|
|
3645
|
+
# probe that fires for an unrelated reason would read as the recurrence being detected.
|
|
3646
|
+
reset_case findings unavailable unavailable
|
|
3647
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus passed-prior-findings --review-chain-id passed-task --autonomous-review-index 2 --prior-review-result-file "$WORK/passed-round-one.json")"; rc=$?
|
|
3648
|
+
check "findings after a clean prior round stay a first findings round" \
|
|
3649
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings self_review_gate.required_triggers.0=findings_returned && ! grep -q recurring_findings_design_check <<<"$out"'
|
|
3650
|
+
|
|
3267
3651
|
# Chain succession. A fix that touches the owner package moves selected_skills_sha256
|
|
3268
3652
|
# and ends the chain by design, so the post-fix candidate can never be challenged
|
|
3269
3653
|
# inside it. Succession opens ONE new chain whose first Agent round is a challenge,
|
|
@@ -3287,7 +3671,8 @@ python3 - "$WORK/succ-round-two.json" \
|
|
|
3287
3671
|
"$WORK/succ-predecessor-owner-moved.json" \
|
|
3288
3672
|
"$WORK/succ-predecessor-forged-controller.json" \
|
|
3289
3673
|
"$WORK/succ-predecessor-foreign-scope.json" \
|
|
3290
|
-
"$WORK/succ-predecessor-forged-terminal.json"
|
|
3674
|
+
"$WORK/succ-predecessor-forged-terminal.json" \
|
|
3675
|
+
"$WORK/succ-predecessor-complete-mode.json" <<'PY'
|
|
3291
3676
|
import json
|
|
3292
3677
|
from pathlib import Path
|
|
3293
3678
|
import sys
|
|
@@ -3313,6 +3698,10 @@ forged_terminal["challenge_index"] = 0
|
|
|
3313
3698
|
forged_terminal["autonomous_reviews_remaining"] = 1
|
|
3314
3699
|
forged_terminal["autonomous_review_allowed"] = True
|
|
3315
3700
|
Path(sys.argv[5]).write_text(json.dumps(forged_terminal, separators=(",", ":")))
|
|
3701
|
+
# Neither lane: a completion checkpoint is not a round the succession may carry.
|
|
3702
|
+
complete_mode = json.loads(json.dumps(source))
|
|
3703
|
+
complete_mode["mode"] = "complete"
|
|
3704
|
+
Path(sys.argv[6]).write_text(json.dumps(complete_mode, separators=(",", ":")))
|
|
3316
3705
|
PY
|
|
3317
3706
|
|
|
3318
3707
|
reset_case passed unavailable unavailable
|
|
@@ -3325,15 +3714,53 @@ out="$(run_challenge_gate --focus owner-moved --review-chain-id succ-owner-moved
|
|
|
3325
3714
|
check "a succession accepts the owner-package hash move that ended the prior chain" \
|
|
3326
3715
|
'[ "$rc" = 0 ] && json_fields "$out" mode=challenge predecessor_chain_id=succ-phase-one'
|
|
3327
3716
|
|
|
3717
|
+
# Budget is one review plus one challenge, so a fix ends the chain and the SECOND
|
|
3718
|
+
# findings round usually lands in the SUCCEEDING chain. Counting only in-chain rounds
|
|
3719
|
+
# would therefore never see the recurrence the trigger exists for.
|
|
3720
|
+
reset_case findings unavailable unavailable
|
|
3721
|
+
out="$(run_challenge_gate --focus recurring-findings --review-chain-id succ-recurrence --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-round-two.json")"; rc=$?
|
|
3722
|
+
check "findings after a predecessor chain that also returned findings raise the design check" \
|
|
3723
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings self_review_gate.required_triggers.0=findings_returned && grep -q recurring_findings_design_check <<<"$out"'
|
|
3724
|
+
|
|
3328
3725
|
reset_case passed unavailable unavailable
|
|
3329
3726
|
out="$(run_challenge_gate --focus no-predecessor --review-chain-id succ-orphan --autonomous-review-index 1)"; rc=$?
|
|
3330
3727
|
check "a tracked challenge cannot open a chain without a predecessor receipt" \
|
|
3331
3728
|
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession"*) false;; *) true;; esac'
|
|
3332
3729
|
|
|
3730
|
+
# A fix applied straight after the REVIEW ends the chain exactly as a fix after the
|
|
3731
|
+
# challenge does -- the owner digest moves either way -- so the ended chain's terminal
|
|
3732
|
+
# receipt is its review. Requiring a challenge receipt here forced that challenge to be
|
|
3733
|
+
# spent on a candidate the author had already decided to change, and bought no evidence
|
|
3734
|
+
# about the candidate that lands: the succession challenge covers it either way. What is
|
|
3735
|
+
# exempted is exactly one class -- a challenge on a candidate that will never land.
|
|
3333
3736
|
reset_case passed unavailable unavailable
|
|
3334
3737
|
out="$(run_challenge_gate --focus review-predecessor --review-chain-id succ-review-predecessor --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-round-one.json")"; rc=$?
|
|
3335
|
-
check "a succession
|
|
3336
|
-
'[ "$rc" =
|
|
3738
|
+
check "a succession may carry a chain whose terminal receipt is its review" \
|
|
3739
|
+
'[ "$rc" = 0 ] && json_fields "$out" mode=challenge review_chain_tracked=true review_chain_id=succ-review-predecessor autonomous_review_index=1 predecessor_chain_id=succ-phase-one'
|
|
3740
|
+
|
|
3741
|
+
# The exemption is bounded by the receipt's own arithmetic. This is a FORGERY guard and
|
|
3742
|
+
# is asserted as one: the fixture below is a shape the controller never emits, because a
|
|
3743
|
+
# genuine round-1 review reads the same whether its chain later ran a challenge or not.
|
|
3744
|
+
# A caller who spent the challenge and presents only the review is accepted here -- the
|
|
3745
|
+
# stateless controller cannot see omitted history -- so no test claims otherwise.
|
|
3746
|
+
python3 - "$WORK/succ-round-one.json" "$WORK/succ-predecessor-spent-review.json" <<'PLAN'
|
|
3747
|
+
import json, sys
|
|
3748
|
+
from pathlib import Path
|
|
3749
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
3750
|
+
spent = json.loads(json.dumps(source))
|
|
3751
|
+
spent["autonomous_reviews_remaining"] = 0
|
|
3752
|
+
spent["autonomous_review_allowed"] = False
|
|
3753
|
+
Path(sys.argv[2]).write_text(json.dumps(spent, separators=(",", ":")))
|
|
3754
|
+
PLAN
|
|
3755
|
+
reset_case passed unavailable unavailable
|
|
3756
|
+
out="$(run_challenge_gate --focus spent-review --review-chain-id succ-spent-review --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-predecessor-spent-review.json")"; rc=$?
|
|
3757
|
+
check "a succession rejects a forged review receipt whose own arithmetic says its chain is spent" \
|
|
3758
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession predecessor is not its chain'"'"'s terminal round"*) true;; *) false;; esac'
|
|
3759
|
+
|
|
3760
|
+
reset_case passed unavailable unavailable
|
|
3761
|
+
out="$(run_challenge_gate --focus complete-predecessor --review-chain-id succ-complete-predecessor --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-predecessor-complete-mode.json")"; rc=$?
|
|
3762
|
+
check "a succession rejects a predecessor that is neither a review nor a challenge round" \
|
|
3763
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession predecessor is not a tracked review or challenge receipt"*) true;; *) false;; esac'
|
|
3337
3764
|
|
|
3338
3765
|
# A mid-chain challenge is a live chain, not an ended one: succeeding it would
|
|
3339
3766
|
# silently retire rounds the wrapper still owes. chain-round-two above is round 2
|
|
@@ -4170,6 +4597,11 @@ reset_case findings unavailable unavailable
|
|
|
4170
4597
|
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 2 --focus final-findings --review-chain-id long-task --autonomous-review-index 3 --prior-review-result-file "$WORK/chain-round-one.json" --prior-review-result-file "$WORK/chain-round-two.json")"; rc=$?
|
|
4171
4598
|
check "findings in the last tracked Agent round return to a post-budget checkpoint" \
|
|
4172
4599
|
'[ "$rc" = 0 ] && json_fields "$out" status=findings review_chain_tracked=true autonomous_review_index=3 autonomous_reviews_remaining=0 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true next_action=triage_findings_and_continue_independent_work self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.blocks.0=external_review self_review_gate.allowed_next_actions.2=continue_independent_work'
|
|
4600
|
+
# Round 1 of this chain also returned findings, so this is the second one: the rule the
|
|
4601
|
+
# trigger carries is about the RECURRENCE, and the agent reads it here rather than in a
|
|
4602
|
+
# skill it never loads while inside the chain.
|
|
4603
|
+
check "a second findings round in one chain raises the design check" \
|
|
4604
|
+
'[ "$rc" = 0 ] && json_fields "$out" self_review_gate.required_triggers.2=recurring_findings_design_check && grep -q decide_keep_delete_narrow_replace <<<"$out"'
|
|
4173
4605
|
|
|
4174
4606
|
reset_case passed unavailable unavailable
|
|
4175
4607
|
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus missing-history --review-chain-id long-task --autonomous-review-index 2)"; rc=$?
|
|
@@ -30,15 +30,22 @@ CORE="$TMP/core.txt"
|
|
|
30
30
|
LATEST="$TMP/latest.txt"
|
|
31
31
|
OLD_INTENT="$TMP/old-intent.txt"
|
|
32
32
|
|
|
33
|
-
python3 - "$PLAN" "$APPEND" "$CORE" "$LATEST" "$OLD_INTENT" <<'PY'
|
|
33
|
+
python3 - "$PLAN" "$APPEND" "$CORE" "$LATEST" "$OLD_INTENT" "$SCRIPT_DIR" <<'PY'
|
|
34
34
|
import json
|
|
35
35
|
import hashlib
|
|
36
|
+
import subprocess
|
|
36
37
|
import sys
|
|
37
38
|
from pathlib import Path
|
|
38
39
|
|
|
39
40
|
plan_path, append_path, core_path, latest_path, old_intent_path = map(
|
|
40
|
-
Path, sys.argv[1:]
|
|
41
|
+
Path, sys.argv[1:6]
|
|
41
42
|
)
|
|
43
|
+
required_concerns = subprocess.run(
|
|
44
|
+
[sys.executable, str(Path(sys.argv[6]) / "review_gate.py"),
|
|
45
|
+
"--print-required-concerns", "--stage", "build"],
|
|
46
|
+
capture_output=True, text=True, check=True,
|
|
47
|
+
).stdout.split()
|
|
48
|
+
assert required_concerns, "the controller printed no required concerns"
|
|
42
49
|
old = "scope:" + ("x" * (3995 - len("scope:") - len("c27"))) + "c27"
|
|
43
50
|
latest = "latest-round:c28"
|
|
44
51
|
core = old[: 3892 - len("\n\n") - len(latest)]
|
|
@@ -55,12 +62,12 @@ plan = {
|
|
|
55
62
|
"conclusion": conclusion,
|
|
56
63
|
"evidence_refs": ["focused-test"],
|
|
57
64
|
}
|
|
65
|
+
# Derived, not copied: a fixture holding its own copy of the required set
|
|
66
|
+
# stops satisfying the gate the moment that set changes, and the runner
|
|
67
|
+
# aborts at its first failing target so the drift surfaces rounds later.
|
|
58
68
|
for concern, conclusion in (
|
|
59
|
-
(
|
|
60
|
-
|
|
61
|
-
("failure_paths", "The focused checks cover overflow, stale input, and malformed text paths."),
|
|
62
|
-
("tests_evidence", "The focused regression fails when bounded update guarantees are removed."),
|
|
63
|
-
("compatibility", "The focused checks preserve the plan schema and original file permissions."),
|
|
69
|
+
(concern, f"The focused updater checks cover {concern} on this fixture.")
|
|
70
|
+
for concern in required_concerns
|
|
64
71
|
)
|
|
65
72
|
],
|
|
66
73
|
"evidence": [
|