@ccoalm/ccl-skills 0.16.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (18) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -4
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +73 -123
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/wording-only-review.md +136 -0
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +120 -11
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +17 -1
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_order.sh +30 -15
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +209 -12
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +14 -7
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +1 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +22 -0
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +10 -0
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +4 -0
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +56 -18
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +41 -1
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +3 -2
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/closeout-reread.md +40 -0
  17. package/dist/assets/release.json +27 -17
  18. package/package.json +1 -1
@@ -20,6 +20,22 @@ from unittest import mock
20
20
 
21
21
 
22
22
  SCRIPT_DIR = Path(__file__).resolve().parent
23
+
24
+
25
+ def required_concerns(stage: str, *risk_tags: str) -> list[str]:
26
+ """The concern ids a plan owes, asked of the controller that enforces them.
27
+
28
+ A fixture holding its own copy of this list silently stops satisfying the gate
29
+ when the set changes, and the suite runner aborts at its first failing target,
30
+ so the drift surfaces rounds later. There is one owner; ask it.
31
+ """
32
+ argv = [sys.executable, str(SCRIPT_DIR / "review_gate.py"),
33
+ "--print-required-concerns", "--stage", stage]
34
+ for tag in risk_tags:
35
+ argv += ["--risk-tag", tag]
36
+ printed = subprocess.run(argv, capture_output=True, text=True, check=True).stdout.split()
37
+ assert printed, "the controller printed no required concerns"
38
+ return printed
23
39
  if str(SCRIPT_DIR) not in sys.path:
24
40
  sys.path.insert(0, str(SCRIPT_DIR))
25
41
  import review_gate
@@ -736,7 +752,7 @@ class CompletionFindingDispositionTest(unittest.TestCase):
736
752
  {"concern": concern,
737
753
  "conclusion": f"The synthetic completion fixture preserves {concern} boundaries.",
738
754
  "evidence_refs": ["fixture"]}
739
- for concern in ("correctness", "safety", "failure_paths", "tests_evidence", "compatibility")
755
+ for concern in required_concerns("build")
740
756
  ],
741
757
  "evidence": [{"id": "fixture", "result": "Synthetic exact-candidate completion and history fixture."}],
742
758
  })
@@ -47,7 +47,7 @@ case "$client" in
47
47
  codex) family=openai; provider=openai; model=codex-local-default ;;
48
48
  *) exit 2 ;;
49
49
  esac
50
- concern_results='[{"concern":"correctness","conclusion":"Checked routing correctness."},{"concern":"safety","conclusion":"Checked fail-closed safety."},{"concern":"failure_paths","conclusion":"Checked fallback failure paths."},{"concern":"tests_evidence","conclusion":"Checked deterministic test evidence."},{"concern":"compatibility","conclusion":"Checked client-order compatibility."}]'
50
+ concern_results='[{"concern":"correctness","conclusion":"Checked routing correctness."},{"concern":"safety","conclusion":"Checked fail-closed safety."},{"concern":"failure_paths","conclusion":"Checked fallback failure paths."},{"concern":"tests_evidence","conclusion":"Checked deterministic test evidence."},{"concern":"compatibility","conclusion":"Checked client-order compatibility."},{"concern":"claim_strength","conclusion":"Checked that no claim reaches past the client-order fixture."}]'
51
51
 
52
52
  if [ "$client" = "claude" ]; then
53
53
  case "$behavior" in
@@ -105,20 +105,35 @@ done
105
105
 
106
106
  printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n' >"$WORK/diff.patch"
107
107
  printf 'diff --git a/c b/c\n--- a/c\n+++ b/c\n@@ -1 +1 @@\n-x\n+aws_key = "AKIAIOSFODNN7EXAMPLE"\n' >"$WORK/secret-diff.patch"
108
- cat >"$WORK/review-plan.json" <<'JSON'
109
- {
110
- "intent": "Preserve independent reviewer routing while adding staged review.",
111
- "acceptance": ["Client ordering and model-family exclusion remain deterministic."],
112
- "self_review": [
113
- {"concern": "correctness", "conclusion": "Routing preserves the first eligible independent result.", "evidence_refs": ["e1"]},
114
- {"concern": "safety", "conclusion": "Terminal boundaries remain fail closed across clients.", "evidence_refs": ["e1"]},
115
- {"concern": "failure_paths", "conclusion": "Candidate-local failures alone enter the fallback chain.", "evidence_refs": ["e1"]},
116
- {"concern": "tests_evidence", "conclusion": "Deterministic stubs cover client order and attribution.", "evidence_refs": ["e1"]},
117
- {"concern": "compatibility", "conclusion": "Existing client-order customization remains supported.", "evidence_refs": ["e1"]}
118
- ],
119
- "evidence": [{"id": "e1", "result": "Deterministic client routing contract fixture."}]
120
- }
121
- JSON
108
+ # The plan's required concern set has ONE owner. A fixture keeping its own copy
109
+ # stops satisfying the gate the moment that set changes, and the suite runner
110
+ # aborts at its first failing target, so the drift surfaces rounds later -- five
111
+ # fixtures drifted that way at once. Ask the controller instead.
112
+ python3 - "$WORK/review-plan.json" "$DIR" <<'PY'
113
+ import json
114
+ import subprocess
115
+ import sys
116
+ from pathlib import Path
117
+
118
+ plan_path, script_dir = Path(sys.argv[1]), Path(sys.argv[2])
119
+ required = subprocess.run(
120
+ [sys.executable, str(script_dir / "review_gate.py"),
121
+ "--print-required-concerns", "--stage", "build"],
122
+ capture_output=True, text=True, check=True,
123
+ ).stdout.split()
124
+ assert required, "the controller printed no required concerns"
125
+ plan_path.write_text(json.dumps({
126
+ "intent": "Preserve independent reviewer routing while adding staged review.",
127
+ "acceptance": ["Client ordering and model-family exclusion remain deterministic."],
128
+ "self_review": [
129
+ {"concern": concern,
130
+ "conclusion": f"The deterministic client-routing stubs cover {concern}.",
131
+ "evidence_refs": ["e1"]}
132
+ for concern in required
133
+ ],
134
+ "evidence": [{"id": "e1", "result": "Deterministic client routing contract fixture."}],
135
+ }, indent=2) + "\n", encoding="utf-8")
136
+ PY
122
137
 
123
138
  reset_case() {
124
139
  rm -f "$WORK/state"/*
@@ -951,7 +951,8 @@ cat >"$WORK/review-plan.json" <<'JSON'
951
951
  {"concern": "tests_evidence", "conclusion": "Focused deterministic contract tests cover the change.", "evidence_refs": ["e1"]},
952
952
  {"concern": "compatibility", "conclusion": "Existing provider routing remains backward compatible.", "evidence_refs": ["e1"]},
953
953
  {"concern": "rollout_rollback", "conclusion": "The local CLI change has a direct revert path.", "evidence_refs": ["e1"]},
954
- {"concern": "observability_operations", "conclusion": "The JSON envelope exposes stage and depth for diagnosis.", "evidence_refs": ["e1"]}
954
+ {"concern": "observability_operations", "conclusion": "The JSON envelope exposes stage and depth for diagnosis.", "evidence_refs": ["e1"]},
955
+ {"concern": "claim_strength", "conclusion": "Each claim is scoped to the fixture it was observed on.", "evidence_refs": ["e1"]}
955
956
  ],
956
957
  "evidence": [
957
958
  {"id": "e1", "result": "Deterministic fake-wrapper contract fixture."}
@@ -1033,7 +1034,8 @@ cat >"$WORK/placeholder-plan.json" <<'JSON'
1033
1034
  {"concern": "safety", "conclusion": "Packet and tool boundaries remain fail closed.", "evidence_refs": ["e1"]},
1034
1035
  {"concern": "failure_paths", "conclusion": "Invalid and inconclusive paths remain terminal.", "evidence_refs": ["e1"]},
1035
1036
  {"concern": "tests_evidence", "conclusion": "A focused regression proves filler is rejected.", "evidence_refs": ["e1"]},
1036
- {"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]}
1037
+ {"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]},
1038
+ {"concern": "claim_strength", "conclusion": "No claim reaches past the filler-rejection fixture.", "evidence_refs": ["e1"]}
1037
1039
  ],
1038
1040
  "evidence": [{"id": "e1", "result": "Deterministic placeholder-validation fixture."}]
1039
1041
  }
@@ -1058,6 +1060,7 @@ cat >"$WORK/high-risk-plan.json" <<'JSON'
1058
1060
  {"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]},
1059
1061
  {"concern": "rollout_rollback", "conclusion": "The local contract change has a direct revert path.", "evidence_refs": ["e1"]},
1060
1062
  {"concern": "observability_operations", "conclusion": "The result exposes depth and risk tags for diagnosis.", "evidence_refs": ["e1"]},
1063
+ {"concern": "claim_strength", "conclusion": "Each claim is scoped to the high-risk fixture it was observed on.", "evidence_refs": ["e1"]},
1061
1064
  {"concern": "high_risk_boundary", "conclusion": "Bypass attempts cannot remove controller-required concerns.", "evidence_refs": ["e1"]}
1062
1065
  ],
1063
1066
  "evidence": [{"id": "e1", "result": "Deterministic high-risk gate fixture."}]
@@ -1081,7 +1084,8 @@ cat >"$WORK/near-limit-plan.json" <<JSON
1081
1084
  {"concern": "safety", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
1082
1085
  {"concern": "failure_paths", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
1083
1086
  {"concern": "tests_evidence", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
1084
- {"concern": "compatibility", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]}
1087
+ {"concern": "compatibility", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
1088
+ {"concern": "claim_strength", "conclusion": "Scoped to this size fixture.", "evidence_refs": ["e1"]}
1085
1089
  ],
1086
1090
  "evidence": [{"id": "e1", "result": "$large_evidence"}]
1087
1091
  }
@@ -1374,6 +1378,144 @@ for malformed_plan in non-string-owner-review-plan empty-owner-review-plan; do
1374
1378
  '[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true'
1375
1379
  done
1376
1380
 
1381
+ # The claim-strength walk is a required concern, so it is owed BEFORE round 1 --
1382
+ # the one point in a round where correcting an overstated claim costs nothing. The
1383
+ # class it covers (absolutes, universals, causal and exhaustiveness claims the cited
1384
+ # evidence does not carry) otherwise keeps arriving as a LATE correction, after the
1385
+ # receipts are bound, where any candidate edit voids them.
1386
+ python3 - "$WORK/review-plan.json" "$WORK/no-claim-strength-review-plan.json" <<'PLAN'
1387
+ import json, sys
1388
+ from pathlib import Path
1389
+ plan = json.loads(Path(sys.argv[1]).read_text())
1390
+ plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != "claim_strength"]
1391
+ Path(sys.argv[2]).write_text(json.dumps(plan))
1392
+ PLAN
1393
+ reset_case passed unavailable unavailable
1394
+ out="$(run_gate --review-plan-file "$WORK/no-claim-strength-review-plan.json")"; rc=$?
1395
+ check "a plan that skips the claim-strength walk fails before any provider runs" \
1396
+ '[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true self_review_gate.required_triggers.0=before_external_review'
1397
+
1398
+ reset_case passed unavailable unavailable
1399
+ out="$(run_gate --allow-fallback-egress)"; rc=$?
1400
+ check "the build reviewer is asked to check claim strength" \
1401
+ '[ "$rc" = 0 ] && json_fields "$out" reviewed_concerns.5=claim_strength'
1402
+
1403
+ # The exported list is only worth deriving from if it IS the enforced one. Build a
1404
+ # plan covering exactly what the controller prints, and then drop each printed
1405
+ # concern in turn: acceptance proves the print covers everything the gate demands,
1406
+ # and every single-drop rejection proves nothing printed is decorative. Without
1407
+ # both directions a caller could derive from a list that had quietly diverged --
1408
+ # which is the drift this export exists to remove.
1409
+ # The exit status is asserted too. This suite runs without errexit, so a command
1410
+ # substitution silently discards it: a printer that emits the right concerns and then
1411
+ # fails would satisfy a non-empty check and report agreement it never reached.
1412
+ printed_rc=0
1413
+ printed_concerns="$("$DIR/review_gate.sh" --print-required-concerns --stage build)" || printed_rc=$?
1414
+ check "the controller can print the concern set a plan owes, and succeeds doing it" \
1415
+ '[ -n "$printed_concerns" ] && [ "$printed_rc" = 0 ]'
1416
+ # Proving agreement at ONE depth leaves the other branch free to diverge with every
1417
+ # test green -- and release/high-risk is the branch that carries the most concerns.
1418
+ # Assert the depth-raising branch answers what the gate itself derives for it.
1419
+ # Parity includes what each side REFUSES. A printer that answers for tags the enforcer
1420
+ # rejects reintroduces the divergence this export removes: a caller deriving from a
1421
+ # malformed tag would get a list where the real round fails closed.
1422
+ for bad_tag in "a b" "" "$(printf 'x%.0s' $(seq 81))"; do
1423
+ # rc captured without touching shell options: this suite runs under `set -uo pipefail`
1424
+ # and enabling errexit here would abort every later case at its first non-zero command.
1425
+ bad_tag_rc=0
1426
+ "$DIR/review_gate.sh" --print-required-concerns --stage build --risk-tag "$bad_tag" >/dev/null 2>&1 || bad_tag_rc=$?
1427
+ # Parity is a claim about TWO sides, so both are exercised: asserting only the
1428
+ # printer would keep these checks green if the enforcer's own rejection were
1429
+ # removed, which is the half this pair exists to tie together.
1430
+ enforcer_tag_rc=0
1431
+ run_gate --risk-tag "$bad_tag" >/dev/null 2>&1 || enforcer_tag_rc=$?
1432
+ check "printer and enforcer both refuse the same malformed risk tag (${#bad_tag} chars)" \
1433
+ '[ "$bad_tag_rc" != 0 ] && [ "$enforcer_tag_rc" != 0 ]'
1434
+ done
1435
+ printed_release_rc=0
1436
+ printed_release="$("$DIR/review_gate.sh" --print-required-concerns --stage explore --risk-tag shared-gate)" || printed_release_rc=$?
1437
+ check "the raised-depth print succeeds" '[ "$printed_release_rc" = 0 ]'
1438
+ # Hoisted for the same reason as the calls above: nested inside the comparison, this
1439
+ # printer call's exit status was discarded, so a regression failing only for explicit
1440
+ # release depth would have compared equal and passed. Every printer invocation in this
1441
+ # suite now has its status asserted.
1442
+ printed_plain_release_rc=0
1443
+ printed_plain_release="$("$DIR/review_gate.sh" --print-required-concerns --stage release)" || printed_plain_release_rc=$?
1444
+ check "the plain release print succeeds" '[ "$printed_plain_release_rc" = 0 ]'
1445
+ check "a high-risk tag raises the printed set to release depth and adds the boundary concern" \
1446
+ '[ "$(printf %s "$printed_release" | tr "\n" " ")" = "$(printf "%s\nhigh_risk_boundary" "$printed_plain_release" | tr "\n" " ")" ]'
1447
+ python3 - "$WORK/review-plan.json" "$WORK/printed-plan.json" $printed_concerns <<'PLAN'
1448
+ import json, sys
1449
+ from pathlib import Path
1450
+ source = json.loads(Path(sys.argv[1]).read_text())
1451
+ printed = sys.argv[3:]
1452
+ by_concern = {row["concern"]: row for row in source["self_review"]}
1453
+ source["self_review"] = [
1454
+ by_concern.get(concern, {"concern": concern,
1455
+ "conclusion": f"The fixture covers {concern}.",
1456
+ "evidence_refs": ["e1"]})
1457
+ for concern in printed
1458
+ ]
1459
+ Path(sys.argv[2]).write_text(json.dumps(source))
1460
+ PLAN
1461
+ reset_case passed unavailable unavailable
1462
+ out="$(run_gate --review-plan-file "$WORK/printed-plan.json" --allow-fallback-egress)"; rc=$?
1463
+ check "a plan built from the printed set satisfies the gate" '[ "$rc" = 0 ]'
1464
+ printed_drop_failures=0
1465
+ for dropped in $printed_concerns; do
1466
+ python3 - "$WORK/printed-plan.json" "$WORK/printed-plan-minus.json" "$dropped" <<'PLAN'
1467
+ import json, sys
1468
+ from pathlib import Path
1469
+ plan = json.loads(Path(sys.argv[1]).read_text())
1470
+ plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != sys.argv[3]]
1471
+ Path(sys.argv[2]).write_text(json.dumps(plan))
1472
+ PLAN
1473
+ reset_case passed unavailable unavailable
1474
+ out="$(run_gate --review-plan-file "$WORK/printed-plan-minus.json")"; rc=$?
1475
+ if [ "$rc" = 2 ] && json_fields "$out" reason_code=self_review_incomplete; then
1476
+ printed_drop_failures=$((printed_drop_failures+1))
1477
+ fi
1478
+ done
1479
+ check "every printed concern is one the gate actually demands" \
1480
+ '[ "$printed_drop_failures" = "$(printf %s "$printed_concerns" | wc -w | tr -d " ")" ]'
1481
+
1482
+ # The same two directions at the OTHER depth. A printer that agreed with the gate at
1483
+ # build and diverged at release/high-risk would keep every test above green, and
1484
+ # release is the branch carrying the most concerns.
1485
+ python3 - "$WORK/high-risk-plan.json" "$WORK/printed-release-plan.json" $printed_release <<'PLAN'
1486
+ import json, sys
1487
+ from pathlib import Path
1488
+ source = json.loads(Path(sys.argv[1]).read_text())
1489
+ by_concern = {row["concern"]: row for row in source["self_review"]}
1490
+ source["self_review"] = [
1491
+ by_concern.get(concern, {"concern": concern,
1492
+ "conclusion": f"The high-risk fixture covers {concern}.",
1493
+ "evidence_refs": ["e1"]})
1494
+ for concern in sys.argv[3:]
1495
+ ]
1496
+ Path(sys.argv[2]).write_text(json.dumps(source))
1497
+ PLAN
1498
+ reset_case passed unavailable unavailable
1499
+ out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/printed-release-plan.json" --review-chain-id printed-release --autonomous-review-index 1 --allow-fallback-egress)"; rc=$?
1500
+ check "a plan built from the printed release set satisfies the raised-depth gate" '[ "$rc" = 0 ]'
1501
+ printed_release_drop_failures=0
1502
+ for dropped in $printed_release; do
1503
+ python3 - "$WORK/printed-release-plan.json" "$WORK/printed-release-minus.json" "$dropped" <<'PLAN'
1504
+ import json, sys
1505
+ from pathlib import Path
1506
+ plan = json.loads(Path(sys.argv[1]).read_text())
1507
+ plan["self_review"] = [row for row in plan["self_review"] if row["concern"] != sys.argv[3]]
1508
+ Path(sys.argv[2]).write_text(json.dumps(plan))
1509
+ PLAN
1510
+ reset_case passed unavailable unavailable
1511
+ out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/printed-release-minus.json" --review-chain-id printed-release-minus --autonomous-review-index 1)"; rc=$?
1512
+ if [ "$rc" = 2 ] && json_fields "$out" reason_code=self_review_incomplete; then
1513
+ printed_release_drop_failures=$((printed_release_drop_failures+1))
1514
+ fi
1515
+ done
1516
+ check "every printed release concern is one the raised-depth gate actually demands" \
1517
+ '[ "$printed_release_drop_failures" = "$(printf %s "$printed_release" | wc -w | tr -d " ")" ]'
1518
+
1377
1519
  owner_lstat_classification="$(python3 - "$DIR/review_gate.py" <<'PY'
1378
1520
  import errno
1379
1521
  import importlib.util
@@ -2073,7 +2215,7 @@ check "complete but placeholder concern conclusions cannot false-green" \
2073
2215
  reset_case findings unavailable unavailable
2074
2216
  out="$(run_gate --allow-fallback-egress)"; rc=$?
2075
2217
  check "Claude findings remain findings" \
2076
- '[ "$rc" = 0 ] && json_fields "$out" status=findings selected_client=claude next_action=triage_findings_and_continue_independent_work autonomous_review_budget=1 autonomous_review_index=1 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim self_review_gate.allowed_next_actions.0=deep_self_review self_review_gate.allowed_next_actions.1=continue_implementation self_review_gate.allowed_next_actions.2=continue_independent_work'
2218
+ '[ "$rc" = 0 ] && json_fields "$out" status=findings selected_client=claude next_action=triage_findings_and_continue_independent_work autonomous_review_budget=1 autonomous_review_index=1 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim self_review_gate.allowed_next_actions.0=deep_self_review self_review_gate.allowed_next_actions.1=continue_implementation self_review_gate.allowed_next_actions.2=continue_independent_work && ! grep -q recurring_findings_design_check <<<"$out"'
2077
2219
 
2078
2220
  reset_case quota passed unavailable
2079
2221
  out="$(run_gate --allow-fallback-egress)"; rc=$?
@@ -2351,7 +2493,7 @@ check "an initial review with challenge capacity requires a tracked Agent chain"
2351
2493
  reset_case passed unavailable unavailable
2352
2494
  out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/high-risk-plan.json" --review-chain-id high-risk-task --autonomous-review-index 1)"; rc=$?
2353
2495
  check "high-risk tags raise explore to release depth and default one challenge" \
2354
- '[ "$rc" = 0 ] && json_fields "$out" stage=explore stage_source=caller-declared review_depth=release risk_tags_source=caller-declared challenge_budget=1 challenge_rounds_remaining=1 review_chain_tracked=true review_chain_id=high-risk-task autonomous_review_budget=2 autonomous_review_index=1 autonomous_reviews_remaining=1 autonomous_review_allowed=true next_action=run_challenge completion_gated=true risk_tags.0=shared-gate reviewed_concerns.7=high_risk_boundary self_review_gate.required=false self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.satisfied_triggers.1=risk_or_scope_escalation'
2496
+ '[ "$rc" = 0 ] && json_fields "$out" stage=explore stage_source=caller-declared review_depth=release risk_tags_source=caller-declared challenge_budget=1 challenge_rounds_remaining=1 review_chain_tracked=true review_chain_id=high-risk-task autonomous_review_budget=2 autonomous_review_index=1 autonomous_reviews_remaining=1 autonomous_review_allowed=true next_action=run_challenge completion_gated=true risk_tags.0=shared-gate reviewed_concerns.8=high_risk_boundary self_review_gate.required=false self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.satisfied_triggers.1=risk_or_scope_escalation'
2355
2497
 
2356
2498
  # Release/high-risk normally requires a challenge. The only single-review
2357
2499
  # exception is a candidate-bound deterministic wording-only proof whose result
@@ -3160,7 +3302,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
3160
3302
  --wording-only-proof-file "$WORK/wording-punctuation-proof.json")"; rc=$?
3161
3303
  punctuation_proof_hash="$(shasum -a 256 "$WORK/wording-punctuation-proof.json" | awk '{print $1}')"
3162
3304
  check "release wording-only punctuation scope can take one proof-bound review" \
3163
- '[ "$rc" = 0 ] && json_fields "$out" challenge_budget=0 review_chain_tracked=false wording_only_scope.status=passed wording_only_scope.check_kind=markdown-punctuation-only wording_only_proof_sha256="$punctuation_proof_hash" reviewed_concerns.7=wording_only_boundary'
3305
+ '[ "$rc" = 0 ] && json_fields "$out" challenge_budget=0 review_chain_tracked=false wording_only_scope.status=passed wording_only_scope.check_kind=markdown-punctuation-only wording_only_proof_sha256="$punctuation_proof_hash" reviewed_concerns.8=wording_only_boundary'
3164
3306
 
3165
3307
  reset_case passed unavailable unavailable
3166
3308
  out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
@@ -3169,7 +3311,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
3169
3311
  --implementer-family openai --review-plan-file "$WORK/review-plan.json" \
3170
3312
  --wording-only-proof-file "$WORK/wording-punctuation-proof.json")"; rc=$?
3171
3313
  check "build wording-only review records the same controller-bound proof" \
3172
- '[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-punctuation-only reviewed_concerns.5=wording_only_boundary'
3314
+ '[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-punctuation-only reviewed_concerns.6=wording_only_boundary'
3173
3315
  printf '%s\n' "$out" >"$WORK/wording-punctuation-review.json"
3174
3316
  reset_case passed unavailable unavailable
3175
3317
  out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
@@ -3245,7 +3387,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
3245
3387
  --implementer-family openai --review-plan-file "$WORK/review-plan.json" \
3246
3388
  --wording-only-proof-file "$WORK/wording-token-proof.json")"; rc=$?
3247
3389
  check "build exact typo replacement can take one proof-bound review" \
3248
- '[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.old_token=teh wording_only_scope.new_token=the wording_only_scope.expected_count=1 wording_only_scope.replacement_count=1 reviewed_concerns.5=wording_only_boundary'
3390
+ '[ "$rc" = 0 ] && json_fields "$out" review_depth=build challenge_budget=0 wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.old_token=teh wording_only_scope.new_token=the wording_only_scope.expected_count=1 wording_only_scope.replacement_count=1 reviewed_concerns.6=wording_only_boundary'
3249
3391
 
3250
3392
  reset_case passed unavailable unavailable
3251
3393
  out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
@@ -3272,7 +3414,7 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
3272
3414
  --implementer-family openai --review-plan-file "$WORK/review-plan.json" \
3273
3415
  --wording-only-proof-file "$WORK/wording-base-proof.json")"; rc=$?
3274
3416
  check "base-mode build wording proof freezes full context from line one" \
3275
- '[ "$rc" = 0 ] && json_fields "$out" review_depth=build wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.replacement_count=1 reviewed_concerns.5=wording_only_boundary'
3417
+ '[ "$rc" = 0 ] && json_fields "$out" review_depth=build wording_only_scope.check_kind=markdown-token-replacement wording_only_scope.replacement_count=1 reviewed_concerns.6=wording_only_boundary'
3276
3418
 
3277
3419
  for rejected_scope in multi-skill-wording truncated-context-wording symlink-mode-wording frontmatter-shift-insert-wording frontmatter-shift-delete-wording no-final-newline-wording invalid-octal-wording huge-hunk-number-wording zero-width-wording bidi-control-wording emoji-symbol-wording currency-symbol-wording decomposed-boundary-wording zwj-boundary-wording; do
3278
3420
  reset_case passed unavailable unavailable
@@ -3499,6 +3641,13 @@ printf '%s\n' "$passed_round_one" >"$WORK/passed-round-one.json"
3499
3641
  check "a passed first tracked round still owes its challenge before completion" \
3500
3642
  '[ "$passed_round_one_rc" = 0 ] && json_fields "$passed_round_one" status=passed autonomous_review_index=1 autonomous_reviews_remaining=2 autonomous_review_allowed=true next_action=run_challenge completion_gated=true'
3501
3643
 
3644
+ # Control leg for the recurrence trigger: same shape, first findings round. Without it a
3645
+ # probe that fires for an unrelated reason would read as the recurrence being detected.
3646
+ reset_case findings unavailable unavailable
3647
+ out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus passed-prior-findings --review-chain-id passed-task --autonomous-review-index 2 --prior-review-result-file "$WORK/passed-round-one.json")"; rc=$?
3648
+ check "findings after a clean prior round stay a first findings round" \
3649
+ '[ "$rc" = 0 ] && json_fields "$out" status=findings self_review_gate.required_triggers.0=findings_returned && ! grep -q recurring_findings_design_check <<<"$out"'
3650
+
3502
3651
  # Chain succession. A fix that touches the owner package moves selected_skills_sha256
3503
3652
  # and ends the chain by design, so the post-fix candidate can never be challenged
3504
3653
  # inside it. Succession opens ONE new chain whose first Agent round is a challenge,
@@ -3522,7 +3671,8 @@ python3 - "$WORK/succ-round-two.json" \
3522
3671
  "$WORK/succ-predecessor-owner-moved.json" \
3523
3672
  "$WORK/succ-predecessor-forged-controller.json" \
3524
3673
  "$WORK/succ-predecessor-foreign-scope.json" \
3525
- "$WORK/succ-predecessor-forged-terminal.json" <<'PY'
3674
+ "$WORK/succ-predecessor-forged-terminal.json" \
3675
+ "$WORK/succ-predecessor-complete-mode.json" <<'PY'
3526
3676
  import json
3527
3677
  from pathlib import Path
3528
3678
  import sys
@@ -3548,6 +3698,10 @@ forged_terminal["challenge_index"] = 0
3548
3698
  forged_terminal["autonomous_reviews_remaining"] = 1
3549
3699
  forged_terminal["autonomous_review_allowed"] = True
3550
3700
  Path(sys.argv[5]).write_text(json.dumps(forged_terminal, separators=(",", ":")))
3701
+ # Neither lane: a completion checkpoint is not a round the succession may carry.
3702
+ complete_mode = json.loads(json.dumps(source))
3703
+ complete_mode["mode"] = "complete"
3704
+ Path(sys.argv[6]).write_text(json.dumps(complete_mode, separators=(",", ":")))
3551
3705
  PY
3552
3706
 
3553
3707
  reset_case passed unavailable unavailable
@@ -3560,15 +3714,53 @@ out="$(run_challenge_gate --focus owner-moved --review-chain-id succ-owner-moved
3560
3714
  check "a succession accepts the owner-package hash move that ended the prior chain" \
3561
3715
  '[ "$rc" = 0 ] && json_fields "$out" mode=challenge predecessor_chain_id=succ-phase-one'
3562
3716
 
3717
+ # Budget is one review plus one challenge, so a fix ends the chain and the SECOND
3718
+ # findings round usually lands in the SUCCEEDING chain. Counting only in-chain rounds
3719
+ # would therefore never see the recurrence the trigger exists for.
3720
+ reset_case findings unavailable unavailable
3721
+ out="$(run_challenge_gate --focus recurring-findings --review-chain-id succ-recurrence --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-round-two.json")"; rc=$?
3722
+ check "findings after a predecessor chain that also returned findings raise the design check" \
3723
+ '[ "$rc" = 0 ] && json_fields "$out" status=findings self_review_gate.required_triggers.0=findings_returned && grep -q recurring_findings_design_check <<<"$out"'
3724
+
3563
3725
  reset_case passed unavailable unavailable
3564
3726
  out="$(run_challenge_gate --focus no-predecessor --review-chain-id succ-orphan --autonomous-review-index 1)"; rc=$?
3565
3727
  check "a tracked challenge cannot open a chain without a predecessor receipt" \
3566
3728
  '[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession"*) false;; *) true;; esac'
3567
3729
 
3730
+ # A fix applied straight after the REVIEW ends the chain exactly as a fix after the
3731
+ # challenge does -- the owner digest moves either way -- so the ended chain's terminal
3732
+ # receipt is its review. Requiring a challenge receipt here forced that challenge to be
3733
+ # spent on a candidate the author had already decided to change, and bought no evidence
3734
+ # about the candidate that lands: the succession challenge covers it either way. What is
3735
+ # exempted is exactly one class -- a challenge on a candidate that will never land.
3568
3736
  reset_case passed unavailable unavailable
3569
3737
  out="$(run_challenge_gate --focus review-predecessor --review-chain-id succ-review-predecessor --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-round-one.json")"; rc=$?
3570
- check "a succession rejects a predecessor that is not a challenge receipt" \
3571
- '[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession predecessor is not a tracked challenge receipt"*) true;; *) false;; esac'
3738
+ check "a succession may carry a chain whose terminal receipt is its review" \
3739
+ '[ "$rc" = 0 ] && json_fields "$out" mode=challenge review_chain_tracked=true review_chain_id=succ-review-predecessor autonomous_review_index=1 predecessor_chain_id=succ-phase-one'
3740
+
3741
+ # The exemption is bounded by the receipt's own arithmetic. This is a FORGERY guard and
3742
+ # is asserted as one: the fixture below is a shape the controller never emits, because a
3743
+ # genuine round-1 review reads the same whether its chain later ran a challenge or not.
3744
+ # A caller who spent the challenge and presents only the review is accepted here -- the
3745
+ # stateless controller cannot see omitted history -- so no test claims otherwise.
3746
+ python3 - "$WORK/succ-round-one.json" "$WORK/succ-predecessor-spent-review.json" <<'PLAN'
3747
+ import json, sys
3748
+ from pathlib import Path
3749
+ source = json.loads(Path(sys.argv[1]).read_text())
3750
+ spent = json.loads(json.dumps(source))
3751
+ spent["autonomous_reviews_remaining"] = 0
3752
+ spent["autonomous_review_allowed"] = False
3753
+ Path(sys.argv[2]).write_text(json.dumps(spent, separators=(",", ":")))
3754
+ PLAN
3755
+ reset_case passed unavailable unavailable
3756
+ out="$(run_challenge_gate --focus spent-review --review-chain-id succ-spent-review --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-predecessor-spent-review.json")"; rc=$?
3757
+ check "a succession rejects a forged review receipt whose own arithmetic says its chain is spent" \
3758
+ '[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession predecessor is not its chain'"'"'s terminal round"*) true;; *) false;; esac'
3759
+
3760
+ reset_case passed unavailable unavailable
3761
+ out="$(run_challenge_gate --focus complete-predecessor --review-chain-id succ-complete-predecessor --autonomous-review-index 1 --predecessor-chain-result-file "$WORK/succ-predecessor-complete-mode.json")"; rc=$?
3762
+ check "a succession rejects a predecessor that is neither a review nor a challenge round" \
3763
+ '[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid && case "$out" in *"chain succession predecessor is not a tracked review or challenge receipt"*) true;; *) false;; esac'
3572
3764
 
3573
3765
  # A mid-chain challenge is a live chain, not an ended one: succeeding it would
3574
3766
  # silently retire rounds the wrapper still owes. chain-round-two above is round 2
@@ -4405,6 +4597,11 @@ reset_case findings unavailable unavailable
4405
4597
  out="$(run_challenge_gate --challenge-budget 2 --challenge-index 2 --focus final-findings --review-chain-id long-task --autonomous-review-index 3 --prior-review-result-file "$WORK/chain-round-one.json" --prior-review-result-file "$WORK/chain-round-two.json")"; rc=$?
4406
4598
  check "findings in the last tracked Agent round return to a post-budget checkpoint" \
4407
4599
  '[ "$rc" = 0 ] && json_fields "$out" status=findings review_chain_tracked=true autonomous_review_index=3 autonomous_reviews_remaining=0 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true next_action=triage_findings_and_continue_independent_work self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.blocks.0=external_review self_review_gate.allowed_next_actions.2=continue_independent_work'
4600
+ # Round 1 of this chain also returned findings, so this is the second one: the rule the
4601
+ # trigger carries is about the RECURRENCE, and the agent reads it here rather than in a
4602
+ # skill it never loads while inside the chain.
4603
+ check "a second findings round in one chain raises the design check" \
4604
+ '[ "$rc" = 0 ] && json_fields "$out" self_review_gate.required_triggers.2=recurring_findings_design_check && grep -q decide_keep_delete_narrow_replace <<<"$out"'
4408
4605
 
4409
4606
  reset_case passed unavailable unavailable
4410
4607
  out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus missing-history --review-chain-id long-task --autonomous-review-index 2)"; rc=$?
@@ -30,15 +30,22 @@ CORE="$TMP/core.txt"
30
30
  LATEST="$TMP/latest.txt"
31
31
  OLD_INTENT="$TMP/old-intent.txt"
32
32
 
33
- python3 - "$PLAN" "$APPEND" "$CORE" "$LATEST" "$OLD_INTENT" <<'PY'
33
+ python3 - "$PLAN" "$APPEND" "$CORE" "$LATEST" "$OLD_INTENT" "$SCRIPT_DIR" <<'PY'
34
34
  import json
35
35
  import hashlib
36
+ import subprocess
36
37
  import sys
37
38
  from pathlib import Path
38
39
 
39
40
  plan_path, append_path, core_path, latest_path, old_intent_path = map(
40
- Path, sys.argv[1:]
41
+ Path, sys.argv[1:6]
41
42
  )
43
+ required_concerns = subprocess.run(
44
+ [sys.executable, str(Path(sys.argv[6]) / "review_gate.py"),
45
+ "--print-required-concerns", "--stage", "build"],
46
+ capture_output=True, text=True, check=True,
47
+ ).stdout.split()
48
+ assert required_concerns, "the controller printed no required concerns"
42
49
  old = "scope:" + ("x" * (3995 - len("scope:") - len("c27"))) + "c27"
43
50
  latest = "latest-round:c28"
44
51
  core = old[: 3892 - len("\n\n") - len(latest)]
@@ -55,12 +62,12 @@ plan = {
55
62
  "conclusion": conclusion,
56
63
  "evidence_refs": ["focused-test"],
57
64
  }
65
+ # Derived, not copied: a fixture holding its own copy of the required set
66
+ # stops satisfying the gate the moment that set changes, and the runner
67
+ # aborts at its first failing target so the drift surfaces rounds later.
58
68
  for concern, conclusion in (
59
- ("correctness", "The focused checks cover the updater's accepted state transitions."),
60
- ("safety", "The focused checks cover no-write failures and file integrity boundaries."),
61
- ("failure_paths", "The focused checks cover overflow, stale input, and malformed text paths."),
62
- ("tests_evidence", "The focused regression fails when bounded update guarantees are removed."),
63
- ("compatibility", "The focused checks preserve the plan schema and original file permissions."),
69
+ (concern, f"The focused updater checks cover {concern} on this fixture.")
70
+ for concern in required_concerns
64
71
  )
65
72
  ],
66
73
  "evidence": [
@@ -25,6 +25,7 @@ The read side already defends against oversized files (chunked reads under ~200
25
25
  - An existing over-limit reference is frozen per invariant 4: shrink or stay level; growth blocks. Additions to a frozen reference are funded by consolidating existing text in the same file.
26
26
  - Append-only ledgers are structurally excluded: `references/source-register.md` grows by contract (append-only, supersede-by-pointer, rows never edited), so a line cap would block the ledger discipline itself; the gate skips it and prints a visibility token when it is over the figure. Residual risk, accepted under the same trusted-contributor model as the entrypoint gate: a prose file named `source-register.md` would dodge the cap — review owns that shape.
27
27
  - A new reference over 100 lines must be structured with `##` sections so chunked reads and greps can navigate it; a heading-less long file draws an advisory token (never a block). A table-of-contents list is optional — section structure is the invariant, not a TOC block.
28
+ - **Funding an addition by trimming prose means editing text that may be pinned — resolve the pins before rewriting, not after.** The ratchet's per-file freeze makes every addition to a legacy surface a rewrite of something else in the same file, and load-bearing sentences are pinned in two places: declaratively in `../../skill-extraction-workflow/scripts/contract-anchors.tsv`, which the fast repo gate checks, and as `grep -Fq` assertions inside owner suites, whose break a full lane run reports half an hour later. Read BOTH for the file you are about to trim — `awk -F'\t' '$2 == "<path>"' skills/skill-extraction-workflow/scripts/contract-anchors.tsv` lists the registry rows pinning it, and `grep -rn 'grep -Fq' skills/*/scripts/*.sh` finds the suite assertions — because a funded trim that silently retired two pinned wait-contract obligations was reported by the slow lane only, long after the edit. A pinned sentence may be reworded only together with whatever pins it, in the same landing.
28
29
  - Authoring anti-patterns (verified against the official skill-authoring checklist, see verdicts below): time-sensitive facts outside an explicit old-patterns section; inconsistent terminology for one concept; abstract examples where a concrete input/output pair fits; Windows-style paths; unexplained constants; scripts that defer error handling to the model instead of solving it.
29
30
 
30
31
  ## Retirement and relocation signal (usage census)
@@ -89,3 +89,25 @@ Firing point: **producing or first-publishing a reader-facing deliverable is an
89
89
  但按该规则自己的定义,**针对外部源的缺口清单就是它所说的 findings 回合**:一旦产出,charter 就只能事后补写。
90
90
 
91
91
  观测实例:一轮里先产出四条「外部有我们没有」的缺口,之后才 invoke 提炼工作流;改前的触发词表逐字检索该轮实际措辞得零命中。
92
+
93
+ ## The review-chain case — the owning skill is not loaded where the situation arises
94
+
95
+ The same-class-recurrence rule is owned by this workflow, but the situation it governs —
96
+ findings coming back round after round — arises inside a `code-review` chain, where this
97
+ skill is typically never loaded. Naming the owner in prose therefore never made it fire.
98
+ The trigger sits at the transition instead:
99
+
100
+ - **The controller must raise it, not the reader:** when a round returns findings and the
101
+ history it carries already holds one, the gate adds `recurring_findings_design_check` to
102
+ that round's required self-review triggers and `decide_keep_delete_narrow_replace` to its
103
+ allowed actions, in the round's own envelope
104
+ (`../../code-review/references/staged-review-contract.md`). What discharges it is the
105
+ `keep` / `delete` / `narrow` / `replace` decision the owning rule defines, ratified by a
106
+ risk owner other than the one proposing it.
107
+ - **What it counts is bounded by what a receipt carries:** the chain the controller is
108
+ handed, plus the predecessor a succession names. Succession does not compose, so a third
109
+ chain opened fresh carries no history and the controller claims none; from there the
110
+ recurrence is the round's own record to keep.
111
+ - **It over-fires by design:** two findings rounds need not share a risk class, so the
112
+ question is sometimes inapplicable. Answering an inapplicable question costs a line; the
113
+ round a missed design question costs does not.
@@ -679,3 +679,13 @@ The pending classification above is superseded by the executed source comparison
679
679
  | A value used as a redaction NEEDLE is rejected when it is only structure: a home directory of `/` is a legitimate environment and a catastrophic needle, because replacing it rewrites every separator in the text and disables every rule that runs after it | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Raised by the landing review against the normalization the previous round added: gathering every spelling of the home directory is right, but a spelling that carries no content is not a path to elide. With `HOME=/` -- root, or an arbitrary-uid container -- the needle set contained `/`, the replacement ran before the URL rules, and the excerpt came out mangled with its credentials intact. The general shape is that a needle derived from the environment needs a content test, not only a presence test. RED-baseline (applied): a row invoking the wrapper with `HOME=/` reds without the content test and greens with it, and removing only that test reds that row alone. |
680
680
  | A filter over free text in a persisted artifact is replaced by having no free text: "nothing secret-shaped survives" is not decidable over arbitrary text, so an adversarial reviewer can always spell one more escape, and the terminal state is a constant the input cannot influence rather than a filter that keeps growing | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. The measurement is the row: eight review chains after the input was already narrowed to CLI-authored error messages, each found a different escape -- an unlisted key name, an assignment form, URL userinfo, a password containing the separator, a fixture whose own shape tripped a neighbouring gate, an escaped quote closing a quoted value early, an uppercase scheme, a separator-only home used as a needle. Every one was real and none was derivable from the previous one. Two intermediate diagnoses were wrong on the way and are recorded above: swapping a key-name list for an assignment-shape rule was called an invariant change and was another enumeration, and narrowing the input was called sufficient when it only slowed the rate. What ends the class is that the receipt now carries a constant and the transport's output stays in the preserved run directory. The property is stated as equality with that constant, which a test can hold, instead of the absence of a list of shapes, which no test can. RED-baseline (applied): echoing the extracted message into the receipt reds the invariant row, and the unmutated control is green. Cost, recorded because the next round should be able to weigh it: each chain was roughly twelve minutes of wall clock, and the merge gate accepts no open challenge finding, so there was no landing state that carried the residue. |
681
681
  | A fixture string is read by every scanner in the repository, not only by the suite it belongs to, so it is chosen to be inert under all of them: a host:port that exists only inside a quoted test payload still reads as a listening port to a lane-isolation scanner | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Fourth guard this round to reject the round's own test data, after the credential scanner, the egress tripwire and the public-sanitization gate. The userinfo fixture carried a port it never needed, and the parallel-lane isolation scanner reads any host:port in a lane member as evidence that concurrent suites could race on it. Dropping the port exercises the same wrapper behaviour. Recorded as one rule with the three before it: the cost of learning this one guard at a time was a full verification cycle each, and the cheaper order is to sweep every local gate after touching a fixture, before spending a review chain on the candidate. RED-baseline (applied): `test_lane_isolation.py` reds on the ported form and greens on the bare host, with the wrapper suite green either way -- which is why the suite alone was not evidence. |
682
+ | A rule that lives in a skill the situation never loads does not fire, however well it is written: the controller that runs the review chain raises the design question itself, inside that round's own envelope, and counts only the history a receipt carries | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md`. Both halves were registered deferrals, reproduced against the current controller before any code was written, each probe paired with a control leg. `recurring_findings_design_check` joins the required self-review triggers, with `decide_keep_delete_narrow_replace` among the allowed actions, whenever a findings round's own history already holds one — an earlier round of this chain, or the predecessor a succession names. `claim_strength` becomes a required self-review concern at build and release depth, owed before round 1 because that is the only point in a round where correcting an overstated claim is free. Applied-mutation RED baseline, differential: control 274 ok / 0 FAIL; removing the succession carry fails exactly `findings after a predecessor chain that also returned findings raise the design check` (273 ok / 1 FAIL); exempting `claim_strength` from the plan-coverage check fails exactly `a plan that skips the claim-strength walk fails before any provider runs` (273 ok / 1 FAIL); no non-owning assertion moves in either mutant. Two control legs ship inside the suite so a FIRST findings round is proved not to raise the trigger. The additions crossed the 500-line reference cap, so the wording-only exception moved verbatim into `code-review/references/wording-only-review.md` (the ratchet's own split-by-subtopic remedy) and the entrypoint's cumulative-budget paragraph became a pointer whose every number already lives in `code-review/references/timeout-auth-and-capabilities.md`; entrypoint body words end 8 below base. Round-1 review (kimi) returned one P2 on exactly that relocation — the numbers were delegated to a file outside the packet with nothing pinning them there — so the three literals are now contract anchors; the applied deletion mutation on one of them turns the anchor gate red (1 of 13) and the unmutated control runs green. Adding a required concern is a repository-wide compatibility event: five suites carried review-plan fixtures that omitted it and only the full lane found them, so the lane is run to green before a review chain opens rather than after. Supporting evidence: `code-review/scripts/review_gate.py`, `code-review/scripts/test_review_gate.sh`, `code-review/references/staged-review-contract.md`. |
683
+ | The same-class-recurrence rule states where it fires and what the firing controller is allowed to count, because the owner skill is not loaded at the transition where the situation arises | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/firing-point-placement.md#The controller must raise it, not the reader | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The rule is owned here, but the situation it governs arises inside a review chain where this skill is never loaded, so naming the owner in prose never made it fire; the firing point moves onto the transition and the statement of it lands in `references/firing-point-placement.md`, the reference the entrypoint already points at for firing-point mechanics, because the attention-budget ratchet holds this entrypoint at its base measure. The reference records what discharges the trigger (the keep/delete/narrow/replace decision, ratified by a risk owner other than the one proposing it), the bound on what the controller may count (this chain plus the predecessor a succession names; succession does not compose, so a third chain opened fresh carries none), and that the trigger over-fires by design. RED baseline is the controller's own suite: with the succession carry removed, the chain this text describes stops raising the check and exactly that assertion fails (273 ok / 1 FAIL against a 274 ok / 0 FAIL control), with no other assertion moving. |
684
+ | A gate that forces evidence to be bought on an artifact that will never land is charging for the wrong thing: a chain ends where the candidate moves, so the round it ended on is whichever round came last -- and requiring that round to be a challenge only moved the spend, never the proof | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md`. A succession may now carry a chain whose terminal receipt is its REVIEW, not only its challenge. Cost of the old shape, recorded as the author's measurement of a prior round rather than as anything this packet can reproduce: a fix applied straight after a review ended the chain by moving the owner digest, and reaching the post-fix candidate then required spending the chain's challenge on the pre-fix candidate first, so challenges were bought on candidates that never landed. Exempted class, stated behaviourally: a challenge on a candidate that will never land, which carries no evidence about the one that does. Every other binding holds -- the candidate must still have moved, succession still does not compose, the per-chain budget is untouched, and this path spends fewer rounds than the old one. What bounds it is the receipt's own arithmetic, and review named that for what it is: a FORGERY guard, not a history check. A genuine round-1 review reads the same whether its chain later ran a challenge or not, so a caller who spent the challenge and presents only the review is accepted, and the successor inherits no challenge focuses -- one that chain did spend can be spent again. The round had claimed that refusal in its acceptance and 'proved' it with a receipt shape the controller never emits; the claim is withdrawn rather than mechanised, because no check at the succession call site can close an omitted-history gap that the rest of this contract already declares. Controller comment, contract text, probe name and acceptance criterion all now say only what is enforced. Applied-mutation RED baseline, differential and attributed to the guard rather than to a diagnostic string: reverting the whole change reds the new probes only because the base rejects EVERY review predecessor, which proves nothing about the arithmetic guard, so the recorded mutation disables that guard ALONE -- review predecessors still admitted, their chain-ended arithmetic no longer checked. Under it exactly one assertion fails, `a succession rejects a forged review receipt whose own arithmetic says its chain is spent`, and it fails because the succession was ACCEPTED; the unmutated control runs the suite green. The same round adds `--print-required-concerns`. Parity is proved over ANSWERS and over REFUSALS, the second only after challenge found the export answering for inputs the enforcer rejects -- a risk tag containing whitespace, an empty tag, an over-long tag -- which is the divergence the export exists to remove, reproduced against the shipped binary and now refused identically on both sides -- asserted on BOTH, after challenge observed the first parity probes invoking only the printer, so removing the enforcer's own rejection would have left them green. Removing it in an isolated clone now reds exactly those three assertions and nothing else. The clone matters: three earlier attempts mutated the worktree and restored it at the end of the same command, and one such restore -- from a backup another still-running task had taken while the file was already mutated -- put a controller with its tag validation stripped back into the tree, caught only because the task's output was shorter than expected. Destructive probes run on a copy. Challenge then found the parity checks themselves discarding exit status: this suite runs without errexit, so a command substitution swallows it and a printer that emitted the right concerns before failing would still have satisfied a non-empty check. The valid calls now assert their status, and a mutant that prints correctly then returns non-zero reds exactly those two assertions. The same failure-propagation blind spot appeared twice in one round -- the first parity probe had instead enabled errexit, aborting the suite at its first non-zero command with zero FAIL lines -- so both directions are now covered by assertions rather than by shell defaults. Asked to sweep the class rather than patch the instance, the next challenge found the remaining one -- a printer call NESTED inside a comparison, whose status no assignment could capture. All four printer invocations in the suite now capture status; a mutant failing only for explicit release depth with no risk tag reds exactly the assertion that covers it, and nothing else. The answer direction is proved in both senses: a plan built from it is accepted, and dropping each printed concern in turn turns the gate red -- at BOTH depths, after the challenge observed that proving it only at build left the release/high-risk branch free to diverge with every test green. |
685
+ | A consumer that keeps its own copy of a set the controller owns drifts the moment that set changes, and the suite runner aborts at its first failing target so the drift surfaces rounds later -- while relocating a working pin mechanism to buy faster feedback costs more than the latency it buys | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. This owner's wrapper suite derives the required concern set from the controller that enforces it instead of holding a copy. RED-baseline, paired control differing in exactly one variable: with one concern added to the controller's build stage, the base fixture fails `self_review_incomplete` on the no-independent-reviewer assertion (rc 1) while the derived fixture on this candidate passes (rc 0); with the mutation removed the derived fixture passes again. Second thread, withdrawn rather than landed: the round first relocated twenty prose pins from a slow suite into the fast registry, and five consecutive challenge findings landed inside the matching normalisation that relocation required -- this repo's own cue to question the capability instead of patching it again -- while the change additionally applied looser whitespace-insensitive matching to fourteen pre-existing anchors that never asked for it. The four mechanism files are byte-identical to base. What lands from that thread is the write-side norm that sends an author to BOTH pin surfaces, with a command for each that was run before it was written down -- the first draft shipped a registry lookup that scanned the wrong scripts directory and returned nothing, which challenge caught; the replacement lists a file's anchor ids and was verified against a file that has them -- before a budget-funded trim; the latency that motivated the relocation is left to its own change, where wiring the owner suite into the fast gate buys the same feedback with no new matching semantics. |
686
+ | 改写既有文档时新写的句子不得把用词水位抬到原文之上——病根是编辑落笔用的是自己的词库而不是宿主文档的;查法是把本轮新句单独拎出、逐个术语查它在原文里出现过没有、没出现的换成原文说法或当场白话解释 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/SKILL.md#改写时新句用词不得高出原文水位、抬高读者门槛 | `updated` | Owner key `tighten-doc/SKILL.md`。Observed failure:一篇面向业务读者的大白话协作文档被逐句就地修订约 25 次,每一次替换单独看都正确;收尾回读的既有触发词是「分享 / 发布之前」,而就地编辑的每一刀落地即发布,永远到不了那个时刻,整体回读因此一次也没跑;新写的句子同时带进了编辑自己的行话,文档 owner 的反应是「改成看不懂的了」,返工重写了全部新句。两个缺陷都只在跨刀整体读时显形。证据边界照实说明:prose 收尾规则在本仓没有可执行的行为 oracle,锚点钉住的是规则在场与措辞,不是模型行为差分,不按行为实测记。owner-generalization map 在 `specs/125-doc-closeout-and-register-drift/evidence/owner-map.md`(十个 owner 逐条 updated/unchanged/routed/not-applicable)。RED-baseline(applied,differential):把该锚定句改成非规范措辞,`register-firing-path-resolution.rb` rc=1 并点名 `source-register.md` 的这一行与该 locator;控制组与恢复后均 rc=0(落地时重跑为 527 locators resolved,与同轮提交的 mutation-walk.txt 一致;518 是本行初稿时的捕获值,账本此后被追加过),同一 diff 的其余检查两侧不变。 |
687
+ | 就地编辑一篇已发布文档时不存在「分享前」这一刻——每一刀落地即发布——所以挂在分享前的收尾回读永远不触发;触发点顺延到本轮最后一次写操作之后,收工前必须整体回读一遍 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/closeout-reread.md#没有「发布前」这一刻**:每一刀落地即发布,读者随时可能正在读。触发点顺延到**最后一次写操作之后**:收工前必须把整篇(或受影响那一面)整体回读一遍 | `updated` | Owner key `tighten-doc/SKILL.md`。Observed failure:一篇面向业务读者的大白话协作文档被逐句就地修订约 25 次,每一次替换单独看都正确;收尾回读的既有触发词是「分享 / 发布之前」,而就地编辑的每一刀落地即发布,永远到不了那个时刻,整体回读因此一次也没跑;新写的句子同时带进了编辑自己的行话,文档 owner 的反应是「改成看不懂的了」,返工重写了全部新句。两个缺陷都只在跨刀整体读时显形。证据边界照实说明:prose 收尾规则在本仓没有可执行的行为 oracle,锚点钉住的是规则在场与措辞,不是模型行为差分,不按行为实测记。owner-generalization map 在 `specs/125-doc-closeout-and-register-drift/evidence/owner-map.md`(十个 owner 逐条 updated/unchanged/routed/not-applicable)。细节落在同包的 `tighten-doc/references/closeout-reread.md`(触发点、这一遍要拿出的证据、三类不能顶替它的东西、语域漂移查法),entrypoint 只留触发与硬规则并因此净缩小(bytes 49905→49454,body words 9822→9806)。RED-baseline(applied,differential):把该锚定的规范列表行改写掉,`register-firing-path-resolution.rb` rc=1 并点名该 locator;控制组与恢复后 rc=0。这一行与上一行分别钉住本轮两条规则,任何一半**被锚定的那段字面**被删或被改写都会红,不靠同一个锚代管两件事;但改动规则的**适用条件**(例如给它加一个前置条件)不动锚内任何字,闸检不出——这一条与本表下方那行同口径,不作更强声称。锚点特意跨过承重从句——「没有发布前这一刻 / 每一刀落地即发布 / 顺延到最后一次写操作之后 / 收工前必须」连成一条字面量,删掉其中任一从句而只留末句都会让 locator 失配;这是评审指出的绕过(只锚末句时,删掉前面的适用条件仍能过闸)。覆盖边界照实说明:entrypoint 里那句一行复述不单独钉锚,本行不声称闸能护住它。 |
688
+ | 语域漂移的查法本身是规则的一部分:新句里原文没出现过的术语是候选,每个候选必须换成原文已有的说法或当场用一句白话解释,原样留着不算处理 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/closeout-reread.md#每个候选必须二选一:换成原文已经在用的说法,或当场用一句白话把它解释掉 | `updated` | Owner key `tighten-doc/SKILL.md`。查法与触发点同属本轮那条收尾规则,分行钉锚是**归属选择**不是解析器限制——守卫会把逗号分隔的多个 locator 拆开各自解析(`register-firing-path-resolution.rb` 的 locator 拆分与 multi_bad/multi_ok 两条用例);分行是为了让红的时候直接指到是哪一步被动了。促成拆分的实测是:只钉规则句时,把查法三步删掉仍能过闸。RED-baseline(applied,differential):删掉该规范列表行,`register-firing-path-resolution.rb` rc=1 并点名该 locator;控制组与恢复后 rc=0。覆盖边界:本行只钉「候选术语必须处理」这一步;隔离新句与对照改前文本两步由下两行分别钉住。 |
689
+ | 语域漂移的查法要先隔离本轮新句:必须把新写的句子单独拎出来单独看,混在整篇里读就看不出用词是谁的 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/closeout-reread.md#必须把本轮**新写的句子**单独拎出来单独看 | `updated` | Owner key `tighten-doc/SKILL.md`。与上三行同属本轮那条收尾规则,分行是归属选择——守卫支持一条 firing-path 里放多个 locator 并各自解析,分行只是为了让红时能指到具体那一步。RED-baseline(applied,differential):删掉或改写该规范列表行,`register-firing-path-resolution.rb` rc=1 并点名该 locator;控制组与恢复后 rc=0。这一行来自人工授权轮的评审:三个 locator 都不覆盖「隔离新句」,删掉它仍能过闸。 |
690
+ | 语域漂移的候选判定必须以改动之前的文本为对照:不得拿改完的文档做对照,否则新词已经在里面,永远查不出候选 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/closeout-reread.md#逐个术语查它**在改动之前的文本里**出现过没有;不得拿改完的文档做对照,否则新词自己就在里面,永远查不出候选。没出现过的就是候选 | `updated` | Owner key `tighten-doc/SKILL.md`。与上三行同属本轮那条收尾规则,分行是归属选择——守卫支持一条 firing-path 里放多个 locator 并各自解析,分行只是为了让红时能指到具体那一步。RED-baseline(applied,differential):删掉或改写该规范列表行,`register-firing-path-resolution.rb` rc=1 并点名该 locator;控制组与恢复后 rc=0。这一行同样来自人工授权轮。锚点从「逐个术语」起,覆盖遍历范围、对照对象、禁令与候选定义四段连写:上一轮 challenge 实测出,只钉禁令时把「在改动之前的文本里」换成「在术语表里」,五条 locator 全部照旧匹配而查法已废;扩锚后该替换直接失配。同类 finding 已连出三轮(只护一半规则 / 锚点截断 / 换掉正面对照对象),据此在账本里把结论写死:substring 锚钉的是字面不是语义,它保证的只是**被锚定的那段字面**被删或被改写时会红;它不保证规则被改写时会红——实测:把查法的引导词改成「以下查法仅在用户明确要求时执行」,五条 locator 全部照常匹配而整条查法已成可选。适用条件与语义完整性由评审与人读负责,本行不作此声称。 |
691
+ | 钉住散文规则的锚点闸必须自带能失败的测试:删掉被锚定的规则行必须让闸变红并点名该 locator,而锚点之外的文字被掏空时闸不得报红——后一条把「锚钉字面不钉语义」这条边界写成被执行的事实,而不是账本里的一句声明 | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`(本轮未改,改动落在同包的该测试脚本)。Observed failure:本轮连续三轮 challenge 都指出同一件事——变异记录依赖的守卫不在按 diff 装的评审包里,包内无法核验;先后用「路径+blob 哈希」和「候选内的风险责任人接受书」作答都被驳回,后者尤其是错的:被评审的东西不能自己给自己发授权。RED-baseline(applied,differential,隔离路径归因):完整套件 fail-fast,任何守卫变异都先撞红最早受影响的既有用例,新增两条根本跑不到——本轮 challenge 正是据此推翻了先前那条「改诊断串」的证据,那次变异撞的是既有的 reworded-anchor 用例,什么也没归因到。改用**每条用例各一份单例副本**:把 `next if body.include?(anchor)` 换成整行相等时,边界用例红、删除用例绿;换成从不报缺失锚点时,删除用例红、边界用例绿;控制组与恢复四格全绿。两次不翻转的格子都是实跑观测;探针已提交为 `specs/125-doc-closeout-and-register-drift/evidence/attribution-probe.sh`,一条命令重跑整张矩阵;它在被测提交的一次性 detached worktree 里执行,调用者的检出只读不写(首版写进活动检出,被本轮评审判为 P1 并已重写)。三次被推翻的归因尝试(改诊断串撞到既有用例/两条用例同副本致后一条不执行/副本临时不可复核)连同原因一并留在走查里。两次变异与原始输出见 `specs/125-doc-closeout-and-register-drift/evidence/mutation-walk.txt` 的第二段。 |
@@ -14,3 +14,7 @@ coverage-tier-provenance skills/testing-strategy/references/test-code-authoring-
14
14
  pairwise-trigger-range skills/test-artifact-management/references/classical-test-design-techniques.md 2-way 累计触发 53–97% 071-chainC-r1f4: NIST SP 800-142 empirical range, externally verified (specs/071 source-verification)
15
15
  bva-two-vs-three-value skills/test-artifact-management/references/classical-test-design-techniques.md 2-value(边界 + 下一格)和 3-value(边界 + 两侧) 071-chainC-r1f4: ISTQB v4 BVA variant definitions, externally verified (specs/071 source-verification)
16
16
  merge-side-ledger-binding-failclosed skills/skill-extraction-workflow/scripts/review_ledger_binding.py return 0 if args.allow_unevaluated else 2 084-r4f1: the no-base fail-closed branch; flipping it to an unconditional 0 restores a gate that passes having checked nothing
17
+ lane-budget-default-range skills/code-review/references/timeout-auth-and-capabilities.md defaults to 2400 seconds and accepts 5 to 3600 125-r1f1: the entrypoint now points here for the cumulative lane budget instead of restating it; dropping or drifting the default/range would silently strip the bound from both surfaces
18
+ lane-budget-mode-minimums skills/code-review/references/timeout-auth-and-capabilities.md 21 total seconds for review and 16 125-r1f1: the per-mode fail-closed minimums the entrypoint delegates here; without the pin the fail-closed limit can vanish with no suite failing
19
+ lane-budget-reserved-seconds skills/code-review/references/timeout-auth-and-capabilities.md while reserving ten controller 125-r1f1: the reserved controller seconds in the per-invocation division the entrypoint delegates here
20
+ recurrence-rule-owner-pointer skills/code-review/references/staged-review-contract.md `../../skill-extraction-workflow/SKILL.md` owns that rule 125-binding-r2f1: the chain-bound reader cannot load the owning skill, so this pointer is the only route to the rule that discharges the trigger; a wrong depth resolves inside code-review and no link checker sees a backticked path