@ccoalm/ccl-skills 0.18.10 → 0.18.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/hooks/headless-background-stop.sh +18 -0
  2. package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +7 -0
  3. package/dist/assets/marketplace/plugins/ccl-skills/hooks/host-input.py +155 -8
  4. package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-post-merge-cleanup.sh +9 -4
  5. package/dist/assets/marketplace/plugins/ccl-skills/hooks/task-entry.sh +24 -3
  6. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_headless_background_stop.sh +150 -0
  7. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_proposed_next.py +71 -0
  8. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_post_merge_cleanup.sh +47 -0
  9. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_task_entry.py +63 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +7 -1
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +2 -2
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +77 -6
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +112 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/references/multi-agent-delegation-playbook.md +1 -0
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +4 -0
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/worktree-mechanics.md +10 -3
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +2 -2
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +6 -1
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-lifecycle-handoff.md +1 -1
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +2 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +15 -0
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-sync-pointers.sh +11 -4
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-paired-eval.py +1546 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +140 -1
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +4 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_sync_pointers.sh +13 -4
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +12 -1
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_skill_paired_eval.py +1052 -0
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_teardown_guard_pins.sh +354 -0
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/SKILL.md +8 -108
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/merge-and-teardown.md +47 -0
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/pre-merge-landing-checks.md +73 -0
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/shared-branch-rebase.md +1 -1
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/scripts/test_worktree_sweep.sh +1 -1
  35. package/dist/assets/release.json +65 -30
  36. package/package.json +1 -1
@@ -80,6 +80,69 @@ class TaskEntryTests(unittest.TestCase):
80
80
  self.assertIn('A report alone does not complete it', context)
81
81
  self.assertIn('required user decision or unavailable authority/resource', context)
82
82
 
83
+ def test_host_task_notification_turn_keeps_only_the_boundary(self):
84
+ # The host also runs UserPromptSubmit on turns it starts itself; a background
85
+ # completion arrives as one <task-notification> envelope. SessionStart keeps the
86
+ # routing list in context (also after compaction), so that turn drops only the
87
+ # list; the skill-loading and unfinished-work boundary appears nowhere else and
88
+ # matters most when a background check has just failed.
89
+ expected, _ = self.run_hook(prompt='Add a feature')
90
+ full = expected['hookSpecificOutput']['additionalContext']
91
+ for prompt in ['<task-notification>\n<task-id>b1</task-id>\n<status>failed</status>\n</task-notification>',
92
+ '\n <task-notification><task-id>b2</task-id></task-notification>']:
93
+ with self.subTest(prompt=prompt[:30]):
94
+ output, error = self.run_hook(prompt=prompt)
95
+ self.assertEqual(error, '')
96
+ context = output['hookSpecificOutput']['additionalContext']
97
+ self.assertIn('unrun, failed or inconclusive verification is unfinished work', context)
98
+ self.assertIn('Before task-specific investigation or substantive analysis', context)
99
+ self.assertTrue(context.startswith('<ccl-task-entry>') and context.endswith('</ccl-task-entry>'))
100
+ self.assertNotIn('<!-- ccl:entry-routing:start -->', context)
101
+ self.assertNotIn('**product-rd-workflow**', context)
102
+ self.assertLess(len(context), len(full) // 2)
103
+ # Controls: a human prompt that merely mentions the envelope still gets the
104
+ # entry, as does input the hook cannot parse.
105
+ for prompt in ['Why did <task-notification> fire twice?', 'task-notification arrived',
106
+ '<task-notification><task-id>b3</task-id></task-notification>\nWhat does this mean?',
107
+ '<task-notification> pasted without its closing tag',
108
+ '<task-notification>first</task-notification>\nCompare these.\n'
109
+ '<task-notification>second</task-notification>']:
110
+ with self.subTest(prompt=prompt):
111
+ output, _ = self.run_hook(prompt=prompt)
112
+ self.assertEqual(output, expected)
113
+
114
+ def test_unreadable_hook_input_keeps_the_entry(self):
115
+ expected, _ = self.run_hook(prompt='Add a feature')
116
+ for raw in ['', 'not json', json.dumps(['<task-notification>']),
117
+ json.dumps({'prompt': 42}), json.dumps({'prompt': '<task-notification>' + 'x' * 2_000_000}),
118
+ '[' * 500000 + ']' * 500000]:
119
+ with self.subTest(raw=raw[:30]):
120
+ with tempfile.TemporaryDirectory(prefix='ccl-task-entry-') as directory:
121
+ root = Path(directory)
122
+ (root / 'hooks').mkdir()
123
+ (root / 'agent-context').mkdir()
124
+ shutil.copyfile(ROOT / 'hooks/task-entry.sh', root / 'hooks/task-entry.sh')
125
+ (root / 'agent-context/session-start.md').write_text(
126
+ (ROOT / 'agent-context/session-start.md').read_text())
127
+ result = subprocess.run(['bash', str(root / 'hooks/task-entry.sh')], input=raw,
128
+ text=True, capture_output=True, timeout=5,
129
+ cwd=root, env={'PATH': os.environ['PATH']})
130
+ self.assertEqual(result.returncode, 0, result.stderr)
131
+ self.assertEqual(json.loads(result.stdout), expected)
132
+ # A closed stdin is read as empty input, never a silent hook.
133
+ with tempfile.TemporaryDirectory(prefix='ccl-task-entry-') as directory:
134
+ root = Path(directory)
135
+ (root / 'hooks').mkdir()
136
+ (root / 'agent-context').mkdir()
137
+ shutil.copyfile(ROOT / 'hooks/task-entry.sh', root / 'hooks/task-entry.sh')
138
+ (root / 'agent-context/session-start.md').write_text(
139
+ (ROOT / 'agent-context/session-start.md').read_text())
140
+ result = subprocess.run(['bash', '-c', 'bash "$1" <&-', 'x', str(root / 'hooks/task-entry.sh')],
141
+ text=True, capture_output=True, timeout=5,
142
+ cwd=root, env={'PATH': os.environ['PATH']})
143
+ self.assertEqual(result.returncode, 0, result.stderr)
144
+ self.assertEqual(json.loads(result.stdout), expected)
145
+
83
146
  def test_missing_or_invalid_source_is_observable_and_fail_soft(self):
84
147
  for source, missing in [('', True), ('not a routing document', False),
85
148
  ('<!-- ccl:entry-routing:end -->', False),
@@ -68,6 +68,8 @@ const OPENCODE_HOOK_BINDINGS = Object.freeze({
68
68
  "owner-dispatch-stop.sh": "event:session.idle/session.status",
69
69
  "skill-extraction-gate-stop.sh": "event:session.idle/session.status",
70
70
  "proposed-next-stop.sh": "event:session.idle/session.status",
71
+ // Inert here: it acts only for a Claude Code sdk entrypoint, which runHook never passes on.
72
+ "headless-background-stop.sh": "event:session.idle/session.status",
71
73
  })
72
74
 
73
75
  type HookJson = {
@@ -105,10 +107,13 @@ function runHook(root: string | null, script: keyof typeof OPENCODE_HOOK_BINDING
105
107
  if (!existsSync(path) || !lstatSync(path).isFile() || lstatSync(path).isSymbolicLink()) {
106
108
  return { status: "missing", message: `${script} is missing from the OpenCode hook runtime` }
107
109
  }
110
+ // A Claude Code entrypoint inherited from a parent process does not describe this OpenCode
111
+ // session, so no hook here may act on it.
112
+ const { CLAUDE_CODE_ENTRYPOINT: _inheritedEntrypoint, ...inherited } = process.env
108
113
  const result = spawnSync("bash", [path], {
109
114
  cwd,
110
115
  encoding: "utf8",
111
- env: { ...process.env, CLAUDE_PLUGIN_ROOT: root },
116
+ env: { ...inherited, CLAUDE_PLUGIN_ROOT: root },
112
117
  input: JSON.stringify(payload),
113
118
  maxBuffer: 256 * 1024,
114
119
  timeout,
@@ -664,6 +669,7 @@ export const CclSkills = async (context: {
664
669
  runHook(hooksRoot, "owner-dispatch-stop.sh", stopPayload, directory, 10_000),
665
670
  runHook(hooksRoot, "skill-extraction-gate-stop.sh", stopPayload, directory, 15_000),
666
671
  runHook(hooksRoot, "proposed-next-stop.sh", stopPayload, directory, 5_000),
672
+ runHook(hooksRoot, "headless-background-stop.sh", stopPayload, directory, 5_000),
667
673
  ]
668
674
  const reasons = results
669
675
  .filter((result) => result.output?.decision === "block" && typeof result.output.reason === "string")
@@ -34,10 +34,10 @@ Before completion or landing handoff, report the actual diff classification and
34
34
 
35
35
  ## Review continuation checkpoint
36
36
 
37
- Necessary in-scope review inherits the existing task authority, including review of a fix made after an earlier pass. Five renewed runs trigger a progress checkpoint, not an authorization request. At that checkpoint, and before each later renewed run:
37
+ Necessary in-scope review inherits the existing task authority, including review of a fix made after an earlier pass. Five renewed runs trigger a progress checkpoint, not an authorization request. The controller counts conclusive runs in the worktree receipt and, from the sixth, returns `continuation_checkpoint` in its output; a review that records no receipt is counted by you. At that checkpoint, and before each later renewed run:
38
38
 
39
39
  1. Check current scope and any explicit user stop, count, cost or time limit. Honor a host permission denial or unavailable resource through its normal recovery path; task authority never bypasses it. Ask only for a missing decision or authority that the next action actually needs.
40
- 2. Record what changed, the disposition of the previous findings, and what new evidence the next pass should obtain. When findings recur without progress, change the method or gather different evidence before calling again. Continue available in-scope repair and tests; park only work that needs an unavailable decision or resource.
40
+ 2. Record what changed, the disposition of the previous findings, and what new evidence the next pass should obtain. When findings recur without progress, change the method or gather different evidence before calling again. When the same class of finding keeps returning, even though each instance was fixed, decide whether the capability that keeps producing it should be narrowed or removed before patching it again, and record that decision. Continue available in-scope repair and tests; park only work that needs an unavailable decision or resource.
41
41
  3. Run the necessary review of the changed candidate without requesting permission per pass. Keep the tracked chain's own round ceiling and each invocation's timeout. A terminal chain stays terminal; use the owning lane's documented fresh-review or delta-pass path, and never reset an unchanged candidate's chain merely to obtain zero findings. Record the real pass history; an automatic pass is not a newly human-requested one.
42
42
 
43
43
  The checkpoint waives no review, challenge, evidence or readiness requirement. A missing conclusive pass or unresolved P0/P1 remains pending regardless of how many passes have run. If the current candidate already has valid review and dispositions, reuse them and finish.
@@ -5,6 +5,7 @@ from __future__ import annotations
5
5
 
6
6
  import argparse
7
7
  import errno
8
+ import fcntl
8
9
  import hashlib
9
10
  import json
10
11
  import math
@@ -1646,10 +1647,67 @@ def open_directory_without_links(path: str) -> int:
1646
1647
  return fd
1647
1648
 
1648
1649
 
1649
- def record_local_review(anchor: dict[str, Any] | None, result: dict[str, Any]) -> None:
1650
- """Best-effort local receipt of the last conclusive review; never fails it."""
1650
+ # development-completion.md "Review continuation checkpoint": the first review plus
1651
+ # five renewed runs. The count lives in the worktree's receipt because a count kept
1652
+ # in prose does not survive long sessions, compaction or a new session on the same
1653
+ # worktree, and a loop in which every round fixes something never looks stuck.
1654
+ REVIEW_CHECKPOINT_RUNS = 6
1655
+
1656
+
1657
+ def prior_local_review(receipt_fd: int) -> dict[str, Any]:
1658
+ try:
1659
+ descriptor = os.open(
1660
+ LOCAL_REVIEW_RECEIPT_FILE, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, dir_fd=receipt_fd
1661
+ )
1662
+ except OSError:
1663
+ return {}
1664
+ try:
1665
+ if not stat.S_ISREG(os.fstat(descriptor).st_mode):
1666
+ return {}
1667
+ data = os.read(descriptor, 65537)
1668
+ except OSError:
1669
+ # A prior receipt that cannot be read only loses the count, never the new receipt.
1670
+ return {}
1671
+ finally:
1672
+ os.close(descriptor)
1673
+ try:
1674
+ value = json.loads(data) if len(data) <= 65536 else None
1675
+ except (ValueError, RecursionError):
1676
+ return {}
1677
+ return value if isinstance(value, dict) else {}
1678
+
1679
+
1680
+ def conclusive_review_runs(prior: dict[str, Any], mode: Any) -> int:
1681
+ runs = prior.get("conclusive_runs")
1682
+ if not (isinstance(runs, int) and not isinstance(runs, bool) and runs >= 0):
1683
+ # A receipt written before the count existed still proves one run.
1684
+ runs = 1 if prior.get("mode") in ("review", "challenge") else 0
1685
+ return runs + (1 if mode in ("review", "challenge") else 0)
1686
+
1687
+
1688
+ def continuation_checkpoint(runs: Any, first_recorded_at: Any) -> dict[str, Any] | None:
1689
+ if not isinstance(runs, int) or isinstance(runs, bool) or runs < REVIEW_CHECKPOINT_RUNS:
1690
+ return None
1691
+ return {
1692
+ "conclusive_review_runs": runs,
1693
+ "first_review_recorded_at": first_recorded_at if isinstance(first_recorded_at, str) else None,
1694
+ "rule": "code-review references/development-completion.md#review-continuation-checkpoint",
1695
+ "before_next_run": [
1696
+ "record what changed and the disposition of the previous findings",
1697
+ "when the same class of finding keeps returning, even if each instance was fixed, decide whether the "
1698
+ "capability producing it should be narrowed or removed before patching it again",
1699
+ "check the remaining findings against the requester's own words; findings outside the request are "
1700
+ "dispositioned, not implemented",
1701
+ ],
1702
+ }
1703
+
1704
+
1705
+ def record_local_review(anchor: dict[str, Any] | None, result: dict[str, Any]) -> dict[str, Any] | None:
1706
+ """Best-effort local receipt of the last conclusive review; never fails it.
1707
+
1708
+ Returns the receipt it wrote, or None when nothing was recorded."""
1651
1709
  if not anchor:
1652
- return
1710
+ return None
1653
1711
  receipt = {
1654
1712
  "schema_version": 1,
1655
1713
  "head": anchor["head"],
@@ -1673,12 +1731,19 @@ def record_local_review(anchor: dict[str, Any] | None, result: dict[str, Any]) -
1673
1731
  # under it since then is not the one the anchor described.
1674
1732
  opened = os.fstat(git_fd)
1675
1733
  if [opened.st_dev, opened.st_ino] != anchor.get("git_dir_identity"):
1676
- return
1734
+ return None
1677
1735
  try:
1678
1736
  os.mkdir(LOCAL_REVIEW_RECEIPT_DIR, 0o700, dir_fd=git_fd)
1679
1737
  except FileExistsError:
1680
1738
  pass
1681
1739
  receipt_fd = os.open(LOCAL_REVIEW_RECEIPT_DIR, directory_flags, dir_fd=git_fd)
1740
+ # Serialize read-increment-replace so overlapping review and challenge
1741
+ # completions in one worktree cannot both read N and write N + 1.
1742
+ fcntl.flock(receipt_fd, fcntl.LOCK_EX)
1743
+ prior = prior_local_review(receipt_fd)
1744
+ receipt["conclusive_runs"] = conclusive_review_runs(prior, receipt["mode"])
1745
+ first = prior.get("first_recorded_at") or prior.get("recorded_at")
1746
+ receipt["first_recorded_at"] = first if isinstance(first, str) else receipt["recorded_at"]
1682
1747
  temporary = f".last-review.{os.getpid()}.{secrets.token_hex(8)}"
1683
1748
  file_fd = os.open(
1684
1749
  temporary,
@@ -1697,8 +1762,9 @@ def record_local_review(anchor: dict[str, Any] | None, result: dict[str, Any]) -
1697
1762
  dst_dir_fd=receipt_fd,
1698
1763
  )
1699
1764
  temporary = None
1765
+ return receipt
1700
1766
  except OSError:
1701
- pass
1767
+ return None
1702
1768
  finally:
1703
1769
  if temporary is not None and receipt_fd is not None:
1704
1770
  try:
@@ -5190,7 +5256,12 @@ def main(argv: list[str] | None = None) -> int:
5190
5256
  if time.monotonic() >= gate_deadline:
5191
5257
  apply_gate_timeout(result)
5192
5258
  return emit(result, 2)
5193
- record_local_review(review_anchor, result)
5259
+ receipt = record_local_review(review_anchor, result)
5260
+ checkpoint = continuation_checkpoint(
5261
+ (receipt or {}).get("conclusive_runs"), (receipt or {}).get("first_recorded_at")
5262
+ )
5263
+ if checkpoint:
5264
+ result["continuation_checkpoint"] = checkpoint
5194
5265
  return emit(result, 0)
5195
5266
  if completed.returncode == 0 and status in {"passed", "findings"}:
5196
5267
  payload.update(
@@ -4931,6 +4931,118 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
4931
4931
  check "a bare --diff-file review records no receipt" \
4932
4932
  '[ "$rc" = 0 ] && json_fields "$out" status=passed && [ ! -e "$contract_repo/.git/ccl-code-review" ]'
4933
4933
 
4934
+ # The continuation checkpoint is counted, not remembered: the receipt carries the
4935
+ # worktree's conclusive review runs, and the sixth (the first review plus five
4936
+ # renewed runs) brings the checkpoint into the gate's own output.
4937
+ counting_out="$(PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY' 2>&1
4938
+ import json, os, sys, review_gate
4939
+ root = os.path.realpath(os.path.join(sys.argv[1], "run-count"))
4940
+ gd = os.path.join(root, "gitdir")
4941
+ os.makedirs(gd)
4942
+ st = os.stat(gd)
4943
+ anchor = {"git_dir": gd, "git_dir_identity": [st.st_dev, st.st_ino], "head": "a" * 40, "worktree_clean": True}
4944
+ receipt_path = os.path.join(gd, "ccl-code-review", "last-review.json")
4945
+ runs = [review_gate.record_local_review(anchor, {"mode": mode, "status": "findings"})["conclusive_runs"]
4946
+ for mode in ["review", "challenge", "review", "complete", "review", "review", "challenge"]]
4947
+ assert runs == [1, 2, 3, 3, 4, 5, 6], runs
4948
+ first = json.load(open(receipt_path))["first_recorded_at"]
4949
+ assert review_gate.continuation_checkpoint(5, first) is None
4950
+ assert review_gate.continuation_checkpoint(True, first) is None
4951
+ cp = review_gate.continuation_checkpoint(6, first)
4952
+ assert cp["conclusive_review_runs"] == 6 and cp["first_review_recorded_at"] == first, cp
4953
+ assert any("narrowed or removed" in step for step in cp["before_next_run"]), cp
4954
+ assert any("own words" in step for step in cp["before_next_run"]), cp
4955
+ # A receipt written before the count existed still proves one run.
4956
+ json.dump({"mode": "review", "recorded_at": "2026-01-01T00:00:00Z"}, open(receipt_path, "w"))
4957
+ receipt = review_gate.record_local_review(anchor, {"mode": "review"})
4958
+ assert receipt["conclusive_runs"] == 2 and receipt["first_recorded_at"] == "2026-01-01T00:00:00Z", receipt
4959
+ # Corrupt or hostile prior values restart the count; they never cost the new receipt.
4960
+ for bad in ["not json", json.dumps({"conclusive_runs": True, "mode": "complete"}),
4961
+ json.dumps({"conclusive_runs": -3}), json.dumps([1]), "x" * 70000]:
4962
+ open(receipt_path, "w").write(bad)
4963
+ receipt = review_gate.record_local_review(anchor, {"mode": "review"})
4964
+ assert receipt is not None and receipt["conclusive_runs"] == 1, (bad[:20], receipt)
4965
+ # A prior receipt that is a link is never read through, and the write replaces the link.
4966
+ target = os.path.join(root, "elsewhere.json")
4967
+ json.dump({"conclusive_runs": 40, "mode": "review"}, open(target, "w"))
4968
+ os.unlink(receipt_path)
4969
+ os.symlink(target, receipt_path)
4970
+ receipt = review_gate.record_local_review(anchor, {"mode": "review"})
4971
+ assert receipt is not None and receipt["conclusive_runs"] == 1 and not os.path.islink(receipt_path), receipt
4972
+ assert json.load(open(target))["conclusive_runs"] == 40
4973
+ # A prior receipt the parser cannot finish (nesting too deep) is invalid, never fatal.
4974
+ real_loads = review_gate.json.loads
4975
+ def too_deep(*args, **kwargs):
4976
+ raise RecursionError("maximum recursion depth exceeded")
4977
+ review_gate.json.loads = too_deep
4978
+ open(receipt_path, "w").write("[" * 30000 + "]" * 30000)
4979
+ receipt = review_gate.record_local_review(anchor, {"mode": "review"})
4980
+ review_gate.json.loads = real_loads
4981
+ assert receipt is not None and receipt["conclusive_runs"] == 1, receipt
4982
+ # Overlapping completions cannot lose an increment because the whole
4983
+ # read-increment-replace runs under an exclusive lock on the receipt directory.
4984
+ # Checked where it matters, not by racing writers: when the controller opens the
4985
+ # prior receipt and when it replaces it, a second open of that directory cannot
4986
+ # take even a shared lock, and no lock call falls between the two, so the lock
4987
+ # is never released or changed in between.
4988
+ import fcntl
4989
+ receipt_dir = os.path.join(gd, "ccl-code-review")
4990
+ real_open, real_replace, real_flock = review_gate.os.open, review_gate.os.replace, review_gate.fcntl.flock
4991
+ def exclusively_locked():
4992
+ probe = real_open(receipt_dir, os.O_RDONLY | os.O_DIRECTORY)
4993
+ try:
4994
+ real_flock(probe, fcntl.LOCK_SH | fcntl.LOCK_NB)
4995
+ except BlockingIOError:
4996
+ return True
4997
+ finally:
4998
+ os.close(probe)
4999
+ return False
5000
+ seen = []
5001
+ def watched_open(path, flags, *args, **kwargs):
5002
+ if path == review_gate.LOCAL_REVIEW_RECEIPT_FILE and not flags & (os.O_WRONLY | os.O_RDWR):
5003
+ seen.append(("read", exclusively_locked()))
5004
+ return real_open(path, flags, *args, **kwargs)
5005
+ def watched_replace(src, dst, *args, **kwargs):
5006
+ if dst == review_gate.LOCAL_REVIEW_RECEIPT_FILE:
5007
+ seen.append(("replace", exclusively_locked()))
5008
+ return real_replace(src, dst, *args, **kwargs)
5009
+ def watched_flock(fd, operation):
5010
+ seen.append(("flock", operation))
5011
+ return real_flock(fd, operation)
5012
+ with open(receipt_path, "w") as handle:
5013
+ json.dump({"mode": "review", "conclusive_runs": 40, "first_recorded_at": "2026-01-01T00:00:00Z"}, handle)
5014
+ review_gate.os.open, review_gate.os.replace, review_gate.fcntl.flock = watched_open, watched_replace, watched_flock
5015
+ try:
5016
+ receipt = review_gate.record_local_review(anchor, {"mode": "challenge"})
5017
+ finally:
5018
+ review_gate.os.open, review_gate.os.replace, review_gate.fcntl.flock = real_open, real_replace, real_flock
5019
+ kinds = [kind for kind, _ in seen]
5020
+ assert kinds.count("read") == 1 and kinds.count("replace") == 1, seen
5021
+ read_at, replace_at = kinds.index("read"), kinds.index("replace")
5022
+ assert read_at < replace_at and seen[read_at][1] and seen[replace_at][1], seen
5023
+ assert "flock" not in kinds[read_at:replace_at], seen
5024
+ stored = json.load(open(receipt_path))
5025
+ assert stored == receipt and stored["conclusive_runs"] == 41, (stored, receipt)
5026
+ assert stored["first_recorded_at"] == "2026-01-01T00:00:00Z", stored
5027
+ print("run_count_ok")
5028
+ PY
5029
+ )"
5030
+ check "conclusive review runs are counted per worktree; the checkpoint starts at the sixth" \
5031
+ '[ "$counting_out" = run_count_ok ]'
5032
+
5033
+ mkdir -p "$contract_repo/.git/ccl-code-review"
5034
+ printf '{"mode":"review","conclusive_runs":5,"first_recorded_at":"2026-01-01T00:00:00Z"}\n' >"$receipt_file"
5035
+ reset_case passed unavailable unavailable
5036
+ out="$(run_contract_gate --mode review)"; rc=$?
5037
+ check "the sixth conclusive run carries the continuation checkpoint in the gate output" \
5038
+ '[ "$rc" = 0 ] && json_fields "$out" continuation_checkpoint.conclusive_review_runs=6 continuation_checkpoint.first_review_recorded_at=2026-01-01T00:00:00Z && [ "$(jq -r .conclusive_runs "$receipt_file")" = 6 ]'
5039
+ printf '{"mode":"review","conclusive_runs":3}\n' >"$receipt_file"
5040
+ reset_case passed unavailable unavailable
5041
+ out="$(run_contract_gate --mode review)"; rc=$?
5042
+ check "an earlier run carries no checkpoint" \
5043
+ '[ "$rc" = 0 ] && json_lacks "$out" continuation_checkpoint && [ "$(jq -r .conclusive_runs "$receipt_file")" = 4 ]'
5044
+ rm -rf "$contract_repo/.git/ccl-code-review"
5045
+
4934
5046
  # The completion checkpoint is what the pull-request reminder reads as disposed:
4935
5047
  # on the same whole-worktree candidate it replaces the review's receipt with a
4936
5048
  # passed one; a checkpoint that fails leaves the receipt as it was.
@@ -95,6 +95,7 @@ Controller-authored, non-shared edits stay in the owning local workflow. Use thi
95
95
 
96
96
  5. Clean up after integration.
97
97
  - After the reviewed ref is non-interim landed, sync the target checkout. If landing used a mutable source-branch merge without an atomic reviewed-ref guard, do not clean up yet; keep the branch as evidence.
98
+ - Every worktree removal below must run the pre-removal scan that the closeout section (`## 收尾`) of `worktree-isolation/references/merge-and-teardown.md` requires: `git -C <worktree-path> status --ignored -s` must exit 0, and a failed scan counts as no scan. Copy costly gitignored outputs the worker produced (long-running results, collected data, trained artifacts) into the target checkout before removal and drop regenerable ones (dependencies, build and test outputs, caches, logs); a worktree that `git status` reports clean still loses its gitignored files on `git worktree remove`. Record the scan exit status and each kept or dropped entry in `cleanup_proof`.
98
99
  - For local-commit-only handoff, prefer fast-forward or merge that preserves the exact reviewed SHA. First sync or fetch the target checkout, verify its current `HEAD` contains the reviewed SHA, and prove the landed content matches the reviewed delta with `git range-diff` or a scoped `git diff` showing no unexpected delta. Only after that content-equivalence proof and `git merge-base --is-ancestor <reviewed-sha> HEAD` both pass in the synced target checkout may the controller remove the isolated worktree and delete the local branch with `git branch -d` (not `-D`) from the same checkout. There is no remote source branch to prune.
99
100
  - For pushed-branch/MR handoff, check the platform merge mode first. If it is squash, rebase, cherry-pick, or another rewrite mode, use the non-ancestor path below. If the reviewed branch tip equals the recorded reviewed SHA and is a target ancestor, sync the target checkout, verify it contains the reviewed SHA, re-fetch and verify the local source branch tip still equals the reviewed SHA, remove the isolated worktree, delete the local branch with `git branch -d` from that synced target checkout, confirm the remote source branch was removed by the platform or delete it explicitly, run `git fetch --prune`, run `git worktree prune`, and verify `git worktree list` no longer shows the path and the remote source branch is absent. If the branch has post-review commits, review those commits before cleanup or keep the branch.
100
101
  - For squash, rebase, cherry-pick, or platform rewrites where ancestry does not prove integration, require a concrete non-ancestor proof before partial cleanup: platform merge state tied to the reviewed SHA, or an explicit patch-equivalence check such as `git range-diff` / scoped `git diff` showing the reviewed hunks are present on target with no unexpected differences. Under the default no-force-delete policy, remove only the clean worktree, keep the local branch, and report `branch retained: non-ancestor integration`; do not claim full cleanup unless `worktree-isolation` explicitly permits that merge-mode cleanup.
@@ -104,6 +104,10 @@ On hosts providing a current final message, `proposed-next-stop.sh` returns one
104
104
 
105
105
  The hook recognizes declarations, not authorization or actual task completion, and cannot force the model to follow through. OpenCode idle does not expose the required final-message evidence; its Stop behavior remains unverified.
106
106
 
107
+ A session nothing re-invokes gets no second chance at the in-flight-work rule below: Claude Code reports an `sdk` entrypoint for `claude -p` and the SDKs, the session ends at the stop, and the host stops its background tasks seconds later, so work waiting on them is lost.
108
+
109
+ - In such a session, `headless-background-stop.sh` blocks such a stop once per background task still running and names them: wait in the foreground for any the request depends on, without starting it again, and stop the rest with TaskStop or say why they can be dropped. It reads only the hook input, so it also fires when session persistence is off and the transcript-reading reminders above cannot run. Interactive sessions, which the host re-invokes when a task finishes, are left alone.
110
+
107
111
  ## Gate triggers and outcome contract
108
112
 
109
113
  An eligible next slice comes from an explicit status/task/acceptance source or active user continuation, is low-risk, local-only/already-authenticated, in accepted scope, clearly owned and verifiable with existing commands. It needs no destructive action, external purchase/financial commitment, production access, legal/compliance/product-strategy decision or high-impact architecture choice. Existing configured internal developer-self-use metered model/tool accounts are not an external purchase. Apply the following conditions to each action.
@@ -38,12 +38,19 @@ A per-line WIP branch stays local/private until the normal shared-branch-push an
38
38
 
39
39
  ## Closeout Cleanup
40
40
 
41
- At closeout, after the work lands or is abandoned, clean up the worktree and private branch:
41
+ At closeout, after the work lands or is abandoned, clean up the worktree and private branch. The canonical procedure is the closeout section (`## 收尾`) of `worktree-isolation/references/merge-and-teardown.md`, which also holds the integration evidence and the remote-branch rules; its guards apply to every removal:
42
+
43
+ - Before removing the worktree directory, you must run `git -C <path> status --ignored -s` from the primary checkout. It must exit 0; a failed scan counts as no scan, so stop and find the cause instead of reading empty output as nothing to keep. `git worktree remove` without `--force` still deletes gitignored files.
44
+ - Judge each listed entry by what it costs to recreate: drop regenerable outputs (dependency directories, build and test outputs, caches, logs) and copy costly ones (long-running results, collected data, trained artifacts) back to the primary checkout before removal. When unsure, treat an entry as costly.
45
+ - If a task with unfinished external side effects, such as a migration or a deployment, still runs from the worktree, wait for it to finish; never kill it to clean up.
46
+ - Never pass `--force` to `git worktree remove` or use `git branch -D`: a refusal means unmerged or uncommitted work. An abandoned line's branch usually fails `-d`; keep it and report it.
47
+ - Never delete a permanent or integration branch, or any branch whose name contains `release`, as part of this cleanup.
42
48
 
43
49
  ```bash
44
- git worktree remove <path>
50
+ git -C <path> status --ignored -s # must exit 0; copy costly ignored outputs out first
51
+ git worktree remove <path> # no --force
45
52
  git worktree prune
46
- git branch -d <line-branch>
53
+ git branch -d <line-branch> # -d, not -D
47
54
  ```
48
55
 
49
56
  ## Owner Routing
@@ -25,7 +25,7 @@ The read side already defends against oversized files (chunked reads under ~200
25
25
  - An existing over-limit reference is frozen per invariant 4: shrink or stay level; growth blocks. Additions to a frozen reference are funded by consolidating existing text in the same file.
26
26
  - Append-only ledgers are structurally excluded: `references/source-register.md` grows by contract (append-only, supersede-by-pointer, rows never edited), so a line cap would block the ledger discipline itself; the gate skips it and prints a visibility token when it is over the figure. Residual risk, accepted under the same trusted-contributor model as the entrypoint gate: a prose file named `source-register.md` would dodge the cap — review owns that shape.
27
27
  - A new reference over 100 lines must be structured with `##` sections so chunked reads and greps can navigate it; a heading-less long file draws an advisory token (never a block). A table-of-contents list is optional — section structure is the invariant, not a TOC block.
28
- - **Funding an addition by trimming prose means editing text that may be pinned — resolve the pins before rewriting, not after.** The ratchet's per-file freeze makes every addition to a legacy surface a rewrite of something else in the same file, and load-bearing sentences are pinned in two places: declaratively in `../../skill-extraction-workflow/scripts/contract-anchors.tsv`, which the fast repo gate checks, and as `grep -Fq` assertions inside owner suites, whose break a full lane run reports half an hour later. Read BOTH for the file you are about to trim — `awk -F'\t' '$2 == "<path>"' skills/skill-extraction-workflow/scripts/contract-anchors.tsv` lists the registry rows pinning it, and `grep -rn 'grep -Fq' skills/*/scripts/*.sh` finds the suite assertions — because a funded trim that silently retired two pinned wait-contract obligations was reported by the slow lane only, long after the edit. A pinned sentence may be reworded only together with whatever pins it, in the same landing.
28
+ - **Funding an addition by trimming prose means editing text that may be pinned — resolve the pins before rewriting, not after.** The ratchet's per-file freeze makes every addition to a legacy surface a rewrite of something else in the same file, and load-bearing sentences are pinned by literal from places that keep multiplying: `../../skill-extraction-workflow/scripts/contract-anchors.tsv`, the always-on pairs registered in `scripts/check-sync-pointers.sh`, assertions in owner and hook suites (some through helpers such as `assert_contains`, so a search for one assertion idiom misses them), and `file:<path>#<anchor>` firing paths of `source-register.md` rows, which `scripts/register-firing-path-resolution.rb` resolves against the tree. Because the classes grow, search by path rather than by class: you must search every non-Markdown file in the repository for the path you are about to trim or move text out of (`git grep -n -F '<path>' -- ':!*.md' ':!specs/'` lists the registries, gates and suites that name it; round evidence under `specs/` records history and pins nothing) and list the ledger anchors with `grep -o 'file:<path>#[^;|]*' skills/skill-extraction-workflow/references/source-register.md`; after the edit, run `bash skills/skill-extraction-workflow/scripts/check-ccl-skills.sh .`, which resolves the contract anchors, the sync registry and every ledger anchor, plus the suites the path search named. A funded trim that silently retired two pinned wait-contract obligations was reported by the slow lane only, long after the edit, and a relocation whose pin read covered only two of these sources moved a ledger-anchored rule out of its entrypoint. A pinned sentence may be reworded or moved only together with whatever pins it, in the same landing.
29
29
  - Authoring anti-patterns (verified against the official skill-authoring checklist, see verdicts below): time-sensitive facts outside an explicit old-patterns section; inconsistent terminology for one concept; abstract examples where a concrete input/output pair fits; Windows-style paths; unexplained constants; scripts that defer error handling to the model instead of solving it.
30
30
 
31
31
  ## Retirement and relocation signal (usage census)
@@ -33,7 +33,7 @@ The read side already defends against oversized files (chunked reads under ~200
33
33
  The ratchet only stops growth; it never says *what* to retire or relocate, and "not pulling its weight" is an author's opinion until something is measured. Three independent lines converge on the same instrument: context-evolution methods keep per-bullet usage counters (helpful/harmful marks) and prune or merge on them rather than on a single monolithic rewrite that collapses detail; trajectory-distillation work places broadly applicable procedure in the root document and *lower-frequency* detail in auxiliary files, and finds joint consolidation over many traces stronger than order-dependent one-lesson-at-a-time edits; and the official skill-authoring guidance tells authors to watch how the agent navigates a skill — a bundled file the agent never accesses is unnecessary or poorly signaled, one it reads on every run belongs in the entrypoint. The repo instantiation:
34
34
 
35
35
  - `scripts/reference-access-census.sh [--skill <name>] [--days <n>]` reads the host's own agent transcripts (Claude Code and Codex session logs; both consume this tree) and prints, per `SKILL.md`/reference file, how many sessions in the window mentioned it and when it was last touched (the last-touched column must be the newest touching transcript's date, never the oldest or an unparsed value). Counts only — no transcript text, prompts, absolute log paths, or session ids — so the output is safe for a private charter; it is still per-host data and never lands in the shared tree.
36
- - **Placement by firing frequency, not only by kind.** The entrypoint's content-placement rule sorts by kind (trigger / routing / core workflow / non-negotiables stay; detail moves). Add the frequency axis: a rule that fires on a narrow source class or correction type (one client surface, one correction shape, one artifact kind) is *low-frequency detail* even when it is non-negotiable, and must live verbatim in the reference the entrypoint already points at, with a one-bullet summary that keeps the load-bearing obligations inline (`rule-consolidation.md` condensing rule). Ledger `file:` anchors and script pins name a path, so a pinned phrase stays where it is — enumerate them before choosing what moves.
36
+ - **Placement by firing frequency and firing time, not only by kind.** The entrypoint's content-placement rule sorts by kind (trigger / routing / core workflow / non-negotiables stay; detail moves). Add the frequency axis: a rule that fires on a narrow source class or correction type (one client surface, one correction shape, one artifact kind) is *low-frequency detail* even when it is non-negotiable, and must live verbatim in the reference the entrypoint already points at, with a one-bullet summary that keeps the load-bearing obligations inline (`rule-consolidation.md` condensing rule). Ledger `file:` anchors and script pins name a path, so a pinned phrase stays where it is — enumerate them before choosing what moves. The time axis works the same way: a section that applies only at a later point of the same task (push and merge, post-merge teardown) is dead weight at activation and sits mid-context by the time it applies, so move it verbatim to a reference and name that reference where the later point is reached — a firing-point table in the entrypoint and the hook that fires there. Moving a section also breaks every pointer that names it by its old location, such as `SKILL.md「X」` or a link to the entrypoint followed by the section title; no gate resolves those, so before landing you must search the Markdown for the name of each moved section, outside `specs/`, and retarget every hit. Exact command conventions and destructive-operation guards that must be applied verbatim stay inline. A hook or always-on text injected at a firing point is what drives the action at that moment, so it must name the canonical section it summarizes, and what it restates must not read as complete when it is not — a digest that calls one exception the only one, or a command list that skips a mandatory pre-step, out-votes the canonical rule loaded earlier; pin the injected guards and the pointer target in the hook's own suite.
37
37
  - **Advisory, never a gate** (Goodhart, same as the health roll-up): a count that becomes a target gets gamed by mentioning files. A zero-session reference is a relocation/merge *candidate* that still owes the zero-loss obligation map; a high-share reference is a promotion candidate, not an automatic move. Pair the census with the closeout cost row in `extraction-quickstart.md` §4 so that a round records what it cost and what it retired, and the next round can tell whether the corpus is shrinking toward the cap or only holding level.
38
38
 
39
39
  - **A mention count is not an open count — classify HOW the file is reached before relocating on a census figure.** The census matches the path anywhere in a transcript line, so a file that a gate names in its own output, that an agent appends to, or that a bounded `sed`/`tail`/`grep` touches for one row scores the same as one an agent loads whole. Before a census figure justifies a split, a promotion, or a retirement, the read shape must be counted in the same window — whole-file reads versus bounded reads versus gate echo versus writes — and let the whole-read count, not the mention share, carry the read-side cost argument. Observed: the append-only ledger scored the package's highest mention share, and the read-shape count showed whole-file loads in a small minority of those mentions, with bounded reads and gate echo making up the rest — the split its share seemed to demand would have bought nothing on the read side.
@@ -113,8 +113,13 @@ Referenced from `SKILL.md`'s "The mechanism underneath" rule. This section holds
113
113
  | [OpenAI, *GPT-4.1 Prompting Guide*](https://developers.openai.com/cookbook/examples/gpt4-1_prompting_guide) | conflicting instructions tend to resolve to the one nearer the end; instructions at both ends of long context beat either alone; check-conflicts-first; a single clear sentence usually steers | same class; explicitly model-generation-bound ("GPT-4.1 tends to…") |
114
114
  | [OpenAI, *GPT-5.1 Prompting Guide*](https://cookbook.openai.com/examples/gpt-5/gpt-5-1_prompting_guide) | check-conflicts-first; a published metaprompt recipe for finding contradictions in your own system prompt | same class |
115
115
  | [RECAST](https://arxiv.org/html/2505.19030) | joint satisfaction degrades sharply as constraint count grows — and it is already low at small counts. The metric for "all of them at once" is the paper's **OSR** (§4.1: "the HSR of all constraints, both rule-based and model-based, that are successfully satisfied simultaneously"). Across all 29 model rows of Table 1 the **ceiling** on OSR is **25.0** at Level 1, falling to **19.0 / 13.0 / 13.5** at Levels 2–4, whose constraint counts are **5 / 10 / 15 / all** (§B.4). So at five constraints no model held the whole set more than about a quarter of the time, and by fifteen none exceeded ~13% | benchmark paper proposing its own dataset and method — a low baseline flatters the contribution; these are one hard benchmark's order of magnitude, not a usage failure rate, and its constraints are generation-task instruction constraints rather than preconditions of a procedure, so transfer is by analogy. **Citation corrected 2026-08 against Table 1:** the widely quoted 39.75% is the **Average column of the single best-by-average row** (Gemini-2.5-Pro) — the arithmetic mean of that row's twelve MSR/RSR/OSR cells (sum 477; 477/12 = 39.75, confirmed) — **not** an all-constraints-satisfied rate, and it must not be cited as one. The two orderings differ: Gemini leads on Average while Qwen3-235B-A22B holds the highest Level-1 OSR, so do not carry "best model" across from one column to the other. Cite the OSR ceilings above |
116
-
117
116
  | [IFScale](https://arxiv.org/abs/2507.11538) | instruction-following accuracy degrades as instruction density rises (500 keyword-inclusion instructions; best frontier model 68% at max density across 20 models / 7 providers); three degradation shapes correlated with model size and reasoning; a **bias toward earlier instructions** (primacy), and omission as a distinct error category | benchmark paper on a synthetic keyword-inclusion task — density and primacy transfer by analogy only; it measures a list of independent constraints, not a procedure's preconditions; note it reports primacy where the vendor guides report recency, so position is a bias with no single direction |
117
+ | [AgentIF](https://arxiv.org/abs/2505.16944) | real agent system prompts (707 instructions from 50 applications) average 1,723 words and 11.9 constraints; the best model followed fewer than 30% of them perfectly, adherence fell as instructions grew longer, and condition and tool constraints were the hardest | benchmark paper on production-shaped prompts — the closest public analog to a loaded skill body, but it scores one response per instruction, not a multi-step procedure |
118
+ | [Context Length Alone Hurts LLM Performance Despite Perfect Retrieval](https://arxiv.org/abs/2510.05381) | with every relevant token retrievable, accuracy still fell 13.9–85% as input grew within the advertised window, even when the added tokens were whitespace or masked; reciting the relevant evidence before answering recovered part of it | 5 models on math, QA and code — supports "length itself costs attention", not any particular budget figure |
119
+ | [SkillsBench](https://arxiv.org/abs/2602.12670) (§6, App. F, Tables 8–9) | curated Skills raised an 87-task pass rate from 33.9% to 50.5% across 18 model–harness configurations; by SKILL.md size bucket the lift was compact +19.0, standard +21.5, detailed +14.5 and comprehensive +0.7 points, and by Skills per task 1 / 2–3 / ≥4 gave +18.0 / +19.0 / +10.1; where Skills hurt, the paired-trajectory audit found a heavyweight pipeline crowding out a simpler path, a generic recipe displacing a stronger native strategy, or a brittle framework the agent could not debug; in the three configurations that tested self-generated Skills, they fell below the no-Skills baseline | observational on size and count (task and Skill are confounded, the comprehensive bucket has five tasks) on terminal-based tasks whose authors flag long-horizon workflows as possibly not transferring — a direction to test here, not a budget |
120
+ | [SkillJuror](https://arxiv.org/abs/2606.11543) | holding the knowledge fixed, a concise root that points to on-demand resources versus one flat file raised distinct resources touched per trajectory from 1.18 to 3.85 and added 17 verifier-passing trials of 410 (+4.1%); it helped where resources guide implementation, checking or repair and was weaker where success hinges on exact output conventions, numerical thresholds or long artifact pipelines | controlled variants on 82 tasks — organization changes how the agent searches before it changes outcomes, and the outcome gain is small and task-dependent |
121
+ | [Skill availability and presentation granularity](https://arxiv.org/abs/2605.31408) | having the Skill raised task-mean pass rates by 18–36 points, while moving the same guidance between abstraction levels or adding one worked example changed them by −6.7 to +1.3 points (both abstraction contrasts' bootstrap intervals cross zero) | controlled, 30 tasks × 2 models × 5 trials — do not expect a rewording of the same knowledge to move outcomes measurably |
122
+ | [Anthropic, *Prompting best practices*](https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) | newer Claude models follow the system prompt more closely, so prompts written to stop tools or skills under-triggering may now over-trigger; the fix is to dial back aggressive emphasis ("CRITICAL: You MUST use this tool when…" → "Use this tool when…") | vendor guidance bound to named model generations; not yet measured against this repository's process-heavy skills |
118
123
 
119
124
  **Why recency is a hazard, not a rule.** Vendor guidance reports that models *tend to follow* whichever instruction sits later — an observation about behaviour, not a licence to resolve conflicts by position. Two ways position becomes dangerous if read as a rule: a later permissive line beats an earlier stricter one (directly contradicting `Conflict Resolution`, which keeps the stricter data-loss/security/contract guard); and text embedded in **untrusted data** — a diff under review, a retrieved document, tool output — sits later within the same authority level and would win by placement alone, which is prompt injection with extra steps. Treat recency as a bias to design against: put the load-bearing rule where the decision happens, and never let placement confer authority.
120
125
 
@@ -59,7 +59,7 @@ When the workflow is consumed as a plugin (or any read-only install mechanism),
59
59
  - **Discover the source URL from the install.** The plugin/marketplace registration that shipped with the install records it: the marketplace clone's git remote (`git -C <marketplace-clone> remote get-url origin`), the marketplace `source.url` in the host's plugin config/registry, and the install URLs documented in the consuming project's README each help identify the canonical repo. Cross-check them rather than trusting one — a README can be stale or absent, and an install/marketplace URL can differ from the canonical source repo (e.g. http vs ssh, or a mirror).
60
60
  - **Author in one standing checkout — clone once, reuse.** Clone that URL a single time (or reuse an existing checkout) and reuse it for every later extraction. Do not clone per change.
61
61
  - **Never edit the consumption copies.** The plugin cache and the marketplace-managed clone are overwritten on update; edits there are lost and never reach the repo. Author only in your own checkout, then land through the normal review/MR gates.
62
- - **Isolate each change with a worktree, not a clone.** Make every extraction in a dedicated worktree off the standing checkout — worktrees share the object store and are not full repo copies — and remove it once the change lands (`git worktree remove`). This meets the concurrent-session isolation rule without proliferating repo copies.
62
+ - **Isolate each change with a worktree, not a clone.** Make every extraction in a dedicated worktree off the standing checkout — worktrees share the object store and are not full repo copies — and remove it once the change lands, following the closeout section (`## 收尾`) of `worktree-isolation/references/merge-and-teardown.md`: you must scan its gitignored outputs before removal (`git -C <worktree> status --ignored -s`, which must exit 0) and copy any output that is costly to recreate back to the standing checkout, because `git worktree remove` deletes gitignored files without asking. This meets the concurrent-session isolation rule without proliferating repo copies.
63
63
  - **Accumulation is consumption-side, not authoring.** Plugin caches on some hosts keep one directory per installed version, so old version dirs can linger after updates — unrelated to authoring. Prune them with the host's plugin prune command if disk matters.
64
64
 
65
65
  After landing, the change reaches every install through the normal update path (marketplace refresh + plugin update); the exact update command lives in the consuming project's README, not in the shared skill tree.
@@ -129,6 +129,8 @@ Result inflation 没有 MAST 对应——它是 context / 成本问题,不是
129
129
 
130
130
  **用**:本 skill `skill-extraction-workflow` 自身的 R0 / drafting 类大改动;产品 skill 的 routing 调整。
131
131
 
132
+ **落地形态**:`scripts/skill-paired-eval.py` + `eval/paired-tasks/`(`make eval-paired`)。同一合成 git 世界里交错跑 off / base / candidate 三臂,按世界状态判分,实现上面的冻结任务库、同轮起跑、成本列与逐断言读数;每个检查都带能把它判红的坏轨迹(`--check-oracles`)。它的数不是什么、隔离怎么核验,以脚本头为准。
133
+
132
134
  ### 3.2 Golden trace(中量)
133
135
 
134
136
  为每个 stable skill 沉淀 1-2 个 **golden agent trace**:
@@ -762,3 +762,18 @@ The pending classification above is superseded by the executed source comparison
762
762
  | A review does not report a requested fix of a pre-existing defect as droppable, and the requester's words in `--focus` reach every reviewer under the egress scan | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. An adversarial challenge read "each fix for a risk that predates the change" as asking reviewers to drop a requested fix of an older defect. In the build and release concern and in the manual prompt, the clause now covers only a pre-existing risk that the request does not cover and the change does not expose or worsen; the controller test fails on the previous text. A requested-fix replay (neutral domain, read by hand) found no run calling the fix droppable under either wording (Claude 0/6 each, Codex 0/3 each), so the reading did not reproduce and the narrowing aligns the text with its intent. New tests show `--focus` reaching the reviewer profile and a fallback reviewer, and a credential-shaped value blocking non-Claude egress; controller copies that drop the focus or skip the profile scan fail them. A reviewer-scope rerun read by hand: only the new text called an unrequested configuration switch unneeded (0/4 to 4/4). The matched-plan control behind the earlier row first carried an unrequested weekly step that reviewers flagged (Claude 5/6, Codex 3/3) and a regex had miscounted; with the step removed, no run raised a scope finding. |
763
763
  | A deterministic check that `make test` does not run surfaces only in CI after a push; the real-repository ledger audit runs in the fast lane and names its fix, and the delta-pass step names its entrypoint for a delta the extraction lane does not own | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. In one reviewed round CI went red on a stale line-cited obligation ledger after `make test` and the quick checker were green: the real-repository audit ran only in the heavy lane. With one line inserted above a cited carrier, main's fast lane passed and the candidate's fails on that audit, printing a `render` command that clears it when run as printed; a carrier whose text changed fails render and audit with another code, so the hint cannot hide a dropped obligation. `test_obligation_ledger.sh` runs the printed command and fails against the previous tool. Two rounds improvised chain ids after the extraction wrapper refused a delta it did not own; the delta-pass step in `dual-track-review-gate.md` now names both entrypoints, and the generic call it documents passes the controller's preconditions. |
764
764
  | The extraction lane's ownership refusal points at the delta-pass recipe for a delta it does not own | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. A delta pass over files the extraction lane does not own was refused with no route forward, and two rounds improvised a review-chain call to the generic controller. The refusal now names the delta-pass step that documents the call; the controller test asserts it and fails against the previous controller, where it is the only failure. The ownership precondition itself is unchanged. |
765
+ | The review controller counts conclusive runs per worktree and returns the continuation checkpoint from the sixth; the checkpoint names the same-class decision | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. In a month of sessions single changes ran 13 to 49 conclusive review runs, several after the five-run checkpoint existed; the controller kept only the last receipt, so the count lived in the agent's memory across hours and compactions. The receipt now carries `conclusive_runs` (review and challenge only) and the output carries `continuation_checkpoint` from the sixth run; the suite's three new checks fail against the previous controller, which after six runs records no count. Replay with only the latest finding visible: with the checkpoint object 6/6 answers stepped back before patching, without it 2/6. The added same-class sentence alone measured at ceiling (6/6 with the old or the new text when three findings are listed side by side) and is not claimed as a behavior change. |
766
+ | The review receipt counter keeps counting through an unparseable prior receipt and overlapping writers | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. The round's adversarial challenge found that a prior receipt nested deeply enough to make the parser raise RecursionError escaped the corrupt-receipt fallback, and that overlapping review and challenge completions could both read N and write N + 1. The parser failure now counts as an invalid receipt, and the read-increment-replace holds a lock on the receipt directory. In the suite's counting unit the previous controller fails on the recursion case; with the lock removed, 24 overlapping writers recorded 4 to 7 runs in three trials, and 24 with it. |
767
+ | The receipt lock check holds the lock itself instead of relying on writers overlapping | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. The round's delta review found that the overlapping-writer check proves the lock only when the writers happen to overlap; run one after another, they would pass with the lock removed. The check now takes the receipt directory lock itself, requires the writer to be still waiting a second after it started, rewrites the count while holding the lock, and requires the writer to increment that value. Against copies of the controller, removing the lock failed 8 of 8 runs on the waiting check, reading the prior receipt before taking the lock failed 8 of 8 on the count, and the unchanged controller passed 8 of 8. The previous check also failed both mutants in 5 of 5 runs on the same machine, so the gain is independence from scheduling, not a newly caught defect. |
768
+ | The receipt lock check signals from the writer's own lock call, so no wait decides its result | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. Supersedes the previous row's claim of scheduling independence: the next delta review found that a writer paused between signalling its start and reaching the lock lets both mutants pass once the one-second wait expires. The writer now signals from inside its exclusive lock call on the receipt directory, and the holder rewrites the count only after that signal; unlock and cleanup run in `finally`, the writer is a daemon and every wait is bounded. Against copies of the controller, each run with and without a forced two-second pause in the writer: the unchanged controller passed 8 of 8; removing the lock and taking a shared lock each failed 8 of 8 on the missing signal; reading the prior receipt before the lock failed 8 of 8 on the count. |
769
+ | The receipt lock check asserts the exclusive lock at the receipt read and replace instead of racing writers | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. Supersedes the two rows above for the check's design. Three successive delta reviews each found a controller variant that a racing-writer check let through: writers that did not overlap, a writer paused past a wait, a lock released before the read. The class was closed by changing the method instead of patching a fourth time. When the controller opens the prior receipt and when it replaces it, the check requires that a second open of the receipt directory cannot take even a shared lock, and then that the count increments the value read. Against copies of the controller, three runs each, the unchanged controller passed; removing the lock, a shared lock, a lock on another descriptor, reading before the lock, unlocking before the read and unlocking before the replace each failed on the recorded lock states. |
770
+ | The receipt lock check also requires the lock to be held without interruption between the read and the replace, and reads the stored receipt back | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. The delta review of the previous row's check, the sixth conclusive run in the worktree and the first to return the continuation checkpoint, found that probes at the read and the replace pass a controller that unlocks and relocks in between, and that the check trusted the returned receipt over the stored one. The check now records every lock call during the controller's call and requires none between the read and the replace, and it compares the stored receipt with the returned one. Against copies of the controller, three runs each: the unchanged controller passed; unlocking and relocking between the read and the replace failed on the recorded lock calls, and a stored count or start time that differs from the returned one, applied to the last call only, failed on the stored receipt; the six variants from the previous row still failed. |
771
+ | The always-on pointers to the worktree merge protocol and teardown section resolve to the package reference that now carries that text, the entrypoint sentence that forwards to it is pinned, and the gate still blocks when either side loses its anchor | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_sync_pointers.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. Constructed scenarios, runs recorded in `specs/161-firing-point-loading/evidence/validation.md`: with the merge and teardown sections relocated and the previous registry, `check-sync-pointers.sh` exits 1 on exactly the three worktree pins; with the registry pointed at `worktree-isolation/references/merge-and-teardown.md` it passes; removing the protocol anchor or renaming the teardown heading in that reference blocks the matching pins only. Removing the forwarding sentence from `worktree-isolation/SKILL.md` left the gate green until the new forwarding pair, which then blocks it alone. The suite's fixtures mutate the reference and the forwarding sentence. |
772
+ | Trimming or moving text out of a file first finds its pins by path, every non-Markdown file outside round evidence plus the ledger's file anchors, and runs the fast validator after the edit | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#must search every non-Markdown file in the repository for the path | updated | Owner key `skill-extraction-workflow/SKILL.md`. Recorded incident in `specs/161-firing-point-loading/evidence/validation.md`: a relocation that read only contract anchors and suite assertions moved the shared-branch update bullet out of `worktree-isolation/SKILL.md`, where six register rows anchor their firing paths; the full validator stopped on `register_firing_path_unresolved` and the base tree did not. On the base tree the path search lists the sync registry, the `assert_contains` pin and the sync suite fixtures, and the ledger search lists the six anchors, so the move would have been seen before it was made; the bullet stays in the entrypoint. |
773
+ | A section used only at a later point of the task loads at that point, and text injected there names its canonical section without presenting a partial list as complete | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#must name the canonical section it summarizes | updated | Owner key `skill-extraction-workflow/SKILL.md`. Applied to `worktree-isolation`: entrypoint 38,112 to 13,964 bytes with 81 moved lines verbatim in two references. The post-merge reminder called the integration-branch case the only exception and skipped the ignored-artifact scan; against the previous hook the nine new text assertions in `hooks/test_remind_post_merge_cleanup.sh` fail and the 57 existing probes pass, against the new hook all 67 pass, and ten single mutations each fail the assertions they target. In paired runs of a cleanup task on a synthetic repository, neither the previous nor the relocated skill lost the expensive ignored file or the release branch in five runs each, while runs without the plugin lost them in four and three of five; both recorded in `specs/161-firing-point-loading/evidence/validation.md`. |
774
+ | A worktree closeout that removes the worktree first scans its gitignored outputs from the primary checkout, treats a failed scan as no scan, copies costly outputs out, keeps forced removal and protected branches off the path, and names the canonical teardown section | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/product-rd-workflow/references/worktree-mechanics.md#Before removing the worktree directory, you must run | updated | Owner key `product-rd-workflow/SKILL.md`. The Closeout Cleanup listed only `git worktree remove`, `git worktree prune` and `git branch -d`, and `git worktree remove` without `--force` deletes gitignored files and exits 0. Each of the nine product delivery rows of the teardown pin family in the shared implementation-gates fixture fails alone against the text before this round and passes on the new one, and the teardown pin walk reds each under an applied mutation. In paired cleanup runs without the plugin, with the recipe committed into a synthetic repository, agents following the previous text lost the expensive ignored file in 2 of 5 runs and stopped to ask in the other 3, while agents following the new text scanned first, kept the file and finished in 5 of 5; with the plugin loaded, both texts kept the file in 5 of 5. Recorded in `specs/162-teardown-recipe-guard/evidence/validation.md`. |
775
+ | Every worker worktree removal in a delegated handoff runs the ignored-output scan first, requires it to succeed, copies costly outputs out, and records the scan in the cleanup proof | `multi-agent-delegation` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/multi-agent-delegation/references/multi-agent-delegation-playbook.md#Every worktree removal below must run the pre-removal scan | updated | Owner key `multi-agent-delegation/SKILL.md`. Handoff step 5 removed the worker worktree on three paths, one of them for a clean worktree, with no scan, so gitignored results a worker produced were deleted with it. Each of the five delegation rows of the teardown pin family fails alone against the text before this round and passes on the new one, and the teardown pin walk reds each under an applied mutation; recorded in `specs/162-teardown-recipe-guard/evidence/validation.md`. |
776
+ | A Markdown surface that restates worktree removal carries the ignored-output scan with its exit-0 requirement on the same line and names the canonical teardown reference, and one added later without them fails the shared fixture | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. Against the tree before this round the sweep names exactly the extraction lifecycle note and the product delivery closeout (no scan) and the worktree handbook (no pointer); on the new tree it passes. `test_teardown_guard_pins.sh` plants decoy surfaces and failure probes and each reds the sweep for the stated reason (no scan, no exit-0 requirement on the scan line, no pointer or a file-name-only pointer, a new top-level directory, a file name starting with a newline, an unreadable file, directory or index, a canonical file hidden from git, a missing repository, a failed classification), while a compliant surface, a package-relative pointer inside the canonical package, a prune-only mention, ignored paths, round records, evaluation inputs and a register row stay green; nineteen sabotaged copies of the fixture each make the walk fail. The sweep keys on the command, so a removal described only in prose is pinned per surface; recorded in `specs/162-teardown-recipe-guard/evidence/validation.md`. |
777
+ | Moving a section retargets every Markdown pointer that names it by its old location, found by searching Markdown for the moved section's name, because no gate resolves those pointers | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#you must search the Markdown for the name of each moved section | updated | Owner key `skill-extraction-workflow/SKILL.md`. Recorded miss in `specs/162-teardown-recipe-guard/evidence/validation.md`: after round 161 (`specs/161-firing-point-loading/`) moved the worktree merge and teardown sections into references, the worktree handbook still named the entrypoint for three of them. On that tree the path search the recipe asked for, over non-Markdown files, returns registries and suites only, while searching Markdown outside `specs/` for each moved section's name returns the three handbook lines. |
778
+ | A behavior-shaping skill change is measured before it lands with paired runs on synthetic tasks (no plugin, a base export, a candidate export) graded by the world state the agent leaves; each run's isolation is read from its own events and an instruction-file canary, and every task check is proven able to fail by a bad trajectory | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/skill-paired-eval.py | updated | Owner key `skill-extraction-workflow/SKILL.md`. §3.1 of `harness-patterns-and-eval.md` had no tool; rounds 161 and 162 measured with private scripts that passed their whole environment to the tested agent, and in round 162 the no-plugin arm read the installed plugin cache and followed it (`specs/162-teardown-recipe-guard/evidence/validation.md`). `skill-paired-eval.py` strips inherited `CLAUDE*` and `GIT_*` variables, gives each run a private copy of the frozen export and checks its init event for that path, and calibrates a per-sample instruction-file canary that appears with project instructions allowed and stayed absent under the run flags. Each of 108 applied mutations to its graders, isolation checks, guards and batch lifecycle reds the named test in `test_skill_paired_eval.py` on a disposable copy while an unrelated test stays green, and every check of the four tasks in `eval/paired-tasks/` fails on a bad trajectory. First batch, 54 samples on Opus 5.5 comparing the published 0.18.11, main and no plugin: with either plugin the release branch was kept in 5 of 5 cleanup runs against 0 of 5 without it, and the edit on the default branch went into a worktree in 5 of 5 against 0 of 5 (p = 0.008 each); no base-candidate comparison separated, and every sample passed the structural isolation checks. That batch disabled session persistence, so the plugin's hooks that read the session transcript did not run in it; the tool now keeps persistence on and invalidates a run without a transcript. Recorded in `specs/163-paired-behavior-eval/evidence/validation.md`. |
779
+ | A session nothing re-invokes ends at the stop and the host stops its background tasks seconds later, so the rule that awaiting your own work is not a stop condition needs a firing point that reads the running tasks from the Stop hook's input instead of the transcript: one block per still-running task, for SDK entrypoints only | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/pre-final-continuation-gate.md#blocks such a stop once per background task still running | updated | Owner key `product-rd-workflow/SKILL.md`; changed reference `product-rd-workflow/references/pre-final-continuation-gate.md`. In a paired batch of an earlier round, six of seven background reviews started by plugin-arm headless runs were killed after the turn ended, leaving three changes uncommitted and three reviews unread, while the plugin's transcript-reading reminder could not run. In a pre-registered headless probe with session persistence on, the plugin from `main` lost the work in 3 of 3 runs, its background task killed, and the plugin with the guard in none, waiting and then finishing. The guard's suite has 30 cases, and each of 22 applied mutations turns its named case red. Recorded in `specs/164-headless-background-stop/evidence/validation.md`. |
@@ -56,7 +56,12 @@ case " ${RUBYOPT:-} " in *" -Ku "*) ;; *) export RUBYOPT="-Ku${RUBYOPT:+ $RUBYOP
56
56
  root="${1:-.}"
57
57
 
58
58
  bootstrap="$root/agent-context/session-start.md"
59
- wt="$root/skills/worktree-isolation/SKILL.md"
59
+ # The merge protocol and the teardown section load at their firing point, so
60
+ # they live in the package reference the entrypoint points to, not in SKILL.md.
61
+ # The entrypoint's forwarding sentence is the hop an agent takes from the
62
+ # always-on pointer to that reference, so it is pinned too.
63
+ wt_entry="$root/skills/worktree-isolation/SKILL.md"
64
+ wt_teardown="$root/skills/worktree-isolation/references/merge-and-teardown.md"
60
65
  prd="$root/skills/product-rd-workflow/SKILL.md"
61
66
  se="$root/skills/skill-extraction-workflow/SKILL.md"
62
67
  sec4="$root/skills/requirement-doc-writer/references/security-four-questions.md"
@@ -169,11 +174,13 @@ run_pair() { # <name> <bootstrap literal> <target> <target literal> <package-dir
169
174
 
170
175
  if [ "$bootstrap_present" -eq 1 ]; then
171
176
  run_pair "merge-exec-protocol-section" \
172
- "「合并执行协议」(canonical" "$wt" "**合并执行协议(canonical——" "$root/skills/worktree-isolation"
177
+ "「合并执行协议」(canonical" "$wt_teardown" "**合并执行协议(canonical——" "$root/skills/worktree-isolation"
173
178
  run_pair "merge-exec-citation-token" \
174
- "「依据: worktree-isolation 合并执行协议」" "$wt" "**合并执行协议(canonical——" "$root/skills/worktree-isolation"
179
+ "「依据: worktree-isolation 合并执行协议」" "$wt_teardown" "**合并执行协议(canonical——" "$root/skills/worktree-isolation"
175
180
  run_pair "worktree-teardown-section" \
176
- "在 \`worktree-isolation\` 收尾节" "$wt" "## 收尾:" "$root/skills/worktree-isolation"
181
+ "在 \`worktree-isolation\` 收尾节" "$wt_teardown" "## 收尾:" "$root/skills/worktree-isolation"
182
+ run_pair "worktree-reference-forwarding" \
183
+ "在 \`worktree-isolation\` 收尾节" "$wt_entry" "都在 \`references/merge-and-teardown.md\`" "$root/skills/worktree-isolation"
177
184
  run_pair "owner-dispatch-firing-gate" \
178
185
  "product-rd \`Implementation entry / re-entry gate\` + \`Owner-dispatch firing gate\`" "$prd" "- **Owner-dispatch firing gate (" "$root/skills/product-rd-workflow"
179
186
  run_pair "implementation-entry-reentry-gate" \