@ccoalm/ccl-skills 0.18.10 → 0.18.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/host-input.py +84 -8
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/task-entry.sh +24 -3
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_proposed_next.py +71 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_task_entry.py +63 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +77 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +112 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +6 -0
- package/dist/assets/release.json +11 -11
- package/package.json +1 -1
|
@@ -579,8 +579,8 @@ def machine_artifact(text):
|
|
|
579
579
|
return False
|
|
580
580
|
|
|
581
581
|
|
|
582
|
-
def
|
|
583
|
-
"""
|
|
582
|
+
def claim_notice(payload, lane):
|
|
583
|
+
"""True on this lane's first attempt, False on a repeat, None when state is unavailable."""
|
|
584
584
|
try:
|
|
585
585
|
# Reuse the installed runtime's owned-directory/no-follow/atomic claim
|
|
586
586
|
# protections. Minimal vendored runtimes may omit this optional helper.
|
|
@@ -596,14 +596,19 @@ def stop_notice(payload, lane, message):
|
|
|
596
596
|
info = module.regular_info(path)
|
|
597
597
|
state = module.State(key)
|
|
598
598
|
try:
|
|
599
|
-
|
|
600
|
-
return None
|
|
599
|
+
return bool(state.claim_attempt('stop-notice-' + lane, [info.st_dev, info.st_ino]))
|
|
601
600
|
finally:
|
|
602
601
|
state.close()
|
|
603
602
|
except Exception:
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
603
|
+
return None
|
|
604
|
+
|
|
605
|
+
|
|
606
|
+
def stop_notice(payload, lane, message):
|
|
607
|
+
"""Cap notice attempts only; these markers never establish verification."""
|
|
608
|
+
# Missing identity, a broken optional helper or unsafe/unavailable state must
|
|
609
|
+
# not invent success: the notice still shows. This only controls advisory output.
|
|
610
|
+
if claim_notice(payload, lane) is False:
|
|
611
|
+
return None
|
|
607
612
|
return {'systemMessage': message}
|
|
608
613
|
|
|
609
614
|
|
|
@@ -721,6 +726,75 @@ def doc_closeout_note(payload):
|
|
|
721
726
|
'the substance stays as the owning skill decided.'.format(names))
|
|
722
727
|
|
|
723
728
|
|
|
729
|
+
# Every request re-reads the whole context, so a long session's token cost grows
|
|
730
|
+
# with its size, and a 1M-token window compacts only near its limit by default.
|
|
731
|
+
# The notice goes to the user only (systemMessage, never the model's context),
|
|
732
|
+
# once per band per transcript, and reads just the transcript tail.
|
|
733
|
+
CONTEXT_BANDS = (300000, 600000)
|
|
734
|
+
CONTEXT_TAIL_BYTES = 1024 * 1024
|
|
735
|
+
|
|
736
|
+
|
|
737
|
+
def last_context_tokens(path):
|
|
738
|
+
"""Context size of the latest real Claude request, or None when unknown."""
|
|
739
|
+
descriptor = os.open(path, os.O_RDONLY | os.O_NONBLOCK)
|
|
740
|
+
with os.fdopen(descriptor, 'rb') as stream:
|
|
741
|
+
metadata = os.fstat(stream.fileno())
|
|
742
|
+
if not stat.S_ISREG(metadata.st_mode):
|
|
743
|
+
return None
|
|
744
|
+
stream.seek(max(0, metadata.st_size - CONTEXT_TAIL_BYTES))
|
|
745
|
+
tail = stream.read(CONTEXT_TAIL_BYTES)
|
|
746
|
+
for line in reversed(tail.split(b'\n')):
|
|
747
|
+
if b'"usage"' not in line:
|
|
748
|
+
continue
|
|
749
|
+
try:
|
|
750
|
+
event = json.loads(line)
|
|
751
|
+
except ValueError:
|
|
752
|
+
continue
|
|
753
|
+
if not isinstance(event, dict) or event.get('type') != 'assistant' or event.get('isSidechain'):
|
|
754
|
+
continue
|
|
755
|
+
message = event.get('message')
|
|
756
|
+
usage = message.get('usage') if isinstance(message, dict) else None
|
|
757
|
+
if not isinstance(usage, dict):
|
|
758
|
+
continue
|
|
759
|
+
total = 0
|
|
760
|
+
for key in ('input_tokens', 'cache_read_input_tokens', 'cache_creation_input_tokens'):
|
|
761
|
+
value = usage.get(key)
|
|
762
|
+
if isinstance(value, int) and not isinstance(value, bool) and value > 0:
|
|
763
|
+
total += value
|
|
764
|
+
if total:
|
|
765
|
+
return total
|
|
766
|
+
return None
|
|
767
|
+
|
|
768
|
+
|
|
769
|
+
def context_notice(payload):
|
|
770
|
+
path = payload.get('transcript_path')
|
|
771
|
+
if not isinstance(path, str) or not path:
|
|
772
|
+
return None
|
|
773
|
+
tokens = last_context_tokens(path)
|
|
774
|
+
reached = [band for band in CONTEXT_BANDS if tokens is not None and tokens >= band]
|
|
775
|
+
if not reached:
|
|
776
|
+
return None
|
|
777
|
+
message = ('本会话上下文约 {} 万 token:之后每次请求都会重读这些内容,长会话的 token 主要花在这里。'
|
|
778
|
+
'当前交付收口后可先写好交接再 /clear 开新会话;或用 /autocompact 把自动压缩提前'
|
|
779
|
+
'(如 /autocompact 400k)。此提示只显示给你,不影响当前任务。').format(tokens // 10000)
|
|
780
|
+
# Optional information: without the state helper it stays quiet rather than
|
|
781
|
+
# repeating on every stop.
|
|
782
|
+
return message if claim_notice(payload, 'context-{}k'.format(reached[-1] // 1000)) is True else None
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
def with_context_notice(payload, result):
|
|
786
|
+
try:
|
|
787
|
+
notice = context_notice(payload) if isinstance(payload, dict) else None
|
|
788
|
+
except Exception: # advisory: a failed size check never costs another notice
|
|
789
|
+
notice = None
|
|
790
|
+
if not notice:
|
|
791
|
+
return result
|
|
792
|
+
result = dict(result or {})
|
|
793
|
+
result['systemMessage'] = (result['systemMessage'] + '\n' + notice
|
|
794
|
+
if result.get('systemMessage') else notice)
|
|
795
|
+
return result
|
|
796
|
+
|
|
797
|
+
|
|
724
798
|
def proposed_next(payload):
|
|
725
799
|
if (not isinstance(payload, dict) or payload.get('hook_event_name') != 'Stop'
|
|
726
800
|
or payload.get('stop_hook_active') is not False):
|
|
@@ -818,12 +892,14 @@ def main():
|
|
|
818
892
|
raise ValueError('oversized input')
|
|
819
893
|
payload = json.loads(raw)
|
|
820
894
|
result = (extraction_overflow(payload) if sys.argv[1] == 'extraction-overflow'
|
|
821
|
-
else proposed_next(payload))
|
|
895
|
+
else with_context_notice(payload, proposed_next(payload)))
|
|
822
896
|
if result:
|
|
823
897
|
print(json.dumps(result))
|
|
824
898
|
except TranscriptTruncated:
|
|
825
899
|
result = stop_notice(payload, 'handoff-overflow',
|
|
826
900
|
'Delivery handoff reminder unverified: transcript scan exceeded its bounded limit.')
|
|
901
|
+
if sys.argv[1] == 'proposed-next':
|
|
902
|
+
result = with_context_notice(payload, result)
|
|
827
903
|
if result:
|
|
828
904
|
print(json.dumps(result))
|
|
829
905
|
except (OSError, ValueError, TypeError, IndexError, AttributeError):
|
|
@@ -1,15 +1,36 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
|
-
# Deliver the canonical task entry before sampling
|
|
2
|
+
# Deliver the canonical task entry before sampling. The prompt is never classified
|
|
3
|
+
# or echoed. One structural check spots a prompt that is exactly one
|
|
4
|
+
# <task-notification> envelope, the turn the host starts itself to deliver a
|
|
5
|
+
# background completion: that turn keeps the skill-loading and unfinished-work
|
|
6
|
+
# boundary but not the routing list, which SessionStart already keeps in context.
|
|
3
7
|
SCRIPT_DIR="$(cd "$(dirname "$0")" 2>/dev/null && pwd)"
|
|
4
8
|
if ! command -v python3 >/dev/null 2>&1; then
|
|
5
9
|
printf 'ccl-skills task-entry: python3 unavailable; task entry omitted\n' >&2
|
|
6
10
|
printf '{}\n'
|
|
7
11
|
exit 0
|
|
8
12
|
fi
|
|
9
|
-
|
|
13
|
+
# fd 3 carries the hook input; stdin is the heredoc program. A closed stdin reads
|
|
14
|
+
# as empty input, which keeps the entry.
|
|
15
|
+
if ! { : 3<&0; } 2>/dev/null; then exec 0</dev/null; fi
|
|
16
|
+
python3 - "$SCRIPT_DIR/../agent-context/session-start.md" 3<&0 <<'PY'
|
|
10
17
|
import json
|
|
18
|
+
import os
|
|
11
19
|
import sys
|
|
12
20
|
|
|
21
|
+
host_turn = False
|
|
22
|
+
try:
|
|
23
|
+
with os.fdopen(3, 'rb') as hook_input:
|
|
24
|
+
raw = hook_input.read(1048577)
|
|
25
|
+
payload = json.loads(raw) if len(raw) <= 1048576 else None
|
|
26
|
+
prompt = payload.get('prompt') if isinstance(payload, dict) else None
|
|
27
|
+
envelope = prompt.strip() if isinstance(prompt, str) else ''
|
|
28
|
+
host_turn = (envelope.startswith('<task-notification>') and envelope.endswith('</task-notification>')
|
|
29
|
+
and envelope.count('<task-notification>') == 1 and envelope.count('</task-notification>') == 1)
|
|
30
|
+
except (OSError, ValueError, RecursionError):
|
|
31
|
+
# Unreadable input keeps the full entry: trimming is only for a proven host turn.
|
|
32
|
+
host_turn = False
|
|
33
|
+
|
|
13
34
|
try:
|
|
14
35
|
with open(sys.argv[1], 'rb') as stream:
|
|
15
36
|
raw = stream.read(32769)
|
|
@@ -35,7 +56,7 @@ try:
|
|
|
35
56
|
'Inspect failure evidence, research or change the approach, repair safely and rerun the relevant checks. '
|
|
36
57
|
'A report alone does not complete it. Continue available authorized work; hand back only for a '
|
|
37
58
|
'required user decision or unavailable authority/resource, stating the concrete blocker.\n\n')
|
|
38
|
-
context = '<ccl-task-entry>\n' + boundary + entry + '\n</ccl-task-entry>'
|
|
59
|
+
context = '<ccl-task-entry>\n' + (boundary.rstrip('\n') if host_turn else boundary + entry) + '\n</ccl-task-entry>'
|
|
39
60
|
if len(context.encode('utf-8')) > 4096:
|
|
40
61
|
raise ValueError('oversized entry')
|
|
41
62
|
print(json.dumps({'hookSpecificOutput': {
|
|
@@ -541,6 +541,77 @@ class ProposedNextTests(unittest.TestCase):
|
|
|
541
541
|
(self.hooks / 'host-input.py').unlink()
|
|
542
542
|
self.assertIn('unavailable', self.run_hook().get('systemMessage', ''))
|
|
543
543
|
|
|
544
|
+
def usage_event(self, context, cache_read=None, model='claude-opus-5-5'):
|
|
545
|
+
read = context - 2000 if cache_read is None else cache_read
|
|
546
|
+
return {'type': 'assistant', 'message': {'model': model, 'content': [{'type': 'text', 'text': 'ok'}],
|
|
547
|
+
'usage': {'input_tokens': 2, 'cache_read_input_tokens': read,
|
|
548
|
+
'cache_creation_input_tokens': context - 2 - read, 'output_tokens': 50}}}
|
|
549
|
+
|
|
550
|
+
def run_with_state(self, payload=None):
|
|
551
|
+
# The once-per-band cap uses the optional state helper; give it a private TMPDIR.
|
|
552
|
+
shutil.copyfile(ROOT / 'hooks/skill-loading.py', self.hooks / 'skill-loading.py')
|
|
553
|
+
state = self.root.parent / (self.root.name + '-state')
|
|
554
|
+
state.mkdir(exist_ok=True)
|
|
555
|
+
value = self.payload if payload is None else payload
|
|
556
|
+
result = subprocess.run(['bash', str(self.hooks / 'proposed-next-stop.sh')],
|
|
557
|
+
input=json.dumps(value), text=True, capture_output=True, cwd=self.root,
|
|
558
|
+
env=dict(os.environ, TMPDIR=str(state)))
|
|
559
|
+
self.assertEqual(result.returncode, 0, result.stderr)
|
|
560
|
+
self.assertEqual(result.stderr, '')
|
|
561
|
+
return json.loads(result.stdout) if result.stdout else {}
|
|
562
|
+
|
|
563
|
+
def test_large_context_notice_is_user_only_and_once_per_band(self):
|
|
564
|
+
self.events([self.usage_event(120000), self.usage_event(299999)])
|
|
565
|
+
self.assertEqual(self.run_with_state(), {})
|
|
566
|
+
self.events([self.usage_event(120000), self.usage_event(321000)])
|
|
567
|
+
first = self.run_with_state()
|
|
568
|
+
self.assertEqual(set(first), {'systemMessage'}) # shown to the user, never a block or model context
|
|
569
|
+
self.assertIn('32 万', first['systemMessage'])
|
|
570
|
+
self.assertIn('/autocompact', first['systemMessage'])
|
|
571
|
+
self.assertIn('/clear', first['systemMessage'])
|
|
572
|
+
self.assertEqual(self.run_with_state(), {}) # same band: no repeat
|
|
573
|
+
self.events([self.usage_event(321000), self.usage_event(612000)])
|
|
574
|
+
second = self.run_with_state()
|
|
575
|
+
self.assertIn('61 万', second.get('systemMessage', ''))
|
|
576
|
+
self.assertEqual(self.run_with_state(), {})
|
|
577
|
+
|
|
578
|
+
def test_large_context_notice_stays_quiet_without_its_state_helper(self):
|
|
579
|
+
# Without the once-per-band record the notice would repeat on every stop, so
|
|
580
|
+
# it is withheld; notices that must show (handoff overflow) still show.
|
|
581
|
+
self.events([self.usage_event(450000)])
|
|
582
|
+
self.assertFalse((self.hooks / 'skill-loading.py').exists())
|
|
583
|
+
self.assertEqual(self.run_hook(), {})
|
|
584
|
+
self.assertEqual(self.run_hook(), {})
|
|
585
|
+
|
|
586
|
+
def test_large_context_notice_rides_with_a_continuation_reminder(self):
|
|
587
|
+
self.events([self.usage_event(450000)])
|
|
588
|
+
payload = dict(self.payload, last_assistant_message='Fixed.\n\nproposed-next: run the integration suite')
|
|
589
|
+
value = self.run_with_state(payload)
|
|
590
|
+
self.assertEqual(value.get('decision'), 'block')
|
|
591
|
+
self.assertIn('proposed-next:', value['reason'])
|
|
592
|
+
self.assertIn('45 万', value.get('systemMessage', ''))
|
|
593
|
+
self.assertNotIn('万 token', value['reason']) # the model-facing reason stays unchanged
|
|
594
|
+
|
|
595
|
+
def test_large_context_notice_reads_only_the_latest_real_claude_usage(self):
|
|
596
|
+
# A trailing zero-usage (synthetic) entry is skipped; Codex shapes and huge
|
|
597
|
+
# trailing tool output without usage stay quiet; the notice never needs the
|
|
598
|
+
# bounded full-transcript scan.
|
|
599
|
+
synthetic = self.usage_event(0, cache_read=0, model='<synthetic>')
|
|
600
|
+
synthetic['message']['usage'] = {'input_tokens': 0, 'cache_read_input_tokens': 0,
|
|
601
|
+
'cache_creation_input_tokens': 0, 'output_tokens': 0}
|
|
602
|
+
self.events([self.usage_event(330000), synthetic])
|
|
603
|
+
self.assertIn('33 万', self.run_with_state().get('systemMessage', ''))
|
|
604
|
+
self.setUp()
|
|
605
|
+
self.events([{'type': 'event_msg', 'payload': {'type': 'token_count', 'info': {
|
|
606
|
+
'last_token_usage': {'input_tokens': 900000}}}}])
|
|
607
|
+
self.assertEqual(self.run_with_state(), {})
|
|
608
|
+
self.setUp()
|
|
609
|
+
self.events([self.usage_event(500000)] + [{'type': 'user', 'text': 'x' * 400000}] * 3)
|
|
610
|
+
self.assertEqual(self.run_with_state(), {})
|
|
611
|
+
self.setUp()
|
|
612
|
+
self.events([{'type': 'ignored'}] * 20000 + [self.usage_event(700000)])
|
|
613
|
+
self.assertIn('70 万', self.run_with_state().get('systemMessage', ''))
|
|
614
|
+
|
|
544
615
|
def test_scan_is_bounded_and_no_filesystem_markers_are_written(self):
|
|
545
616
|
self.events([{'type': 'ignored'}] * 20000 + self.claude_load())
|
|
546
617
|
before = set(self.root.rglob('*'))
|
|
@@ -80,6 +80,69 @@ class TaskEntryTests(unittest.TestCase):
|
|
|
80
80
|
self.assertIn('A report alone does not complete it', context)
|
|
81
81
|
self.assertIn('required user decision or unavailable authority/resource', context)
|
|
82
82
|
|
|
83
|
+
def test_host_task_notification_turn_keeps_only_the_boundary(self):
|
|
84
|
+
# The host also runs UserPromptSubmit on turns it starts itself; a background
|
|
85
|
+
# completion arrives as one <task-notification> envelope. SessionStart keeps the
|
|
86
|
+
# routing list in context (also after compaction), so that turn drops only the
|
|
87
|
+
# list; the skill-loading and unfinished-work boundary appears nowhere else and
|
|
88
|
+
# matters most when a background check has just failed.
|
|
89
|
+
expected, _ = self.run_hook(prompt='Add a feature')
|
|
90
|
+
full = expected['hookSpecificOutput']['additionalContext']
|
|
91
|
+
for prompt in ['<task-notification>\n<task-id>b1</task-id>\n<status>failed</status>\n</task-notification>',
|
|
92
|
+
'\n <task-notification><task-id>b2</task-id></task-notification>']:
|
|
93
|
+
with self.subTest(prompt=prompt[:30]):
|
|
94
|
+
output, error = self.run_hook(prompt=prompt)
|
|
95
|
+
self.assertEqual(error, '')
|
|
96
|
+
context = output['hookSpecificOutput']['additionalContext']
|
|
97
|
+
self.assertIn('unrun, failed or inconclusive verification is unfinished work', context)
|
|
98
|
+
self.assertIn('Before task-specific investigation or substantive analysis', context)
|
|
99
|
+
self.assertTrue(context.startswith('<ccl-task-entry>') and context.endswith('</ccl-task-entry>'))
|
|
100
|
+
self.assertNotIn('<!-- ccl:entry-routing:start -->', context)
|
|
101
|
+
self.assertNotIn('**product-rd-workflow**', context)
|
|
102
|
+
self.assertLess(len(context), len(full) // 2)
|
|
103
|
+
# Controls: a human prompt that merely mentions the envelope still gets the
|
|
104
|
+
# entry, as does input the hook cannot parse.
|
|
105
|
+
for prompt in ['Why did <task-notification> fire twice?', 'task-notification arrived',
|
|
106
|
+
'<task-notification><task-id>b3</task-id></task-notification>\nWhat does this mean?',
|
|
107
|
+
'<task-notification> pasted without its closing tag',
|
|
108
|
+
'<task-notification>first</task-notification>\nCompare these.\n'
|
|
109
|
+
'<task-notification>second</task-notification>']:
|
|
110
|
+
with self.subTest(prompt=prompt):
|
|
111
|
+
output, _ = self.run_hook(prompt=prompt)
|
|
112
|
+
self.assertEqual(output, expected)
|
|
113
|
+
|
|
114
|
+
def test_unreadable_hook_input_keeps_the_entry(self):
|
|
115
|
+
expected, _ = self.run_hook(prompt='Add a feature')
|
|
116
|
+
for raw in ['', 'not json', json.dumps(['<task-notification>']),
|
|
117
|
+
json.dumps({'prompt': 42}), json.dumps({'prompt': '<task-notification>' + 'x' * 2_000_000}),
|
|
118
|
+
'[' * 500000 + ']' * 500000]:
|
|
119
|
+
with self.subTest(raw=raw[:30]):
|
|
120
|
+
with tempfile.TemporaryDirectory(prefix='ccl-task-entry-') as directory:
|
|
121
|
+
root = Path(directory)
|
|
122
|
+
(root / 'hooks').mkdir()
|
|
123
|
+
(root / 'agent-context').mkdir()
|
|
124
|
+
shutil.copyfile(ROOT / 'hooks/task-entry.sh', root / 'hooks/task-entry.sh')
|
|
125
|
+
(root / 'agent-context/session-start.md').write_text(
|
|
126
|
+
(ROOT / 'agent-context/session-start.md').read_text())
|
|
127
|
+
result = subprocess.run(['bash', str(root / 'hooks/task-entry.sh')], input=raw,
|
|
128
|
+
text=True, capture_output=True, timeout=5,
|
|
129
|
+
cwd=root, env={'PATH': os.environ['PATH']})
|
|
130
|
+
self.assertEqual(result.returncode, 0, result.stderr)
|
|
131
|
+
self.assertEqual(json.loads(result.stdout), expected)
|
|
132
|
+
# A closed stdin is read as empty input, never a silent hook.
|
|
133
|
+
with tempfile.TemporaryDirectory(prefix='ccl-task-entry-') as directory:
|
|
134
|
+
root = Path(directory)
|
|
135
|
+
(root / 'hooks').mkdir()
|
|
136
|
+
(root / 'agent-context').mkdir()
|
|
137
|
+
shutil.copyfile(ROOT / 'hooks/task-entry.sh', root / 'hooks/task-entry.sh')
|
|
138
|
+
(root / 'agent-context/session-start.md').write_text(
|
|
139
|
+
(ROOT / 'agent-context/session-start.md').read_text())
|
|
140
|
+
result = subprocess.run(['bash', '-c', 'bash "$1" <&-', 'x', str(root / 'hooks/task-entry.sh')],
|
|
141
|
+
text=True, capture_output=True, timeout=5,
|
|
142
|
+
cwd=root, env={'PATH': os.environ['PATH']})
|
|
143
|
+
self.assertEqual(result.returncode, 0, result.stderr)
|
|
144
|
+
self.assertEqual(json.loads(result.stdout), expected)
|
|
145
|
+
|
|
83
146
|
def test_missing_or_invalid_source_is_observable_and_fail_soft(self):
|
|
84
147
|
for source, missing in [('', True), ('not a routing document', False),
|
|
85
148
|
('<!-- ccl:entry-routing:end -->', False),
|
|
@@ -34,10 +34,10 @@ Before completion or landing handoff, report the actual diff classification and
|
|
|
34
34
|
|
|
35
35
|
## Review continuation checkpoint
|
|
36
36
|
|
|
37
|
-
Necessary in-scope review inherits the existing task authority, including review of a fix made after an earlier pass. Five renewed runs trigger a progress checkpoint, not an authorization request. At that checkpoint, and before each later renewed run:
|
|
37
|
+
Necessary in-scope review inherits the existing task authority, including review of a fix made after an earlier pass. Five renewed runs trigger a progress checkpoint, not an authorization request. The controller counts conclusive runs in the worktree receipt and, from the sixth, returns `continuation_checkpoint` in its output; a review that records no receipt is counted by you. At that checkpoint, and before each later renewed run:
|
|
38
38
|
|
|
39
39
|
1. Check current scope and any explicit user stop, count, cost or time limit. Honor a host permission denial or unavailable resource through its normal recovery path; task authority never bypasses it. Ask only for a missing decision or authority that the next action actually needs.
|
|
40
|
-
2. Record what changed, the disposition of the previous findings, and what new evidence the next pass should obtain. When findings recur without progress, change the method or gather different evidence before calling again. Continue available in-scope repair and tests; park only work that needs an unavailable decision or resource.
|
|
40
|
+
2. Record what changed, the disposition of the previous findings, and what new evidence the next pass should obtain. When findings recur without progress, change the method or gather different evidence before calling again. When the same class of finding keeps returning, even though each instance was fixed, decide whether the capability that keeps producing it should be narrowed or removed before patching it again, and record that decision. Continue available in-scope repair and tests; park only work that needs an unavailable decision or resource.
|
|
41
41
|
3. Run the necessary review of the changed candidate without requesting permission per pass. Keep the tracked chain's own round ceiling and each invocation's timeout. A terminal chain stays terminal; use the owning lane's documented fresh-review or delta-pass path, and never reset an unchanged candidate's chain merely to obtain zero findings. Record the real pass history; an automatic pass is not a newly human-requested one.
|
|
42
42
|
|
|
43
43
|
The checkpoint waives no review, challenge, evidence or readiness requirement. A missing conclusive pass or unresolved P0/P1 remains pending regardless of how many passes have run. If the current candidate already has valid review and dispositions, reuse them and finish.
|
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py
CHANGED
|
@@ -5,6 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
|
|
6
6
|
import argparse
|
|
7
7
|
import errno
|
|
8
|
+
import fcntl
|
|
8
9
|
import hashlib
|
|
9
10
|
import json
|
|
10
11
|
import math
|
|
@@ -1646,10 +1647,67 @@ def open_directory_without_links(path: str) -> int:
|
|
|
1646
1647
|
return fd
|
|
1647
1648
|
|
|
1648
1649
|
|
|
1649
|
-
|
|
1650
|
-
|
|
1650
|
+
# development-completion.md "Review continuation checkpoint": the first review plus
|
|
1651
|
+
# five renewed runs. The count lives in the worktree's receipt because a count kept
|
|
1652
|
+
# in prose does not survive long sessions, compaction or a new session on the same
|
|
1653
|
+
# worktree, and a loop in which every round fixes something never looks stuck.
|
|
1654
|
+
REVIEW_CHECKPOINT_RUNS = 6
|
|
1655
|
+
|
|
1656
|
+
|
|
1657
|
+
def prior_local_review(receipt_fd: int) -> dict[str, Any]:
|
|
1658
|
+
try:
|
|
1659
|
+
descriptor = os.open(
|
|
1660
|
+
LOCAL_REVIEW_RECEIPT_FILE, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, dir_fd=receipt_fd
|
|
1661
|
+
)
|
|
1662
|
+
except OSError:
|
|
1663
|
+
return {}
|
|
1664
|
+
try:
|
|
1665
|
+
if not stat.S_ISREG(os.fstat(descriptor).st_mode):
|
|
1666
|
+
return {}
|
|
1667
|
+
data = os.read(descriptor, 65537)
|
|
1668
|
+
except OSError:
|
|
1669
|
+
# A prior receipt that cannot be read only loses the count, never the new receipt.
|
|
1670
|
+
return {}
|
|
1671
|
+
finally:
|
|
1672
|
+
os.close(descriptor)
|
|
1673
|
+
try:
|
|
1674
|
+
value = json.loads(data) if len(data) <= 65536 else None
|
|
1675
|
+
except (ValueError, RecursionError):
|
|
1676
|
+
return {}
|
|
1677
|
+
return value if isinstance(value, dict) else {}
|
|
1678
|
+
|
|
1679
|
+
|
|
1680
|
+
def conclusive_review_runs(prior: dict[str, Any], mode: Any) -> int:
|
|
1681
|
+
runs = prior.get("conclusive_runs")
|
|
1682
|
+
if not (isinstance(runs, int) and not isinstance(runs, bool) and runs >= 0):
|
|
1683
|
+
# A receipt written before the count existed still proves one run.
|
|
1684
|
+
runs = 1 if prior.get("mode") in ("review", "challenge") else 0
|
|
1685
|
+
return runs + (1 if mode in ("review", "challenge") else 0)
|
|
1686
|
+
|
|
1687
|
+
|
|
1688
|
+
def continuation_checkpoint(runs: Any, first_recorded_at: Any) -> dict[str, Any] | None:
|
|
1689
|
+
if not isinstance(runs, int) or isinstance(runs, bool) or runs < REVIEW_CHECKPOINT_RUNS:
|
|
1690
|
+
return None
|
|
1691
|
+
return {
|
|
1692
|
+
"conclusive_review_runs": runs,
|
|
1693
|
+
"first_review_recorded_at": first_recorded_at if isinstance(first_recorded_at, str) else None,
|
|
1694
|
+
"rule": "code-review references/development-completion.md#review-continuation-checkpoint",
|
|
1695
|
+
"before_next_run": [
|
|
1696
|
+
"record what changed and the disposition of the previous findings",
|
|
1697
|
+
"when the same class of finding keeps returning, even if each instance was fixed, decide whether the "
|
|
1698
|
+
"capability producing it should be narrowed or removed before patching it again",
|
|
1699
|
+
"check the remaining findings against the requester's own words; findings outside the request are "
|
|
1700
|
+
"dispositioned, not implemented",
|
|
1701
|
+
],
|
|
1702
|
+
}
|
|
1703
|
+
|
|
1704
|
+
|
|
1705
|
+
def record_local_review(anchor: dict[str, Any] | None, result: dict[str, Any]) -> dict[str, Any] | None:
|
|
1706
|
+
"""Best-effort local receipt of the last conclusive review; never fails it.
|
|
1707
|
+
|
|
1708
|
+
Returns the receipt it wrote, or None when nothing was recorded."""
|
|
1651
1709
|
if not anchor:
|
|
1652
|
-
return
|
|
1710
|
+
return None
|
|
1653
1711
|
receipt = {
|
|
1654
1712
|
"schema_version": 1,
|
|
1655
1713
|
"head": anchor["head"],
|
|
@@ -1673,12 +1731,19 @@ def record_local_review(anchor: dict[str, Any] | None, result: dict[str, Any]) -
|
|
|
1673
1731
|
# under it since then is not the one the anchor described.
|
|
1674
1732
|
opened = os.fstat(git_fd)
|
|
1675
1733
|
if [opened.st_dev, opened.st_ino] != anchor.get("git_dir_identity"):
|
|
1676
|
-
return
|
|
1734
|
+
return None
|
|
1677
1735
|
try:
|
|
1678
1736
|
os.mkdir(LOCAL_REVIEW_RECEIPT_DIR, 0o700, dir_fd=git_fd)
|
|
1679
1737
|
except FileExistsError:
|
|
1680
1738
|
pass
|
|
1681
1739
|
receipt_fd = os.open(LOCAL_REVIEW_RECEIPT_DIR, directory_flags, dir_fd=git_fd)
|
|
1740
|
+
# Serialize read-increment-replace so overlapping review and challenge
|
|
1741
|
+
# completions in one worktree cannot both read N and write N + 1.
|
|
1742
|
+
fcntl.flock(receipt_fd, fcntl.LOCK_EX)
|
|
1743
|
+
prior = prior_local_review(receipt_fd)
|
|
1744
|
+
receipt["conclusive_runs"] = conclusive_review_runs(prior, receipt["mode"])
|
|
1745
|
+
first = prior.get("first_recorded_at") or prior.get("recorded_at")
|
|
1746
|
+
receipt["first_recorded_at"] = first if isinstance(first, str) else receipt["recorded_at"]
|
|
1682
1747
|
temporary = f".last-review.{os.getpid()}.{secrets.token_hex(8)}"
|
|
1683
1748
|
file_fd = os.open(
|
|
1684
1749
|
temporary,
|
|
@@ -1697,8 +1762,9 @@ def record_local_review(anchor: dict[str, Any] | None, result: dict[str, Any]) -
|
|
|
1697
1762
|
dst_dir_fd=receipt_fd,
|
|
1698
1763
|
)
|
|
1699
1764
|
temporary = None
|
|
1765
|
+
return receipt
|
|
1700
1766
|
except OSError:
|
|
1701
|
-
|
|
1767
|
+
return None
|
|
1702
1768
|
finally:
|
|
1703
1769
|
if temporary is not None and receipt_fd is not None:
|
|
1704
1770
|
try:
|
|
@@ -5190,7 +5256,12 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
5190
5256
|
if time.monotonic() >= gate_deadline:
|
|
5191
5257
|
apply_gate_timeout(result)
|
|
5192
5258
|
return emit(result, 2)
|
|
5193
|
-
record_local_review(review_anchor, result)
|
|
5259
|
+
receipt = record_local_review(review_anchor, result)
|
|
5260
|
+
checkpoint = continuation_checkpoint(
|
|
5261
|
+
(receipt or {}).get("conclusive_runs"), (receipt or {}).get("first_recorded_at")
|
|
5262
|
+
)
|
|
5263
|
+
if checkpoint:
|
|
5264
|
+
result["continuation_checkpoint"] = checkpoint
|
|
5194
5265
|
return emit(result, 0)
|
|
5195
5266
|
if completed.returncode == 0 and status in {"passed", "findings"}:
|
|
5196
5267
|
payload.update(
|
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh
CHANGED
|
@@ -4931,6 +4931,118 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
4931
4931
|
check "a bare --diff-file review records no receipt" \
|
|
4932
4932
|
'[ "$rc" = 0 ] && json_fields "$out" status=passed && [ ! -e "$contract_repo/.git/ccl-code-review" ]'
|
|
4933
4933
|
|
|
4934
|
+
# The continuation checkpoint is counted, not remembered: the receipt carries the
|
|
4935
|
+
# worktree's conclusive review runs, and the sixth (the first review plus five
|
|
4936
|
+
# renewed runs) brings the checkpoint into the gate's own output.
|
|
4937
|
+
counting_out="$(PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY' 2>&1
|
|
4938
|
+
import json, os, sys, review_gate
|
|
4939
|
+
root = os.path.realpath(os.path.join(sys.argv[1], "run-count"))
|
|
4940
|
+
gd = os.path.join(root, "gitdir")
|
|
4941
|
+
os.makedirs(gd)
|
|
4942
|
+
st = os.stat(gd)
|
|
4943
|
+
anchor = {"git_dir": gd, "git_dir_identity": [st.st_dev, st.st_ino], "head": "a" * 40, "worktree_clean": True}
|
|
4944
|
+
receipt_path = os.path.join(gd, "ccl-code-review", "last-review.json")
|
|
4945
|
+
runs = [review_gate.record_local_review(anchor, {"mode": mode, "status": "findings"})["conclusive_runs"]
|
|
4946
|
+
for mode in ["review", "challenge", "review", "complete", "review", "review", "challenge"]]
|
|
4947
|
+
assert runs == [1, 2, 3, 3, 4, 5, 6], runs
|
|
4948
|
+
first = json.load(open(receipt_path))["first_recorded_at"]
|
|
4949
|
+
assert review_gate.continuation_checkpoint(5, first) is None
|
|
4950
|
+
assert review_gate.continuation_checkpoint(True, first) is None
|
|
4951
|
+
cp = review_gate.continuation_checkpoint(6, first)
|
|
4952
|
+
assert cp["conclusive_review_runs"] == 6 and cp["first_review_recorded_at"] == first, cp
|
|
4953
|
+
assert any("narrowed or removed" in step for step in cp["before_next_run"]), cp
|
|
4954
|
+
assert any("own words" in step for step in cp["before_next_run"]), cp
|
|
4955
|
+
# A receipt written before the count existed still proves one run.
|
|
4956
|
+
json.dump({"mode": "review", "recorded_at": "2026-01-01T00:00:00Z"}, open(receipt_path, "w"))
|
|
4957
|
+
receipt = review_gate.record_local_review(anchor, {"mode": "review"})
|
|
4958
|
+
assert receipt["conclusive_runs"] == 2 and receipt["first_recorded_at"] == "2026-01-01T00:00:00Z", receipt
|
|
4959
|
+
# Corrupt or hostile prior values restart the count; they never cost the new receipt.
|
|
4960
|
+
for bad in ["not json", json.dumps({"conclusive_runs": True, "mode": "complete"}),
|
|
4961
|
+
json.dumps({"conclusive_runs": -3}), json.dumps([1]), "x" * 70000]:
|
|
4962
|
+
open(receipt_path, "w").write(bad)
|
|
4963
|
+
receipt = review_gate.record_local_review(anchor, {"mode": "review"})
|
|
4964
|
+
assert receipt is not None and receipt["conclusive_runs"] == 1, (bad[:20], receipt)
|
|
4965
|
+
# A prior receipt that is a link is never read through, and the write replaces the link.
|
|
4966
|
+
target = os.path.join(root, "elsewhere.json")
|
|
4967
|
+
json.dump({"conclusive_runs": 40, "mode": "review"}, open(target, "w"))
|
|
4968
|
+
os.unlink(receipt_path)
|
|
4969
|
+
os.symlink(target, receipt_path)
|
|
4970
|
+
receipt = review_gate.record_local_review(anchor, {"mode": "review"})
|
|
4971
|
+
assert receipt is not None and receipt["conclusive_runs"] == 1 and not os.path.islink(receipt_path), receipt
|
|
4972
|
+
assert json.load(open(target))["conclusive_runs"] == 40
|
|
4973
|
+
# A prior receipt the parser cannot finish (nesting too deep) is invalid, never fatal.
|
|
4974
|
+
real_loads = review_gate.json.loads
|
|
4975
|
+
def too_deep(*args, **kwargs):
|
|
4976
|
+
raise RecursionError("maximum recursion depth exceeded")
|
|
4977
|
+
review_gate.json.loads = too_deep
|
|
4978
|
+
open(receipt_path, "w").write("[" * 30000 + "]" * 30000)
|
|
4979
|
+
receipt = review_gate.record_local_review(anchor, {"mode": "review"})
|
|
4980
|
+
review_gate.json.loads = real_loads
|
|
4981
|
+
assert receipt is not None and receipt["conclusive_runs"] == 1, receipt
|
|
4982
|
+
# Overlapping completions cannot lose an increment because the whole
|
|
4983
|
+
# read-increment-replace runs under an exclusive lock on the receipt directory.
|
|
4984
|
+
# Checked where it matters, not by racing writers: when the controller opens the
|
|
4985
|
+
# prior receipt and when it replaces it, a second open of that directory cannot
|
|
4986
|
+
# take even a shared lock, and no lock call falls between the two, so the lock
|
|
4987
|
+
# is never released or changed in between.
|
|
4988
|
+
import fcntl
|
|
4989
|
+
receipt_dir = os.path.join(gd, "ccl-code-review")
|
|
4990
|
+
real_open, real_replace, real_flock = review_gate.os.open, review_gate.os.replace, review_gate.fcntl.flock
|
|
4991
|
+
def exclusively_locked():
|
|
4992
|
+
probe = real_open(receipt_dir, os.O_RDONLY | os.O_DIRECTORY)
|
|
4993
|
+
try:
|
|
4994
|
+
real_flock(probe, fcntl.LOCK_SH | fcntl.LOCK_NB)
|
|
4995
|
+
except BlockingIOError:
|
|
4996
|
+
return True
|
|
4997
|
+
finally:
|
|
4998
|
+
os.close(probe)
|
|
4999
|
+
return False
|
|
5000
|
+
seen = []
|
|
5001
|
+
def watched_open(path, flags, *args, **kwargs):
|
|
5002
|
+
if path == review_gate.LOCAL_REVIEW_RECEIPT_FILE and not flags & (os.O_WRONLY | os.O_RDWR):
|
|
5003
|
+
seen.append(("read", exclusively_locked()))
|
|
5004
|
+
return real_open(path, flags, *args, **kwargs)
|
|
5005
|
+
def watched_replace(src, dst, *args, **kwargs):
|
|
5006
|
+
if dst == review_gate.LOCAL_REVIEW_RECEIPT_FILE:
|
|
5007
|
+
seen.append(("replace", exclusively_locked()))
|
|
5008
|
+
return real_replace(src, dst, *args, **kwargs)
|
|
5009
|
+
def watched_flock(fd, operation):
|
|
5010
|
+
seen.append(("flock", operation))
|
|
5011
|
+
return real_flock(fd, operation)
|
|
5012
|
+
with open(receipt_path, "w") as handle:
|
|
5013
|
+
json.dump({"mode": "review", "conclusive_runs": 40, "first_recorded_at": "2026-01-01T00:00:00Z"}, handle)
|
|
5014
|
+
review_gate.os.open, review_gate.os.replace, review_gate.fcntl.flock = watched_open, watched_replace, watched_flock
|
|
5015
|
+
try:
|
|
5016
|
+
receipt = review_gate.record_local_review(anchor, {"mode": "challenge"})
|
|
5017
|
+
finally:
|
|
5018
|
+
review_gate.os.open, review_gate.os.replace, review_gate.fcntl.flock = real_open, real_replace, real_flock
|
|
5019
|
+
kinds = [kind for kind, _ in seen]
|
|
5020
|
+
assert kinds.count("read") == 1 and kinds.count("replace") == 1, seen
|
|
5021
|
+
read_at, replace_at = kinds.index("read"), kinds.index("replace")
|
|
5022
|
+
assert read_at < replace_at and seen[read_at][1] and seen[replace_at][1], seen
|
|
5023
|
+
assert "flock" not in kinds[read_at:replace_at], seen
|
|
5024
|
+
stored = json.load(open(receipt_path))
|
|
5025
|
+
assert stored == receipt and stored["conclusive_runs"] == 41, (stored, receipt)
|
|
5026
|
+
assert stored["first_recorded_at"] == "2026-01-01T00:00:00Z", stored
|
|
5027
|
+
print("run_count_ok")
|
|
5028
|
+
PY
|
|
5029
|
+
)"
|
|
5030
|
+
check "conclusive review runs are counted per worktree; the checkpoint starts at the sixth" \
|
|
5031
|
+
'[ "$counting_out" = run_count_ok ]'
|
|
5032
|
+
|
|
5033
|
+
mkdir -p "$contract_repo/.git/ccl-code-review"
|
|
5034
|
+
printf '{"mode":"review","conclusive_runs":5,"first_recorded_at":"2026-01-01T00:00:00Z"}\n' >"$receipt_file"
|
|
5035
|
+
reset_case passed unavailable unavailable
|
|
5036
|
+
out="$(run_contract_gate --mode review)"; rc=$?
|
|
5037
|
+
check "the sixth conclusive run carries the continuation checkpoint in the gate output" \
|
|
5038
|
+
'[ "$rc" = 0 ] && json_fields "$out" continuation_checkpoint.conclusive_review_runs=6 continuation_checkpoint.first_review_recorded_at=2026-01-01T00:00:00Z && [ "$(jq -r .conclusive_runs "$receipt_file")" = 6 ]'
|
|
5039
|
+
printf '{"mode":"review","conclusive_runs":3}\n' >"$receipt_file"
|
|
5040
|
+
reset_case passed unavailable unavailable
|
|
5041
|
+
out="$(run_contract_gate --mode review)"; rc=$?
|
|
5042
|
+
check "an earlier run carries no checkpoint" \
|
|
5043
|
+
'[ "$rc" = 0 ] && json_lacks "$out" continuation_checkpoint && [ "$(jq -r .conclusive_runs "$receipt_file")" = 4 ]'
|
|
5044
|
+
rm -rf "$contract_repo/.git/ccl-code-review"
|
|
5045
|
+
|
|
4934
5046
|
# The completion checkpoint is what the pull-request reminder reads as disposed:
|
|
4935
5047
|
# on the same whole-worktree candidate it replaces the review's receipt with a
|
|
4936
5048
|
# passed one; a checkpoint that fails leaves the receipt as it was.
|
|
@@ -762,3 +762,9 @@ The pending classification above is superseded by the executed source comparison
|
|
|
762
762
|
| A review does not report a requested fix of a pre-existing defect as droppable, and the requester's words in `--focus` reach every reviewer under the egress scan | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. An adversarial challenge read "each fix for a risk that predates the change" as asking reviewers to drop a requested fix of an older defect. In the build and release concern and in the manual prompt, the clause now covers only a pre-existing risk that the request does not cover and the change does not expose or worsen; the controller test fails on the previous text. A requested-fix replay (neutral domain, read by hand) found no run calling the fix droppable under either wording (Claude 0/6 each, Codex 0/3 each), so the reading did not reproduce and the narrowing aligns the text with its intent. New tests show `--focus` reaching the reviewer profile and a fallback reviewer, and a credential-shaped value blocking non-Claude egress; controller copies that drop the focus or skip the profile scan fail them. A reviewer-scope rerun read by hand: only the new text called an unrequested configuration switch unneeded (0/4 to 4/4). The matched-plan control behind the earlier row first carried an unrequested weekly step that reviewers flagged (Claude 5/6, Codex 3/3) and a regex had miscounted; with the step removed, no run raised a scope finding. |
|
|
763
763
|
| A deterministic check that `make test` does not run surfaces only in CI after a push; the real-repository ledger audit runs in the fast lane and names its fix, and the delta-pass step names its entrypoint for a delta the extraction lane does not own | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. In one reviewed round CI went red on a stale line-cited obligation ledger after `make test` and the quick checker were green: the real-repository audit ran only in the heavy lane. With one line inserted above a cited carrier, main's fast lane passed and the candidate's fails on that audit, printing a `render` command that clears it when run as printed; a carrier whose text changed fails render and audit with another code, so the hint cannot hide a dropped obligation. `test_obligation_ledger.sh` runs the printed command and fails against the previous tool. Two rounds improvised chain ids after the extraction wrapper refused a delta it did not own; the delta-pass step in `dual-track-review-gate.md` now names both entrypoints, and the generic call it documents passes the controller's preconditions. |
|
|
764
764
|
| The extraction lane's ownership refusal points at the delta-pass recipe for a delta it does not own | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. A delta pass over files the extraction lane does not own was refused with no route forward, and two rounds improvised a review-chain call to the generic controller. The refusal now names the delta-pass step that documents the call; the controller test asserts it and fails against the previous controller, where it is the only failure. The ownership precondition itself is unchanged. |
|
|
765
|
+
| The review controller counts conclusive runs per worktree and returns the continuation checkpoint from the sixth; the checkpoint names the same-class decision | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. In a month of sessions single changes ran 13 to 49 conclusive review runs, several after the five-run checkpoint existed; the controller kept only the last receipt, so the count lived in the agent's memory across hours and compactions. The receipt now carries `conclusive_runs` (review and challenge only) and the output carries `continuation_checkpoint` from the sixth run; the suite's three new checks fail against the previous controller, which after six runs records no count. Replay with only the latest finding visible: with the checkpoint object 6/6 answers stepped back before patching, without it 2/6. The added same-class sentence alone measured at ceiling (6/6 with the old or the new text when three findings are listed side by side) and is not claimed as a behavior change. |
|
|
766
|
+
| The review receipt counter keeps counting through an unparseable prior receipt and overlapping writers | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. The round's adversarial challenge found that a prior receipt nested deeply enough to make the parser raise RecursionError escaped the corrupt-receipt fallback, and that overlapping review and challenge completions could both read N and write N + 1. The parser failure now counts as an invalid receipt, and the read-increment-replace holds a lock on the receipt directory. In the suite's counting unit the previous controller fails on the recursion case; with the lock removed, 24 overlapping writers recorded 4 to 7 runs in three trials, and 24 with it. |
|
|
767
|
+
| The receipt lock check holds the lock itself instead of relying on writers overlapping | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. The round's delta review found that the overlapping-writer check proves the lock only when the writers happen to overlap; run one after another, they would pass with the lock removed. The check now takes the receipt directory lock itself, requires the writer to be still waiting a second after it started, rewrites the count while holding the lock, and requires the writer to increment that value. Against copies of the controller, removing the lock failed 8 of 8 runs on the waiting check, reading the prior receipt before taking the lock failed 8 of 8 on the count, and the unchanged controller passed 8 of 8. The previous check also failed both mutants in 5 of 5 runs on the same machine, so the gain is independence from scheduling, not a newly caught defect. |
|
|
768
|
+
| The receipt lock check signals from the writer's own lock call, so no wait decides its result | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. Supersedes the previous row's claim of scheduling independence: the next delta review found that a writer paused between signalling its start and reaching the lock lets both mutants pass once the one-second wait expires. The writer now signals from inside its exclusive lock call on the receipt directory, and the holder rewrites the count only after that signal; unlock and cleanup run in `finally`, the writer is a daemon and every wait is bounded. Against copies of the controller, each run with and without a forced two-second pause in the writer: the unchanged controller passed 8 of 8; removing the lock and taking a shared lock each failed 8 of 8 on the missing signal; reading the prior receipt before the lock failed 8 of 8 on the count. |
|
|
769
|
+
| The receipt lock check asserts the exclusive lock at the receipt read and replace instead of racing writers | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. Supersedes the two rows above for the check's design. Three successive delta reviews each found a controller variant that a racing-writer check let through: writers that did not overlap, a writer paused past a wait, a lock released before the read. The class was closed by changing the method instead of patching a fourth time. When the controller opens the prior receipt and when it replaces it, the check requires that a second open of the receipt directory cannot take even a shared lock, and then that the count increments the value read. Against copies of the controller, three runs each, the unchanged controller passed; removing the lock, a shared lock, a lock on another descriptor, reading before the lock, unlocking before the read and unlocking before the replace each failed on the recorded lock states. |
|
|
770
|
+
| The receipt lock check also requires the lock to be held without interruption between the read and the replace, and reads the stored receipt back | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md`. The delta review of the previous row's check, the sixth conclusive run in the worktree and the first to return the continuation checkpoint, found that probes at the read and the replace pass a controller that unlocks and relocks in between, and that the check trusted the returned receipt over the stored one. The check now records every lock call during the controller's call and requires none between the read and the replace, and it compares the stored receipt with the returned one. Against copies of the controller, three runs each: the unchanged controller passed; unlocking and relocking between the read and the replace failed on the recorded lock calls, and a stored count or start time that differs from the returned one, applied to the last call only, failed on the stored receipt; the six variants from the previous row still failed. |
|
package/dist/assets/release.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schema": 1,
|
|
3
3
|
"npmPackage": "@ccoalm/ccl-skills",
|
|
4
|
-
"version": "0.18.
|
|
5
|
-
"sourceCommit": "
|
|
4
|
+
"version": "0.18.11",
|
|
5
|
+
"sourceCommit": "1e07f1c777ba010542c5d97d5eaa9c3c64506f38",
|
|
6
6
|
"sourceState": "clean",
|
|
7
7
|
"files": [
|
|
8
8
|
{
|
|
@@ -82,7 +82,7 @@
|
|
|
82
82
|
},
|
|
83
83
|
{
|
|
84
84
|
"path": "marketplace/plugins/ccl-skills/hooks/host-input.py",
|
|
85
|
-
"sha256": "
|
|
85
|
+
"sha256": "34a1f1c7b5dfc22b80d739663683cfebd30434800987db16f50423246d79f0ad",
|
|
86
86
|
"mode": 420
|
|
87
87
|
},
|
|
88
88
|
{
|
|
@@ -157,7 +157,7 @@
|
|
|
157
157
|
},
|
|
158
158
|
{
|
|
159
159
|
"path": "marketplace/plugins/ccl-skills/hooks/task-entry.sh",
|
|
160
|
-
"sha256": "
|
|
160
|
+
"sha256": "577cd5341e2540d3ae3b0ac7910fd281b360ae34a6a09c8c852827c36ec7f6f9",
|
|
161
161
|
"mode": 493
|
|
162
162
|
},
|
|
163
163
|
{
|
|
@@ -187,7 +187,7 @@
|
|
|
187
187
|
},
|
|
188
188
|
{
|
|
189
189
|
"path": "marketplace/plugins/ccl-skills/hooks/test_proposed_next.py",
|
|
190
|
-
"sha256": "
|
|
190
|
+
"sha256": "1de3753e036f87aa1823e0c53ce31c39840501af4ec6667ab62b6c04f0d716f1",
|
|
191
191
|
"mode": 493
|
|
192
192
|
},
|
|
193
193
|
{
|
|
@@ -222,7 +222,7 @@
|
|
|
222
222
|
},
|
|
223
223
|
{
|
|
224
224
|
"path": "marketplace/plugins/ccl-skills/hooks/test_task_entry.py",
|
|
225
|
-
"sha256": "
|
|
225
|
+
"sha256": "f86e65ba1301882568a64d5d4665d8843d0b4871b860fb65dd4ed477744145e8",
|
|
226
226
|
"mode": 493
|
|
227
227
|
},
|
|
228
228
|
{
|
|
@@ -347,7 +347,7 @@
|
|
|
347
347
|
},
|
|
348
348
|
{
|
|
349
349
|
"path": "marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md",
|
|
350
|
-
"sha256": "
|
|
350
|
+
"sha256": "1047377aaf3f42498ec2b300316b801aea2311b2684ed8a944b392d21e26f0ce",
|
|
351
351
|
"mode": 420
|
|
352
352
|
},
|
|
353
353
|
{
|
|
@@ -452,7 +452,7 @@
|
|
|
452
452
|
},
|
|
453
453
|
{
|
|
454
454
|
"path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py",
|
|
455
|
-
"sha256": "
|
|
455
|
+
"sha256": "ba57bfe8478550c1be3eb826de0638890c03a316a2cc9e7d086005e2495fd900",
|
|
456
456
|
"mode": 493
|
|
457
457
|
},
|
|
458
458
|
{
|
|
@@ -557,7 +557,7 @@
|
|
|
557
557
|
},
|
|
558
558
|
{
|
|
559
559
|
"path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh",
|
|
560
|
-
"sha256": "
|
|
560
|
+
"sha256": "888c4293e4fd091a48a715295369566e01c2c06397f2f019431e22c6a55041f9",
|
|
561
561
|
"mode": 493
|
|
562
562
|
},
|
|
563
563
|
{
|
|
@@ -2207,7 +2207,7 @@
|
|
|
2207
2207
|
},
|
|
2208
2208
|
{
|
|
2209
2209
|
"path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md",
|
|
2210
|
-
"sha256": "
|
|
2210
|
+
"sha256": "0c69b7ef4e68d579be780328551153114b7f0960e4adc6f2be54406b21fc5d28",
|
|
2211
2211
|
"mode": 420
|
|
2212
2212
|
},
|
|
2213
2213
|
{
|
|
@@ -3533,5 +3533,5 @@
|
|
|
3533
3533
|
"mode": 420
|
|
3534
3534
|
}
|
|
3535
3535
|
],
|
|
3536
|
-
"snapshotHash": "
|
|
3536
|
+
"snapshotHash": "28d8ffdb54ff423aa8c7947ac1ace08ac32d314ff0a71b80697caf29bb0b85f2"
|
|
3537
3537
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ccoalm/ccl-skills",
|
|
3
|
-
"version": "0.18.
|
|
3
|
+
"version": "0.18.11",
|
|
4
4
|
"description": "Reusable workflows that help coding agents plan, build, test, review, and release software — for Claude Code, Codex, and OpenCode",
|
|
5
5
|
"keywords": ["skills", "agent-skills", "claude", "claude-code", "codex", "opencode", "agent", "ai", "ai-agents", "cli", "anthropic", "developer-tools"],
|
|
6
6
|
"type": "module",
|