@ccoalm/ccl-skills 0.18.5 → 0.18.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-policy.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +11 -9
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/host-input.py +99 -10
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-review-covers-head.sh +34 -9
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/skill-extraction-gate-stop.sh +14 -3
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/skill-loading.py +39 -2
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_host_input.py +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_proposed_next.py +114 -3
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_review_covers_head.sh +19 -3
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_skill_loading.py +77 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh +65 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +4 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +158 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +39 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +2 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +2 -2
- package/dist/assets/release.json +25 -25
- package/package.json +1 -1
|
@@ -77,6 +77,83 @@ class SkillLoadingTests(unittest.TestCase):
|
|
|
77
77
|
self.assertNotIn('permissionDecision":"ask', json.dumps(result))
|
|
78
78
|
self.assertEqual(self.run_hook(self.payload()), {})
|
|
79
79
|
|
|
80
|
+
def ccl_checkout(self):
|
|
81
|
+
root = self.root / 'ccl'
|
|
82
|
+
(root / 'skills' / 'skill-extraction-workflow').mkdir(parents=True)
|
|
83
|
+
(root / 'skills' / 'skill-extraction-workflow' / 'SKILL.md').write_text('marker')
|
|
84
|
+
(root / 'skills' / 'sample-owner' / 'references').mkdir(parents=True)
|
|
85
|
+
(root / 'hooks').mkdir()
|
|
86
|
+
return root
|
|
87
|
+
|
|
88
|
+
def extraction_loaded(self, ident='ext'):
|
|
89
|
+
self.append({'type': 'assistant', 'message': {'content': [
|
|
90
|
+
{'type': 'tool_use', 'id': ident, 'name': 'Skill',
|
|
91
|
+
'input': {'skill': 'ccl-skills:skill-extraction-workflow'}}]}},
|
|
92
|
+
{'type': 'user', 'message': {'content': [
|
|
93
|
+
{'type': 'tool_result', 'tool_use_id': ident, 'content': 'Loaded owner'}]}})
|
|
94
|
+
|
|
95
|
+
def test_shared_skill_edit_names_the_extraction_owner_at_the_first_edit(self):
|
|
96
|
+
root = self.ccl_checkout()
|
|
97
|
+
target = root / 'skills' / 'sample-owner' / 'scripts' / 'gate.sh'
|
|
98
|
+
target.parent.mkdir(parents=True)
|
|
99
|
+
result = self.run_hook(self.payload(tool_input={'file_path': str(target)}))
|
|
100
|
+
self.assertEqual(self.decision(result), 'deny')
|
|
101
|
+
reason = result['hookSpecificOutput']['permissionDecisionReason']
|
|
102
|
+
self.assertIn('skill-extraction-workflow', reason)
|
|
103
|
+
self.assertIn('charter', reason.lower())
|
|
104
|
+
|
|
105
|
+
def test_shared_skill_markdown_edit_reaches_the_checkpoint(self):
|
|
106
|
+
root = self.ccl_checkout()
|
|
107
|
+
target = root / 'skills' / 'sample-owner' / 'references' / 'rule.md'
|
|
108
|
+
result = self.run_hook(self.payload(tool_input={'file_path': str(target)}))
|
|
109
|
+
self.assertEqual(self.decision(result), 'deny')
|
|
110
|
+
self.assertIn('skill-extraction-workflow',
|
|
111
|
+
result['hookSpecificOutput']['permissionDecisionReason'])
|
|
112
|
+
|
|
113
|
+
def test_markdown_outside_a_ccl_checkout_still_skips_the_checkpoint(self):
|
|
114
|
+
outside = self.root / 'product' / 'skills' / 'thing' / 'notes.md'
|
|
115
|
+
outside.parent.mkdir(parents=True)
|
|
116
|
+
self.assertEqual(self.run_hook(self.payload(tool_input={'file_path': str(outside)})), {})
|
|
117
|
+
cached = self.ccl_checkout() / 'skills' / 'sample-owner' / 'SKILL.md'
|
|
118
|
+
# An installed plugin copy carries the same marker; only the cache exclusion skips it.
|
|
119
|
+
cache_root = self.root / 'plugins' / 'cache' / 'ccl'
|
|
120
|
+
(cache_root / 'skills' / 'skill-extraction-workflow').mkdir(parents=True)
|
|
121
|
+
(cache_root / 'skills' / 'skill-extraction-workflow' / 'SKILL.md').write_text('marker')
|
|
122
|
+
plugin_cache = cache_root / 'skills' / 'x' / 'SKILL.md'
|
|
123
|
+
plugin_cache.parent.mkdir(parents=True)
|
|
124
|
+
self.assertEqual(self.run_hook(self.payload(tool_input={'file_path': str(plugin_cache)})), {})
|
|
125
|
+
# The checkpoint is still unspent for a real shared-skill edit.
|
|
126
|
+
result = self.run_hook(self.payload(tool_input={'file_path': str(cached)}))
|
|
127
|
+
self.assertEqual(self.decision(result), 'deny')
|
|
128
|
+
|
|
129
|
+
def test_checkout_under_an_ancestor_named_like_a_surface_is_recognised(self):
|
|
130
|
+
root = self.root / 'skills' / 'ccl'
|
|
131
|
+
(root / 'skills' / 'skill-extraction-workflow').mkdir(parents=True)
|
|
132
|
+
(root / 'skills' / 'skill-extraction-workflow' / 'SKILL.md').write_text('marker')
|
|
133
|
+
target = root / 'skills' / 'sample-owner' / 'SKILL.md'
|
|
134
|
+
target.parent.mkdir(parents=True)
|
|
135
|
+
result = self.run_hook(self.payload(tool_input={'file_path': str(target)}))
|
|
136
|
+
self.assertEqual(self.decision(result), 'deny')
|
|
137
|
+
self.assertIn('skill-extraction-workflow',
|
|
138
|
+
result['hookSpecificOutput']['permissionDecisionReason'])
|
|
139
|
+
|
|
140
|
+
def test_codex_install_copy_is_exempt_like_the_plugin_cache(self):
|
|
141
|
+
install = self.root / '.codex' / 'ccl'
|
|
142
|
+
(install / 'skills' / 'skill-extraction-workflow').mkdir(parents=True)
|
|
143
|
+
(install / 'skills' / 'skill-extraction-workflow' / 'SKILL.md').write_text('marker')
|
|
144
|
+
target = install / 'skills' / 'x' / 'SKILL.md'
|
|
145
|
+
target.parent.mkdir(parents=True)
|
|
146
|
+
self.assertEqual(self.run_hook(self.payload(tool_input={'file_path': str(target)})), {})
|
|
147
|
+
self.assertEqual(self.decision(self.run_hook(self.payload())), 'deny')
|
|
148
|
+
|
|
149
|
+
def test_shared_skill_edit_with_the_owner_loaded_keeps_the_generic_reason(self):
|
|
150
|
+
root = self.ccl_checkout()
|
|
151
|
+
self.extraction_loaded()
|
|
152
|
+
target = root / 'hooks' / 'gate.sh'
|
|
153
|
+
result = self.run_hook(self.payload(tool_input={'file_path': str(target)}))
|
|
154
|
+
self.assertEqual(self.decision(result), 'deny')
|
|
155
|
+
self.assertNotIn('charter', result['hookSpecificOutput']['permissionDecisionReason'].lower())
|
|
156
|
+
|
|
80
157
|
def test_native_patch_and_all_targets(self):
|
|
81
158
|
patch = '*** Begin Patch\n*** Add File: README.md\n+doc\n*** Add File: src/new.ts\n+code\n*** End Patch'
|
|
82
159
|
value = self.run_hook(self.payload('apply_patch', tool_input={'command': patch}))
|
|
@@ -23,7 +23,7 @@ A review quotes the repository's own tracked contract files for the changed path
|
|
|
23
23
|
|
|
24
24
|
Walk these before you open a pull or merge request, push to one that is already open, ask for a merge, or report the work done or ready — every time, not only after the first review:
|
|
25
25
|
|
|
26
|
-
1. Name the commit or packet the last conclusive review covered and compare it with the current candidate: HEAD plus staged, unstaged and untracked implementation files. For a review whose candidate `--base` derived from the whole worktree, the controller records that commit in the worktree's git directory (`ccl-code-review/last-review.json`; a bare `--diff-file` or `--paths` review records nothing), and the plugin's pull-request hook repeats this comparison when you open, ready or merge one; its reminder is this step firing, not a new question.
|
|
26
|
+
1. Name the commit or packet the last conclusive review covered and compare it with the current candidate: HEAD plus staged, unstaged and untracked implementation files. For a review whose candidate `--base` derived from the whole worktree, the controller records that commit in the worktree's git directory (`ccl-code-review/last-review.json`; a bare `--diff-file` or `--paths` review records nothing), and the plugin's pull-request hook repeats this comparison when you open, ready or merge one; its reminder is this step firing, not a new question. A successful completion checkpoint on such a candidate records it as well, so a chain whose findings were disposed reads as passed; coverage alone is not disposition, and a covered receipt with findings or no status must still be dispositioned before a ready report.
|
|
27
27
|
2. Any difference makes the current candidate unreviewed: a fix for a finding of any severity, an added or changed test, a changelog or doc line, a rebase that changed a file the candidate touches. Only the review's own record files (result JSON, disposition notes) are exempt.
|
|
28
28
|
3. An unreviewed candidate must get the owning gate's renewed review now, run by you: a fresh full run per [manual invocation](manual-invocation-and-prompts.md); the skill-extraction lane runs its own delta pass instead. Pushing it for a human to review does not discharge it, and "awaiting human review" never stands in for the run.
|
|
29
29
|
4. Apply the [review continuation checkpoint](#review-continuation-checkpoint) after five renewed runs or earlier when findings recur without progress. Necessary review continues within the task's authority; unresolved P0/P1 or unreviewed changes still prevent a ready report.
|
|
@@ -10,7 +10,8 @@ Rules:
|
|
|
10
10
|
their arguments, returned bytes and completed lifecycle. This does not permit
|
|
11
11
|
arbitrary commands or workspace access. A malformed, timeout, or inconclusive
|
|
12
12
|
wrapper result is not a pass.
|
|
13
|
-
- **Never pin
|
|
13
|
+
- **Never pin a lane to a CLI version's vocabulary — reading it or writing it.**
|
|
14
|
+
`parse_probe_result.py`
|
|
14
15
|
gates on *shape*, not on field/value names: the isolation proof is the exact
|
|
15
16
|
`tools` allowlist plus the tool_use scan, which no init field can bypass.
|
|
16
17
|
Field-name and value whitelists were tried twice (`fast_mode_state`, then
|
|
@@ -19,6 +20,15 @@ Rules:
|
|
|
19
20
|
or empty-container, and `agents`/`capabilities` are list-of-strings checks
|
|
20
21
|
with no value vocabulary. Only `permissionMode` stays value-pinned (it widens
|
|
21
22
|
what the runtime may do with no tool added).
|
|
23
|
+
The write side is the same class and was missed once: config a wrapper
|
|
24
|
+
GENERATES and then submits to the runtime's own validator as an admission
|
|
25
|
+
precondition carries a key that release may not know, and rejecting it turns
|
|
26
|
+
the whole lane off. So a generated setting that a per-invocation environment
|
|
27
|
+
variable already carries is belt, not the gate — on rejection regenerate the
|
|
28
|
+
strict subset without it and revalidate, unconditionally rather than by
|
|
29
|
+
matching the runtime's error wording, which is the same pin relocated. Only a
|
|
30
|
+
setting with no second, independent path to the same property may fail the
|
|
31
|
+
lane closed, and admission must never widen on the retry.
|
|
22
32
|
- **Skill, command and plugin lists are vocabulary, not a boundary.**
|
|
23
33
|
`slash_commands`, `terminal_slash_commands`, `skills` and `plugins` may hold
|
|
24
34
|
any value and are never judged by name, shape, or origin; only `mcp_servers`
|
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh
CHANGED
|
@@ -721,7 +721,7 @@ fi
|
|
|
721
721
|
# prevention. The cooperative probe below attempts only private-workspace
|
|
722
722
|
# Read/Glob/Grep canaries and rejects any observed tool exposure.
|
|
723
723
|
install_packet_only_config() {
|
|
724
|
-
"$PYTHON_BIN_PATH" - "$RUNTIME_HOME/config.toml" "$PACKET_DELIVERY" <<'PY'
|
|
724
|
+
"$PYTHON_BIN_PATH" - "$RUNTIME_HOME/config.toml" "$PACKET_DELIVERY" "${1:-guard}" <<'PY'
|
|
725
725
|
import json
|
|
726
726
|
import os
|
|
727
727
|
from pathlib import Path
|
|
@@ -730,6 +730,15 @@ import tomllib
|
|
|
730
730
|
|
|
731
731
|
config_path = Path(sys.argv[1])
|
|
732
732
|
delivery = sys.argv[2]
|
|
733
|
+
# "guard" writes the watcher table; "omit" leaves it out for a runtime whose
|
|
734
|
+
# own validator does not recognise it. Both are equally isolated: every
|
|
735
|
+
# invocation sets KIMI_CODE_WATCH, which the runtime reads ahead of the config.
|
|
736
|
+
watch_mode = sys.argv[3]
|
|
737
|
+
# A retry regenerates from the previous generation, so the marker is dropped on
|
|
738
|
+
# the way in and re-added below; the output is the same bytes either way.
|
|
739
|
+
generated_marker = (
|
|
740
|
+
"# Generated by code-review; source-home hooks, tools, and permissions are not inherited."
|
|
741
|
+
)
|
|
733
742
|
source = config_path.read_text(encoding="utf-8") if config_path.exists() else ""
|
|
734
743
|
lines = source.splitlines(keepends=True)
|
|
735
744
|
kept = []
|
|
@@ -793,6 +802,8 @@ def decoded_assignment_root(line):
|
|
|
793
802
|
|
|
794
803
|
for line in lines:
|
|
795
804
|
stripped = line.strip()
|
|
805
|
+
if stripped == generated_marker:
|
|
806
|
+
continue
|
|
796
807
|
table_root = decoded_table_root(stripped)
|
|
797
808
|
if table_root is not None:
|
|
798
809
|
table_allowed = table_root in safe_table_roots
|
|
@@ -828,10 +839,19 @@ guard = (
|
|
|
828
839
|
"[tools]\n"
|
|
829
840
|
f"enabled = {json.dumps(enabled_tools, separators=(',', ':'))}\n"
|
|
830
841
|
)
|
|
842
|
+
if watch_mode == "guard":
|
|
843
|
+
guard += "[watch]\nenabled = false\n"
|
|
831
844
|
rendered = "".join(kept).rstrip() + guard
|
|
832
845
|
parsed = tomllib.loads(rendered)
|
|
833
846
|
allowed_roots = safe_root_keys | safe_table_roots | {"tools"}
|
|
834
|
-
|
|
847
|
+
expected_watch = {"enabled": False} if watch_mode == "guard" else None
|
|
848
|
+
if watch_mode == "guard":
|
|
849
|
+
allowed_roots = allowed_roots | {"watch"}
|
|
850
|
+
if (
|
|
851
|
+
set(parsed) - allowed_roots
|
|
852
|
+
or parsed.get("tools") != {"enabled": enabled_tools}
|
|
853
|
+
or parsed.get("watch") != expected_watch
|
|
854
|
+
):
|
|
835
855
|
raise SystemExit("generated packet-only policy failed semantic verification")
|
|
836
856
|
replacement = config_path.with_name(f".{config_path.name}.code-review.tmp")
|
|
837
857
|
replacement.write_text(rendered, encoding="utf-8")
|
|
@@ -839,14 +859,30 @@ os.chmod(replacement, 0o600)
|
|
|
839
859
|
os.replace(replacement, config_path)
|
|
840
860
|
PY
|
|
841
861
|
}
|
|
842
|
-
install_packet_only_config \
|
|
843
|
-
|| die_inconclusive kimi_packet_only_config_failed capability_missing true
|
|
844
862
|
DOCTOR_STDOUT="$RUN_ROOT/doctor.stdout"
|
|
845
863
|
DOCTOR_STDERR="$RUN_ROOT/doctor.stderr"
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
864
|
+
validate_packet_only_config() {
|
|
865
|
+
doctor_rc=0
|
|
866
|
+
KIMI_CODE_HOME="$RUNTIME_HOME" KIMI_CODE_WATCH=0 KIMI_DISABLE_TELEMETRY=1 \
|
|
867
|
+
timeout --kill-after=1s 15s "$KIMI_BIN_PATH" doctor config "$RUNTIME_HOME/config.toml" \
|
|
868
|
+
>"$DOCTOR_STDOUT" 2>"$DOCTOR_STDERR" \
|
|
869
|
+
|| doctor_rc=$?
|
|
870
|
+
[ "$doctor_rc" -eq 0 ]
|
|
871
|
+
}
|
|
872
|
+
install_packet_only_config guard \
|
|
873
|
+
|| die_inconclusive kimi_packet_only_config_failed capability_missing true
|
|
874
|
+
# The watcher table is belt on top of KIMI_CODE_WATCH, which every invocation
|
|
875
|
+
# below sets and which the runtime reads ahead of the config. A runtime whose
|
|
876
|
+
# validator does not know the table must not cost the whole lane, so retry once
|
|
877
|
+
# with a strict subset of the same policy — admission cannot widen, and the
|
|
878
|
+
# retry is unconditional rather than matched against the runtime's error
|
|
879
|
+
# wording, which would be the same release-vocabulary pin in another place.
|
|
880
|
+
if ! validate_packet_only_config; then
|
|
881
|
+
install_packet_only_config omit \
|
|
882
|
+
|| die_inconclusive kimi_packet_only_config_failed capability_missing true
|
|
883
|
+
validate_packet_only_config \
|
|
884
|
+
|| die_inconclusive kimi_packet_only_config_unrecognized capability_missing true "$doctor_rc"
|
|
885
|
+
fi
|
|
850
886
|
|
|
851
887
|
# Exercise the forbidden built-in tool surface without depending on a version
|
|
852
888
|
# string. In MCP mode the generated config still exposes the one packet reader,
|
|
@@ -866,7 +902,7 @@ PROBE_TIMEOUT="$TIMEOUT"
|
|
|
866
902
|
[ "$PROBE_TIMEOUT" -le 60 ] || PROBE_TIMEOUT=60
|
|
867
903
|
(
|
|
868
904
|
cd "$RUN_WORKSPACE" || exit 2
|
|
869
|
-
KIMI_CODE_HOME="$RUNTIME_HOME" KIMI_DISABLE_TELEMETRY=1 \
|
|
905
|
+
KIMI_CODE_HOME="$RUNTIME_HOME" KIMI_CODE_WATCH=0 KIMI_DISABLE_TELEMETRY=1 \
|
|
870
906
|
timeout --kill-after=1s "${PROBE_TIMEOUT}s" "$KIMI_BIN_PATH" \
|
|
871
907
|
--skills-dir "$ACTIVE_SKILLS_DIR" --prompt "$PROBE_PROMPT" \
|
|
872
908
|
--output-format stream-json
|
|
@@ -876,6 +912,24 @@ if [ "$probe_rc" -ne 0 ]; then
|
|
|
876
912
|
if grep -qiE 'EMFILE|too many open files' "$PROBE_STDERR"; then
|
|
877
913
|
die_inconclusive kimi_host_resource_exhausted client_unavailable true "$probe_rc"
|
|
878
914
|
fi
|
|
915
|
+
# NO SUB-CLASSIFICATION OF PROVIDER PROSE. Splitting this failure into quota
|
|
916
|
+
# and auth reasons was tried and withdrawn: four independent review rounds
|
|
917
|
+
# each broke the predicate on a new message. A digit boundary excluded 4290
|
|
918
|
+
# but not an offset of exactly 429; requiring a status word before the number
|
|
919
|
+
# matched `code` inside `decode`; matching what the message said instead
|
|
920
|
+
# matched `hit` inside `whitelisted` and still missed `credits are exhausted`.
|
|
921
|
+
# That is the shape this directory's contract already names — a predicate over
|
|
922
|
+
# a vocabulary the control does not own — and same-class recurrence is the cue
|
|
923
|
+
# to remove the capability rather than guard it again.
|
|
924
|
+
#
|
|
925
|
+
# Nothing about the gate changes: every one of those reasons was
|
|
926
|
+
# die_inconclusive and cascade-eligible, so the split only ever decided an
|
|
927
|
+
# operator hint, and a wrong hint is worse than none. The contract also
|
|
928
|
+
# forbids putting raw provider text in the payload, so the honest replacement
|
|
929
|
+
# is not a looser regex but a classifier over the structured error the probe
|
|
930
|
+
# already streams, the way the Claude lane's envelope classifier works. That
|
|
931
|
+
# is its own change, against a real sample.
|
|
932
|
+
:
|
|
879
933
|
die_inconclusive kimi_tool_capability_unverified capability_missing true "$probe_rc"
|
|
880
934
|
fi
|
|
881
935
|
# MCP mode enables the packet reader for the later formal transport, but this
|
|
@@ -962,14 +1016,14 @@ run_started=$SECONDS
|
|
|
962
1016
|
if [ "$PACKET_DELIVERY" = inline ]; then
|
|
963
1017
|
(
|
|
964
1018
|
cd "$RUN_WORKSPACE" || exit 2
|
|
965
|
-
KIMI_CODE_HOME="$RUNTIME_HOME" KIMI_DISABLE_TELEMETRY=1 \
|
|
1019
|
+
KIMI_CODE_HOME="$RUNTIME_HOME" KIMI_CODE_WATCH=0 KIMI_DISABLE_TELEMETRY=1 \
|
|
966
1020
|
timeout --kill-after=1s "${FORMAL_TIMEOUT}s" "$KIMI_BIN_PATH" --skills-dir "$ACTIVE_SKILLS_DIR" \
|
|
967
1021
|
--prompt "$PROMPT" --output-format stream-json
|
|
968
1022
|
) >"$EVENTS" 2>"$STDERR_FILE"
|
|
969
1023
|
else
|
|
970
1024
|
(
|
|
971
1025
|
cd "$RUN_WORKSPACE" || exit 2
|
|
972
|
-
KIMI_CODE_HOME="$RUNTIME_HOME" KIMI_DISABLE_TELEMETRY=1 \
|
|
1026
|
+
KIMI_CODE_HOME="$RUNTIME_HOME" KIMI_CODE_WATCH=0 KIMI_DISABLE_TELEMETRY=1 \
|
|
973
1027
|
timeout --kill-after=1s "${FORMAL_TIMEOUT}s" "$KIMI_BIN_PATH" --skills-dir "$ACTIVE_SKILLS_DIR" \
|
|
974
1028
|
--agent-file "$AGENT_FILE" --prompt "$PROMPT" --output-format stream-json
|
|
975
1029
|
) >"$EVENTS" 2>"$STDERR_FILE"
|
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py
CHANGED
|
@@ -4661,8 +4661,10 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
4661
4661
|
)
|
|
4662
4662
|
# Only a candidate derived from the whole worktree says what HEAD holds:
|
|
4663
4663
|
# a bare --diff-file packet or a --paths slice may cover less than HEAD.
|
|
4664
|
+
# A completion checkpoint records too: it is what turns a findings
|
|
4665
|
+
# receipt into a passed one once every finding is disposed.
|
|
4664
4666
|
anchors_wanted = (
|
|
4665
|
-
args.mode in {"review", "challenge"}
|
|
4667
|
+
args.mode in {"review", "challenge", "complete"}
|
|
4666
4668
|
and bool(args.base)
|
|
4667
4669
|
and not args.paths
|
|
4668
4670
|
)
|
|
@@ -4831,6 +4833,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
4831
4833
|
),
|
|
4832
4834
|
)
|
|
4833
4835
|
result.update(completion_metadata)
|
|
4836
|
+
record_local_review(review_anchor, result)
|
|
4834
4837
|
return emit(result, 0)
|
|
4835
4838
|
last_reason_code = "no_independent_reviewer_available"
|
|
4836
4839
|
for client in order:
|
|
@@ -64,6 +64,9 @@ base_url = "https://example.invalid"
|
|
|
64
64
|
[future_tools]
|
|
65
65
|
enabled = true
|
|
66
66
|
|
|
67
|
+
[watch]
|
|
68
|
+
enabled = true
|
|
69
|
+
|
|
67
70
|
[permission]
|
|
68
71
|
deny = ["Read"]
|
|
69
72
|
|
|
@@ -164,6 +167,7 @@ cat >"$WORK/bin/kimi" <<'KIMI_STUB'
|
|
|
164
167
|
#!/usr/bin/env bash
|
|
165
168
|
set -u
|
|
166
169
|
state="$REVIEW_WRAPPER_TEST_STATE"
|
|
170
|
+
printf '%s\t%s\n' invocation "${KIMI_CODE_WATCH-}" >>"$state/kimi_watch_all"
|
|
167
171
|
if [ "${1:-}" = --version ]; then
|
|
168
172
|
if [ "${STUB_BEHAVIOR:-}" = version_hang ]; then
|
|
169
173
|
trap '' TERM
|
|
@@ -176,6 +180,23 @@ fi
|
|
|
176
180
|
if [ "${1:-}" = doctor ]; then
|
|
177
181
|
touch "$state/kimi_doctor_checked"
|
|
178
182
|
printf '%s\n' "$*" >"$state/kimi_doctor_args"
|
|
183
|
+
printf '%s\t%s\n' doctor "${KIMI_CODE_WATCH-}" >>"$state/kimi_watch"
|
|
184
|
+
# Keep every validated config so a retry can be compared against the attempt
|
|
185
|
+
# it replaced, rather than only asserted about one property at a time.
|
|
186
|
+
doctor_n=$(( $(cat "$state/kimi_doctor_count" 2>/dev/null || echo 0) + 1 ))
|
|
187
|
+
printf '%s' "$doctor_n" >"$state/kimi_doctor_count"
|
|
188
|
+
if [ -n "${3:-}" ] && [ -f "$3" ]; then
|
|
189
|
+
cp "$3" "$state/kimi_doctor_config.$doctor_n"
|
|
190
|
+
fi
|
|
191
|
+
if [ "${STUB_BEHAVIOR:-pass}" = doctor_reject_all ]; then
|
|
192
|
+
printf '%s\n' 'config rejected' >&2
|
|
193
|
+
exit 1
|
|
194
|
+
fi
|
|
195
|
+
if [ "${STUB_BEHAVIOR:-pass}" = doctor_reject_watch ] \
|
|
196
|
+
&& [ -n "${3:-}" ] && [ -f "$3" ] && grep -q '^\[watch\]' "$3"; then
|
|
197
|
+
printf '%s\n' 'unknown config table: watch' >&2
|
|
198
|
+
exit 1
|
|
199
|
+
fi
|
|
179
200
|
exit 0
|
|
180
201
|
fi
|
|
181
202
|
touch "$state/kimi_invoked"
|
|
@@ -199,6 +220,12 @@ while [ "$#" -gt 0 ]; do
|
|
|
199
220
|
*) shift ;;
|
|
200
221
|
esac
|
|
201
222
|
done
|
|
223
|
+
if [[ "$prompt" = "No-tools capability probe."* ]]; then
|
|
224
|
+
watch_phase=capability
|
|
225
|
+
else
|
|
226
|
+
watch_phase=formal
|
|
227
|
+
fi
|
|
228
|
+
printf '%s\t%s\n' "$watch_phase" "${KIMI_CODE_WATCH-}" >>"$state/kimi_watch"
|
|
202
229
|
printf '%s' "$has_stream" >"$state/kimi_stream"
|
|
203
230
|
printf '%s' "$has_model" >"$state/kimi_model_override"
|
|
204
231
|
printf '%s' "$has_skills_dir" >"$state/kimi_skills_dir"
|
|
@@ -213,7 +240,9 @@ if [ -f "$KIMI_CODE_HOME/config.toml" ]; then
|
|
|
213
240
|
grep -q '/trusted/local/hook-marker' "$KIMI_CODE_HOME/config.toml" && touch "$state/kimi_hooks_preserved"
|
|
214
241
|
grep -q 'enabled = \["\*"\]' "$KIMI_CODE_HOME/config.toml" \
|
|
215
242
|
&& touch "$state/kimi_no_tools_configured"
|
|
216
|
-
|
|
243
|
+
grep -c '^# Generated by code-review' "$KIMI_CODE_HOME/config.toml" \
|
|
244
|
+
>"$state/kimi_generated_marker_count"
|
|
245
|
+
python3 - "$KIMI_CODE_HOME/config.toml" "$state/kimi_runtime_config_mode" "$state/kimi_dotted_config_preserved" "$state/kimi_watch_disabled" "$state/kimi_watch_absent" <<'PY'
|
|
217
246
|
import os, sys, tomllib
|
|
218
247
|
from pathlib import Path
|
|
219
248
|
|
|
@@ -222,6 +251,10 @@ with open(sys.argv[2], "w", encoding="utf-8") as stream:
|
|
|
222
251
|
config = tomllib.loads(Path(sys.argv[1]).read_text(encoding="utf-8"))
|
|
223
252
|
if config.get("providers") and config.get("models"):
|
|
224
253
|
Path(sys.argv[3]).touch()
|
|
254
|
+
if config.get("watch") == {"enabled": False}:
|
|
255
|
+
Path(sys.argv[4]).touch()
|
|
256
|
+
if "watch" not in config:
|
|
257
|
+
Path(sys.argv[5]).touch()
|
|
225
258
|
PY
|
|
226
259
|
fi
|
|
227
260
|
if [ -f "$KIMI_CODE_HOME/mcp.json" ]; then
|
|
@@ -258,6 +291,58 @@ if [[ "$prompt" = "No-tools capability probe."* ]]; then
|
|
|
258
291
|
printf '%s\n' 'EMFILE: too many open files, watch' >&2
|
|
259
292
|
exit 1
|
|
260
293
|
fi
|
|
294
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_quota ]; then
|
|
295
|
+
printf '%s\n' "provider.auth_error: 403 You've reached your weekly (7-day) usage limit" >&2
|
|
296
|
+
exit 1
|
|
297
|
+
fi
|
|
298
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_auth ]; then
|
|
299
|
+
printf '%s\n' 'provider.auth_error: 403 authentication required' >&2
|
|
300
|
+
exit 1
|
|
301
|
+
fi
|
|
302
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_auth_quota_mention ]; then
|
|
303
|
+
printf '%s\n' 'provider.auth_error: 403 authentication required; quota metadata unavailable' >&2
|
|
304
|
+
exit 1
|
|
305
|
+
fi
|
|
306
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_auth_limit ]; then
|
|
307
|
+
printf '%s\n' 'provider.auth_error: 403 authentication required; usage limit metadata unavailable' >&2
|
|
308
|
+
exit 1
|
|
309
|
+
fi
|
|
310
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_auth_word ]; then
|
|
311
|
+
printf '%s\n' 'tool surface rejected; authentication metadata unavailable' >&2
|
|
312
|
+
exit 1
|
|
313
|
+
fi
|
|
314
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_offset ]; then
|
|
315
|
+
printf '%s\n' 'provider failed at byte offset 4290 and line 14030' >&2
|
|
316
|
+
exit 1
|
|
317
|
+
fi
|
|
318
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_exact_offset ]; then
|
|
319
|
+
printf '%s\n' 'provider failed at byte offset 429 and line 403' >&2
|
|
320
|
+
exit 1
|
|
321
|
+
fi
|
|
322
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_rate_limit_403 ]; then
|
|
323
|
+
printf '%s\n' 'provider.auth_error: 403 rate limit exceeded' >&2
|
|
324
|
+
exit 1
|
|
325
|
+
fi
|
|
326
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_http_quota ]; then
|
|
327
|
+
printf '%s\n' 'provider error: HTTP 429 Too Many Requests' >&2
|
|
328
|
+
exit 1
|
|
329
|
+
fi
|
|
330
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_decode_offset ]; then
|
|
331
|
+
printf '%s\n' 'provider failed to decode offset 429' >&2
|
|
332
|
+
exit 1
|
|
333
|
+
fi
|
|
334
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_auth_weekly_metadata ]; then
|
|
335
|
+
printf '%s\n' 'provider.auth_error: 403 authentication required; weekly usage limit metadata unavailable' >&2
|
|
336
|
+
exit 1
|
|
337
|
+
fi
|
|
338
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_auth_too_many ]; then
|
|
339
|
+
printf '%s\n' 'provider.auth_error: HTTP 429 Too Many Requests' >&2
|
|
340
|
+
exit 1
|
|
341
|
+
fi
|
|
342
|
+
if [ "${STUB_BEHAVIOR:-pass}" = capability_forbidden ]; then
|
|
343
|
+
printf '%s\n' 'provider error: 403 Forbidden' >&2
|
|
344
|
+
exit 1
|
|
345
|
+
fi
|
|
261
346
|
if [ "${STUB_BEHAVIOR:-pass}" = capability_missing ]; then
|
|
262
347
|
printf '%s\n' '{"role":"meta","type":"system.version","version":"future"}'
|
|
263
348
|
exit 0
|
|
@@ -546,6 +631,7 @@ content = {
|
|
|
546
631
|
"foreign_tool": clean,
|
|
547
632
|
"nested_tool": clean,
|
|
548
633
|
"unknown_event": clean,
|
|
634
|
+
"doctor_reject_watch": clean,
|
|
549
635
|
"multi_message": clean,
|
|
550
636
|
"mcp_bad_chunk": clean,
|
|
551
637
|
"mcp_eof_confirmation": clean,
|
|
@@ -1321,15 +1407,37 @@ for client in kimi codex; do
|
|
|
1321
1407
|
done
|
|
1322
1408
|
done
|
|
1323
1409
|
|
|
1324
|
-
|
|
1410
|
+
rm -f "$WORK/state/kimi_watch" "$WORK/state/kimi_watch_all" "$WORK/state/kimi_watch_disabled"
|
|
1411
|
+
out="$(KIMI_CODE_WATCH=1 run_kimi pass)"; rc=$?
|
|
1325
1412
|
check "Kimi clean result passes with fixed Moonshot attribution" \
|
|
1326
|
-
'[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ "$(field concern_results.0.concern "$out")" = correctness ] && [ "$(field reviewer_family "$out")" = moonshot ] && [ "$(field provider "$out")" = kimi-cli ] && [ "$(field model "$out")" = None ] && [ "$(dir_mode "$WORK/kimi-source/config.toml")" = 400 ] && [ "$(cat "$WORK/state/kimi_runtime_config_mode")" = 600 ]'
|
|
1413
|
+
'[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ "$(field concern_results.0.concern "$out")" = correctness ] && [ "$(field reviewer_family "$out")" = moonshot ] && [ "$(field provider "$out")" = kimi-cli ] && [ "$(field model "$out")" = None ] && [ "$(dir_mode "$WORK/kimi-source/config.toml")" = 400 ] && [ "$(cat "$WORK/state/kimi_runtime_config_mode")" = 600 ] && [ "$(wc -l < "$WORK/state/kimi_watch" | tr -d " ")" = 3 ] && [ "$(grep -c "^doctor[[:space:]]0$" "$WORK/state/kimi_watch")" = 1 ] && [ "$(grep -c "^capability[[:space:]]0$" "$WORK/state/kimi_watch")" = 1 ] && [ "$(grep -c "^formal[[:space:]]0$" "$WORK/state/kimi_watch")" = 1 ] && [ "$(wc -l < "$WORK/state/kimi_watch_all" | tr -d " ")" = 3 ] && ! grep -qEv "^invocation[[:space:]]0$" "$WORK/state/kimi_watch_all" && [ -e "$WORK/state/kimi_watch_disabled" ]'
|
|
1414
|
+
# The retry's config must differ from the attempt it replaced by exactly the
|
|
1415
|
+
# watcher table: asserting only that watch is absent would also pass a retry
|
|
1416
|
+
# that dropped or widened something else.
|
|
1417
|
+
retry_config_is_watch_only_delta() {
|
|
1418
|
+
python3 -c '
|
|
1419
|
+
import sys
|
|
1420
|
+
first = open(sys.argv[1], encoding="utf-8").read()
|
|
1421
|
+
retry = open(sys.argv[2], encoding="utf-8").read()
|
|
1422
|
+
sys.exit(0 if first.replace("[watch]\nenabled = false\n", "", 1) == retry else 1)
|
|
1423
|
+
' "$1" "$2"
|
|
1424
|
+
}
|
|
1425
|
+
rm -f "$WORK/state/kimi_watch" "$WORK/state/kimi_watch_all" "$WORK/state/kimi_watch_disabled" "$WORK/state/kimi_watch_absent" "$WORK/state/kimi_doctor_count" "$WORK/state/kimi_doctor_config."*
|
|
1426
|
+
out="$(KIMI_CODE_WATCH=1 run_kimi doctor_reject_watch)"; rc=$?
|
|
1427
|
+
# The watcher guard is belt on top of the per-invocation override, which the
|
|
1428
|
+
# runtime reads ahead of the config. A runtime that does not know the table may
|
|
1429
|
+
# not cost the whole lane: drop the table, keep the override, stay admitted.
|
|
1430
|
+
check "Kimi keeps the lane when the runtime rejects only the watcher table" \
|
|
1431
|
+
'[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ -e "$WORK/state/kimi_watch_absent" ] && [ ! -e "$WORK/state/kimi_watch_disabled" ] && [ "$(tr -d " " < "$WORK/state/kimi_generated_marker_count")" = 1 ] && retry_config_is_watch_only_delta "$WORK/state/kimi_doctor_config.1" "$WORK/state/kimi_doctor_config.2" && [ "$(grep -c "^doctor[[:space:]]0$" "$WORK/state/kimi_watch")" = 2 ] && [ "$(grep -c "^capability[[:space:]]0$" "$WORK/state/kimi_watch")" = 1 ] && [ "$(grep -c "^formal[[:space:]]0$" "$WORK/state/kimi_watch")" = 1 ] && ! grep -qEv "^invocation[[:space:]]0$" "$WORK/state/kimi_watch_all"'
|
|
1432
|
+
out="$(run_kimi doctor_reject_all)"; rc=$?
|
|
1433
|
+
check "Kimi still fails closed when no generated config is accepted" \
|
|
1434
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_packet_only_config_unrecognized ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1327
1435
|
# Admission is version-neutral: deliberately unparseable version output must not
|
|
1328
1436
|
# trigger --version or block a capable runtime.
|
|
1329
|
-
rm -f "$WORK/state/kimi_invoked" "$WORK/state/kimi_version_checked" "$WORK/state/kimi_no_tools_configured" "$WORK/state/kimi_no_tools_policy_verified"
|
|
1437
|
+
rm -f "$WORK/state/kimi_invoked" "$WORK/state/kimi_version_checked" "$WORK/state/kimi_no_tools_configured" "$WORK/state/kimi_no_tools_policy_verified" "$WORK/state/kimi_watch_all"
|
|
1330
1438
|
out="$(KIMI_STUB_VERSION='not-a-version' run_kimi pass)"; rc=$?
|
|
1331
1439
|
check "Kimi admission depends on runtime capability, not a version string" \
|
|
1332
|
-
'[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ ! -e "$WORK/state/kimi_version_checked" ] && [ -e "$WORK/state/kimi_doctor_checked" ] && grep -q "^doctor config /.*config.toml$" "$WORK/state/kimi_doctor_args" && [ -e "$WORK/state/kimi_invoked" ] && [ -e "$WORK/state/kimi_no_tools_configured" ] && [ -e "$WORK/state/kimi_no_tools_policy_verified" ]'
|
|
1440
|
+
'[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ ! -e "$WORK/state/kimi_version_checked" ] && [ -e "$WORK/state/kimi_doctor_checked" ] && grep -q "^doctor config /.*config.toml$" "$WORK/state/kimi_doctor_args" && [ -e "$WORK/state/kimi_invoked" ] && [ -e "$WORK/state/kimi_no_tools_configured" ] && [ -e "$WORK/state/kimi_no_tools_policy_verified" ] && [ "$(wc -l < "$WORK/state/kimi_watch_all" | tr -d " ")" = 3 ] && ! grep -qEv "^invocation[[:space:]]0$" "$WORK/state/kimi_watch_all"'
|
|
1333
1441
|
rm -f "$WORK/state/kimi_invoked"
|
|
1334
1442
|
out="$(run_kimi pass claude "$WORK/kimi-source" "$WORK/nul.patch")"; rc=$?
|
|
1335
1443
|
check "Kimi rejects a NUL-bearing diff before inference" \
|
|
@@ -1346,6 +1454,47 @@ check "Kimi rejects any tool exposed during the no-tools probe" \
|
|
|
1346
1454
|
out="$(run_kimi capability_emfile)"; rc=$?
|
|
1347
1455
|
check "Kimi classifies probe-time EMFILE as a local client failure" \
|
|
1348
1456
|
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_host_resource_exhausted ] && [ "$(field reason_code "$out")" = client_unavailable ] && [ "$(field cascade_eligible "$out")" = True ]'
|
|
1457
|
+
# Provider prose is not sub-classified: these shapes each broke a predecessor
|
|
1458
|
+
# predicate, and all of them now land in the one capability class.
|
|
1459
|
+
out="$(run_kimi capability_quota)"; rc=$?
|
|
1460
|
+
check "Kimi does not sub-classify provider prose: quota" \
|
|
1461
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1462
|
+
out="$(run_kimi capability_auth)"; rc=$?
|
|
1463
|
+
check "Kimi does not sub-classify provider prose: auth" \
|
|
1464
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1465
|
+
out="$(run_kimi capability_auth_quota_mention)"; rc=$?
|
|
1466
|
+
check "Kimi does not sub-classify provider prose: auth quota mention" \
|
|
1467
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1468
|
+
out="$(run_kimi capability_auth_limit)"; rc=$?
|
|
1469
|
+
check "Kimi does not sub-classify provider prose: auth limit" \
|
|
1470
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1471
|
+
out="$(run_kimi capability_auth_word)"; rc=$?
|
|
1472
|
+
check "Kimi does not sub-classify provider prose: auth word" \
|
|
1473
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1474
|
+
out="$(run_kimi capability_offset)"; rc=$?
|
|
1475
|
+
check "Kimi does not sub-classify provider prose: offset" \
|
|
1476
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1477
|
+
out="$(run_kimi capability_exact_offset)"; rc=$?
|
|
1478
|
+
check "Kimi does not sub-classify provider prose: exact offset" \
|
|
1479
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1480
|
+
out="$(run_kimi capability_rate_limit_403)"; rc=$?
|
|
1481
|
+
check "Kimi does not sub-classify provider prose: rate limit 403" \
|
|
1482
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1483
|
+
out="$(run_kimi capability_http_quota)"; rc=$?
|
|
1484
|
+
check "Kimi does not sub-classify provider prose: http quota" \
|
|
1485
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1486
|
+
out="$(run_kimi capability_forbidden)"; rc=$?
|
|
1487
|
+
check "Kimi does not sub-classify provider prose: forbidden" \
|
|
1488
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1489
|
+
out="$(run_kimi capability_decode_offset)"; rc=$?
|
|
1490
|
+
check "Kimi does not sub-classify provider prose: decode offset" \
|
|
1491
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1492
|
+
out="$(run_kimi capability_auth_weekly_metadata)"; rc=$?
|
|
1493
|
+
check "Kimi does not sub-classify provider prose: auth weekly metadata" \
|
|
1494
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1495
|
+
out="$(run_kimi capability_auth_too_many)"; rc=$?
|
|
1496
|
+
check "Kimi does not sub-classify provider prose: auth too many" \
|
|
1497
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = kimi_tool_capability_unverified ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ "$(field transport_exit_code "$out")" = 1 ]'
|
|
1349
1498
|
probe_started=$SECONDS
|
|
1350
1499
|
out="$(REVIEW_TEST_TIMEOUT=5 run_kimi capability_hang)"; rc=$?
|
|
1351
1500
|
probe_elapsed=$((SECONDS - probe_started))
|
|
@@ -1406,10 +1555,13 @@ out="$(REVIEW_TEST_TIMEOUT=180 REVIEW_TEST_PATH_PREFIX="$WORK/timeout-probe-bin"
|
|
|
1406
1555
|
check "Kimi caps argv-exposed inline review at 120 seconds" \
|
|
1407
1556
|
'[ "$rc" = 0 ] && grep -q -- "--kill-after=1s 120s " "$WORK/state/timeout_args"'
|
|
1408
1557
|
rm -f "$WORK/state/timeout_args"
|
|
1409
|
-
|
|
1558
|
+
rm -f "$WORK/state/kimi_watch" "$WORK/state/kimi_watch_all" "$WORK/state/kimi_watch_disabled"
|
|
1559
|
+
out="$(KIMI_CODE_WATCH=1 REVIEW_TEST_TIMEOUT=180 REVIEW_TEST_PATH_PREFIX="$WORK/timeout-probe-bin" run_kimi pass claude "$WORK/kimi-source" "$WORK/agent-template.patch")"; rc=$?
|
|
1410
1560
|
formal_timeout="$(tail -n 1 "$WORK/state/timeout_args" | awk '{ value=$2; sub(/s$/, "", value); print value }')"
|
|
1411
1561
|
check "Kimi gives private MCP review only the remaining controller budget" \
|
|
1412
1562
|
'[ "$rc" = 0 ] && [ "$formal_timeout" -ge 1 ] && [ "$formal_timeout" -le 180 ]'
|
|
1563
|
+
check "Kimi sets the watcher override on private MCP formal delivery" \
|
|
1564
|
+
'[ "$(wc -l < "$WORK/state/kimi_watch" | tr -d " ")" = 3 ] && [ "$(grep -c "^doctor[[:space:]]0$" "$WORK/state/kimi_watch")" = 1 ] && [ "$(grep -c "^capability[[:space:]]0$" "$WORK/state/kimi_watch")" = 1 ] && [ "$(grep -c "^formal[[:space:]]0$" "$WORK/state/kimi_watch")" = 1 ] && [ "$(wc -l < "$WORK/state/kimi_watch_all" | tr -d " ")" = 3 ] && ! grep -qEv "^invocation[[:space:]]0$" "$WORK/state/kimi_watch_all" && [ -e "$WORK/state/kimi_watch_disabled" ]'
|
|
1413
1565
|
rm -f "$WORK/state/timeout_args"
|
|
1414
1566
|
out="$(REVIEW_TEST_TIMEOUT=10 REVIEW_TEST_PATH_PREFIX="$WORK/timeout-probe-bin" run_kimi capability_delay claude "$WORK/kimi-source" "$WORK/agent-template.patch")"; rc=$?
|
|
1415
1567
|
formal_timeout="$(tail -n 1 "$WORK/state/timeout_args" | awk '{ value=$2; sub(/s$/, "", value); print value }')"
|
|
@@ -832,11 +832,49 @@ class CompletionFindingDispositionTest(unittest.TestCase):
|
|
|
832
832
|
"failure_path": "A distinct synthetic failure on the same line."})
|
|
833
833
|
return subprocess.CompletedProcess(command, 0, json.dumps(payload).encode("utf-8"), b"")
|
|
834
834
|
|
|
835
|
+
base_repo: Path | None = None
|
|
836
|
+
|
|
835
837
|
def arguments(self, mode: str) -> list[str]:
|
|
836
|
-
|
|
838
|
+
if self.base_repo is not None:
|
|
839
|
+
candidate = ["--cwd", str(self.base_repo), "--base", self.base_commit]
|
|
840
|
+
else:
|
|
841
|
+
candidate = ["--cwd", str(self.root), "--diff-file", str(self.packet)]
|
|
842
|
+
return ["--mode", mode, *candidate,
|
|
837
843
|
"--implementer-family", "openai", "--review-plan-file", str(self.plan),
|
|
838
844
|
"--challenge-budget", "1"]
|
|
839
845
|
|
|
846
|
+
def test_source_refuted_completion_records_a_passed_receipt(self) -> None:
|
|
847
|
+
# The chain the pull-request reminder reads: findings recorded on a
|
|
848
|
+
# committed whole-worktree candidate, then disposed through completion.
|
|
849
|
+
repo = self.root / "repo"
|
|
850
|
+
repo.mkdir()
|
|
851
|
+
def git(*command: str) -> str:
|
|
852
|
+
return subprocess.run(["git", "-C", str(repo), "-c", "user.email=t@example.invalid",
|
|
853
|
+
"-c", "user.name=t", *command],
|
|
854
|
+
capture_output=True, check=True, text=True).stdout.strip()
|
|
855
|
+
git("init", "-q")
|
|
856
|
+
(repo / "x").write_text("a\n", encoding="utf-8")
|
|
857
|
+
git("add", "x")
|
|
858
|
+
git("commit", "-q", "-m", "base")
|
|
859
|
+
self.base_commit = git("rev-parse", "HEAD")
|
|
860
|
+
(repo / "x").write_text("b\n", encoding="utf-8")
|
|
861
|
+
git("commit", "-q", "-am", "candidate")
|
|
862
|
+
head = git("rev-parse", "HEAD")
|
|
863
|
+
self.base_repo = repo
|
|
864
|
+
self.record_receipts()
|
|
865
|
+
receipt = repo / ".git" / "ccl-code-review" / "last-review.json"
|
|
866
|
+
before = json.loads(receipt.read_text(encoding="utf-8"))
|
|
867
|
+
self.assertEqual((before["mode"], before["status"], before["head"], before["worktree_clean"]),
|
|
868
|
+
("challenge", "findings", head, True))
|
|
869
|
+
|
|
870
|
+
code, result = self.complete()
|
|
871
|
+
|
|
872
|
+
self.assertEqual(code, 0, result)
|
|
873
|
+
self.assertEqual(result["status"], "passed", result)
|
|
874
|
+
after = json.loads(receipt.read_text(encoding="utf-8"))
|
|
875
|
+
self.assertEqual((after["mode"], after["status"], after["head"], after["worktree_clean"]),
|
|
876
|
+
("complete", "passed", head, True))
|
|
877
|
+
|
|
840
878
|
def invoke(self, mode: str, extra: list[str]) -> tuple[int, dict]:
|
|
841
879
|
output = io.StringIO()
|
|
842
880
|
with (mock.patch.object(review_gate, "run", side_effect=self.wrapper_result),
|
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh
CHANGED
|
@@ -4863,6 +4863,24 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
|
|
|
4863
4863
|
check "a bare --diff-file review records no receipt" \
|
|
4864
4864
|
'[ "$rc" = 0 ] && json_fields "$out" status=passed && [ ! -e "$contract_repo/.git/ccl-code-review" ]'
|
|
4865
4865
|
|
|
4866
|
+
# The completion checkpoint is what the pull-request reminder reads as disposed:
|
|
4867
|
+
# on the same whole-worktree candidate it replaces the review's receipt with a
|
|
4868
|
+
# passed one; a checkpoint that fails leaves the receipt as it was.
|
|
4869
|
+
reset_case passed unavailable unavailable
|
|
4870
|
+
out="$(run_contract_gate --mode review)"; rc=$?
|
|
4871
|
+
printf '%s\n' "$out" >"$WORK/contract-completion-review.json"
|
|
4872
|
+
check "the review before a completion checkpoint records its own receipt" \
|
|
4873
|
+
'[ "$rc" = 0 ] && [ "$(jq -r .mode "$receipt_file")" = review ]'
|
|
4874
|
+
reset_case passed unavailable unavailable
|
|
4875
|
+
out="$(run_contract_gate --mode complete --completion-review-result-file "$WORK/contract-completion-review.json")"; rc=$?
|
|
4876
|
+
check "a successful completion checkpoint records a passed receipt for the HEAD it bound" \
|
|
4877
|
+
'[ "$rc" = 0 ] && json_fields "$out" mode=complete status=passed && [ "$(jq -r .mode "$receipt_file")" = complete ] && [ "$(jq -r .status "$receipt_file")" = passed ] && [ "$(jq -r .head "$receipt_file")" = "$(git -C "$contract_repo" rev-parse HEAD)" ]'
|
|
4878
|
+
cp "$receipt_file" "$WORK/receipt-before-failed-completion"
|
|
4879
|
+
reset_case passed unavailable unavailable
|
|
4880
|
+
out="$(run_contract_gate --mode complete --review-plan-file "$WORK/changed-review-plan.json" --completion-review-result-file "$WORK/contract-completion-review.json")"; rc=$?
|
|
4881
|
+
check "a failed completion checkpoint leaves the receipt untouched" \
|
|
4882
|
+
'[ "$rc" = 2 ] && cmp -s "$receipt_file" "$WORK/receipt-before-failed-completion"'
|
|
4883
|
+
|
|
4866
4884
|
swap_out="$(PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY' 2>&1
|
|
4867
4885
|
import os, sys, review_gate
|
|
4868
4886
|
root = os.path.realpath(os.path.join(sys.argv[1], "gitdir-swap"))
|