@ccoalm/ccl-skills 0.15.4 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +15 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +70 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +286 -40
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +177 -89
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +396 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +236 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +3 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +20 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +56 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +68 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +14 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/scenario-testing.md +1 -1
- package/dist/assets/release.json +17 -17
- package/package.json +1 -1
|
@@ -134,17 +134,16 @@ echo "code_review_skill_dir=$CODE_REVIEW_SKILL_DIR" >&2
|
|
|
134
134
|
: "${REVIEW_CHAIN_ID:?set REVIEW_CHAIN_ID to a task-scoped chain id: letters, digits, dot, underscore, hyphen only}"
|
|
135
135
|
: "${REVIEW_STAGE:?set REVIEW_STAGE to the stage this candidate is actually at: explore, build, or release}"
|
|
136
136
|
: "${REVIEW_EVIDENCE_DIR:?set REVIEW_EVIDENCE_DIR to a durable directory you control for the per-round result rows}"
|
|
137
|
-
#
|
|
138
|
-
#
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
elif [ -n "${REVIEW_BASE:-}" ]; then
|
|
144
|
-
PACKET_ARGS=(--base "$REVIEW_BASE")
|
|
145
|
-
else
|
|
146
|
-
echo "set exactly one of REVIEW_DIFF_FILE or REVIEW_BASE" >&2; exit 1
|
|
137
|
+
# REVIEW_BASE names the candidate, REVIEW_DIFF_FILE widens what the reviewer reads,
|
|
138
|
+
# and BOTH is a widened packet that must BEGIN with the candidate (composition
|
|
139
|
+
# rules: references/staged-review-contract.md). if-blocks, not `[ -n ... ] && ...`: a
|
|
140
|
+
# trailing false test returns non-zero and `set -e` would kill the caller.
|
|
141
|
+
if [ -z "${REVIEW_DIFF_FILE:-}" ] && [ -z "${REVIEW_BASE:-}" ]; then
|
|
142
|
+
echo "set REVIEW_BASE, REVIEW_DIFF_FILE, or both" >&2; exit 1
|
|
147
143
|
fi
|
|
144
|
+
PACKET_ARGS=()
|
|
145
|
+
if [ -n "${REVIEW_BASE:-}" ]; then PACKET_ARGS+=(--base "$REVIEW_BASE"); fi
|
|
146
|
+
if [ -n "${REVIEW_DIFF_FILE:-}" ]; then PACKET_ARGS+=(--diff-file "$REVIEW_DIFF_FILE"); fi
|
|
148
147
|
# REVIEW_RUN_DIR holds the raw round-1 result only for the chain handoff; the durable
|
|
149
148
|
# per-round evidence is persisted to REVIEW_EVIDENCE_DIR, whose confidentiality you own.
|
|
150
149
|
REVIEW_RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/review-run.XXXXXX")" || exit 1
|
|
@@ -199,10 +198,11 @@ bash "$CODE_REVIEW_SKILL_DIR/scripts/review_gate.sh" \
|
|
|
199
198
|
>"$REVIEW_RUN_DIR/round2.json"
|
|
200
199
|
require_tracked_result "$REVIEW_RUN_DIR/round2.json" challenge 2
|
|
201
200
|
# Both rounds must bind the SAME candidate: the chain accepts older candidate hashes,
|
|
202
|
-
# so a
|
|
201
|
+
# so a candidate edited between rounds would otherwise be persisted as one coherent
|
|
202
|
+
# pair. Their packets may differ; each receipt records the packet it actually read.
|
|
203
203
|
python3 -c 'import json,sys; a=json.load(open(sys.argv[1])); b=json.load(open(sys.argv[2])); h=a.get("candidate_sha256"); sys.exit(0 if h and h==b.get("candidate_sha256") else 1)' \
|
|
204
204
|
"$ROUND1_RESULT_FILE" "$REVIEW_RUN_DIR/round2.json" \
|
|
205
|
-
|| { echo "round 2
|
|
205
|
+
|| { echo "round 2 bound a different candidate than round 1; rerun the pair on one frozen candidate" >&2; exit 1; }
|
|
206
206
|
cp "$REVIEW_RUN_DIR/round2.json" "$EVIDENCE_RUN_DIR/round2-challenge.json" || exit 1
|
|
207
207
|
cat "$EVIDENCE_RUN_DIR/round2-challenge.json"
|
|
208
208
|
# A secret-free diff egresses to non-Claude reviewers automatically; add
|
|
@@ -216,10 +216,10 @@ Run the script by path while keeping `--cwd` pointed at the product repository u
|
|
|
216
216
|
**The packet is the reviewer's whole world — compose it deliberately.** Review and challenge are built packet-bounded — Claude runs `--tools ""` with no `--add-dir`, and the other wrappers run in an isolated run workspace or a packet-only read surface. Treat the packet as the reviewer's whole world when deciding coverage: it is the only content bound by the packet hash and scanned before egress, so anything outside it is neither reliably visible to the reviewer nor covered by the verdict; a diff-only packet surfaces defects visible inside the changed lines and little else, and `--paths` only narrows it further. Whatever is absent from the packet is unreachable, not merely missed: a contradiction with an unchanged sibling clause, drift against a carrier outside the diff, or a silent weakening of upstream wording cannot be found by a reviewer who never saw the other side — that is the packet's shape, not the reviewer's weakness.
|
|
217
217
|
|
|
218
218
|
- Codex permits frozen-packet read/search; see [tool boundaries](references/development-completion.md#review-tools).
|
|
219
|
-
- To widen the packet, assemble it yourself and pass `--diff-file
|
|
220
|
-
- A verdict covers exactly the packet it was taken on
|
|
219
|
+
- To widen the packet, assemble it yourself and pass `--diff-file`, plus `--base` whenever the round must bind a landing candidate ([composition rules](references/staged-review-contract.md#the-packet-and-the-candidate)). It must name a regular file (no symlink or hardlink) holding text without NUL bytes. The gate hard-caps a packet at 200,000 bytes; split a larger candidate as described in the next bullet.
|
|
220
|
+
- A verdict covers exactly the packet it was taken on; the receipt records `packet_sha256` for those bytes and `candidate_sha256` for the base-derived candidate that will land, equal unless the packet was widened. A candidate too large for one packet is split by file group or risk class into a partition that still covers the whole candidate — every part in some packet, none dropped — each partition's verdict recorded against its own packet hash, and the candidate-wide claim withheld until every partition is conclusive; one partition's `no blocking findings` is never a verdict on the landing candidate. Cross-partition contradictions are unreachable by construction, so repeat the shared canonical context in every partition's packet and review anything that spans partitions as its own packet.
|
|
221
221
|
- Added context egresses to the selected reviewer exactly like the diff does, through the same credential tripwire — which catches machine-detectable secrets only. Paste rule text, carriers, and tool output; never paste credentials or material you would not send to that provider.
|
|
222
|
-
- A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane rather than editing the candidate to satisfy it.
|
|
222
|
+
- A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane, keeping `--base` so it still binds the same candidate, rather than editing the candidate to satisfy it.
|
|
223
223
|
|
|
224
224
|
When intentionally reviewing `code-review` itself, override the resolver from the ccl-skills repo under review before invoking the gate:
|
|
225
225
|
|
|
@@ -16,6 +16,36 @@ The budget is a ceiling, not a quota: after a clean or fully source-refuted trac
|
|
|
16
16
|
`autonomous_review_allowed=false`; release/high-risk still requires at least one
|
|
17
17
|
challenge before this early close is eligible.
|
|
18
18
|
|
|
19
|
+
## What `complete` closes, and what it does not
|
|
20
|
+
|
|
21
|
+
`--mode complete` is the checkpoint for a chain whose findings were **shown to
|
|
22
|
+
be wrong**. Every original occurrence must carry a `source_refuted`
|
|
23
|
+
disposition; `unresolved`, `accepted_risk`, `accepted_tradeoff`, and
|
|
24
|
+
`needs_human_decision` are refused, and that refusal is deliberate. The gate
|
|
25
|
+
binds structure and provenance, never authority: it cannot tell a human
|
|
26
|
+
acceptance from an agent that labelled its own findings accepted, so it does
|
|
27
|
+
not let an acceptance close a machine checkpoint.
|
|
28
|
+
|
|
29
|
+
**A chain whose findings are accepted, out of scope, or input defects is not
|
|
30
|
+
stalled — it is simply not closed by this mode.** Such a round ends at its
|
|
31
|
+
`findings` result with a recorded disposition per occurrence, and the round's
|
|
32
|
+
own ledger carries the wider vocabulary. Do not read a refused `complete` as an
|
|
33
|
+
unfinished review; read it as "no refutation was claimed". Reporting the round
|
|
34
|
+
requires the dispositions, not a completion receipt.
|
|
35
|
+
|
|
36
|
+
Two mechanics that cost time when they are discovered by experiment:
|
|
37
|
+
|
|
38
|
+
- **`--stage` and `--risk-tag` must be passed to `complete`, not omitted.** The
|
|
39
|
+
binding predicate compares the prior rounds against the profile derived from
|
|
40
|
+
the arguments given here, so a risk-tagged chain checked without its tags
|
|
41
|
+
fails as an unbound candidate rather than as a mismatch.
|
|
42
|
+
- **A round that edits `skills/code-review/scripts/**` cannot bind its own
|
|
43
|
+
earlier rounds.** `review_controller_sha256` covers every `.py` and `.sh`
|
|
44
|
+
there, so any further edit to the harness mid-round changes the controller
|
|
45
|
+
identity and both chain succession and `complete` refuse the earlier
|
|
46
|
+
receipts. Land every harness edit first, then run review and challenge back
|
|
47
|
+
to back with nothing changed in between.
|
|
48
|
+
|
|
19
49
|
## Plan and owner binding
|
|
20
50
|
|
|
21
51
|
The plan is optional for `review` and `challenge` and required for `complete`.
|
|
@@ -154,6 +184,46 @@ commit an in-scope path when Git should represent it; or compose complete
|
|
|
154
184
|
`--diff-file` partitions when the candidate must be split. Never omit a path
|
|
155
185
|
and report the remaining packet as the whole candidate.
|
|
156
186
|
|
|
187
|
+
## The packet and the candidate
|
|
188
|
+
|
|
189
|
+
They are two objects. The **packet** is what the reviewer reads; the **candidate**
|
|
190
|
+
is what will land and what `review_ledger_binding.py` recomputes at merge time.
|
|
191
|
+
A receipt records both hashes.
|
|
192
|
+
|
|
193
|
+
They hold the same value when the packet came from `--base` alone. Pass
|
|
194
|
+
`--diff-file` **with** `--base`/`--paths` to widen what the reviewer reads while
|
|
195
|
+
the round still binds the landing candidate — the shape an evidence-gap finding
|
|
196
|
+
needs, since editing the candidate would answer an input defect with a candidate
|
|
197
|
+
change. `--diff-file` alone binds no landing; only the combined form rejects a
|
|
198
|
+
`--wording-only-proof-file`.
|
|
199
|
+
|
|
200
|
+
What makes the widened form safe is a **prefix requirement**: the packet begins
|
|
201
|
+
with the base-derived candidate, byte for byte, so nothing lands unread.
|
|
202
|
+
|
|
203
|
+
- **Append context after the candidate diff.** Putting anything before the
|
|
204
|
+
candidate fails, and that is not cosmetic: a packet preceding it with a decoy
|
|
205
|
+
diff would read as the change while the real candidate read as context.
|
|
206
|
+
Interleaving context inside the candidate fails, as does dropping any part of
|
|
207
|
+
it. The reviewer is told where the candidate ends — the profile carries
|
|
208
|
+
`candidate_bytes` and states that exactly the first N packet bytes land — so
|
|
209
|
+
appended hunks that continue or seem to revert the diff cannot pass as it.
|
|
210
|
+
- **Keep the packet file outside the repository, and put nothing else in the
|
|
211
|
+
tree while the rounds run.** The controller counts every untracked path into
|
|
212
|
+
the candidate; the binder counts only committed content minus the receipt JSON
|
|
213
|
+
a round adds. A packet file, a superseded round's receipt, or any scratch
|
|
214
|
+
artifact in the worktree therefore moves the candidate the rounds bind and the
|
|
215
|
+
binder never computes it — the mirror of committing a plain-text attestation
|
|
216
|
+
after the rounds. Bound evidence lands before the rounds, receipts after,
|
|
217
|
+
nothing else present.
|
|
218
|
+
- **Read the candidate identity, do not reconstruct it.** `--print-candidate`
|
|
219
|
+
is the authority: its base is a fork point, its paths carry the round's
|
|
220
|
+
exclusions, and it refuses an uncommitted tree.
|
|
221
|
+
- Worth adding beyond the diff — the canonical rule the changed lines must not
|
|
222
|
+
contradict, sibling clauses, the carriers restating the change, gate output.
|
|
223
|
+
|
|
224
|
+
Rounds in one chain agree on the **candidate**, not the packet, which lets a
|
|
225
|
+
later round read more than an earlier one.
|
|
226
|
+
|
|
157
227
|
## Proof-bound wording-only single review
|
|
158
228
|
|
|
159
229
|
The wording-only exception is one untracked `review` with
|
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh
CHANGED
|
@@ -16,7 +16,7 @@ MAX_PROMPT_BYTES=245000
|
|
|
16
16
|
CHALLENGE_CLASSES="race conditions, data loss, security holes, auth bypass, lost or duplicated work, operational footguns"
|
|
17
17
|
|
|
18
18
|
emit_inconclusive() {
|
|
19
|
-
python3 - "$MODE" "$1" "${2:-invalid_input}" "${3:-false}" "${4:-}" <<'PY'
|
|
19
|
+
python3 - "$MODE" "$1" "${2:-invalid_input}" "${3:-false}" "${4:-}" "${5:-}" "${6:-}" <<'PY'
|
|
20
20
|
import json, sys
|
|
21
21
|
payload = {
|
|
22
22
|
"reviewer": "codex",
|
|
@@ -31,6 +31,10 @@ payload = {
|
|
|
31
31
|
}
|
|
32
32
|
if sys.argv[5]:
|
|
33
33
|
payload["transport_exit_code"] = int(sys.argv[5]) if sys.argv[5].isdigit() else sys.argv[5]
|
|
34
|
+
if sys.argv[6]:
|
|
35
|
+
payload["transport_diagnostic"] = sys.argv[6]
|
|
36
|
+
if sys.argv[7]:
|
|
37
|
+
payload["transport_run_dir"] = sys.argv[7]
|
|
34
38
|
print(json.dumps(payload, ensure_ascii=False, separators=(",", ":")))
|
|
35
39
|
PY
|
|
36
40
|
}
|
|
@@ -179,7 +183,15 @@ if [ "$REVIEW_SKILL_COUNT" -gt 0 ]; then
|
|
|
179
183
|
|| die_inconclusive codex_installed_skill_binding_invalid binding_mismatch false
|
|
180
184
|
fi
|
|
181
185
|
RUN_ROOT="$(mktemp -d "${TMPDIR:-/tmp}/codex-review.XXXXXX")"
|
|
182
|
-
|
|
186
|
+
# A failure that deletes its own evidence is the defect this round started from:
|
|
187
|
+
# rounds 122 and 123 left six receipts and no account of why the lane failed,
|
|
188
|
+
# because both captured streams went out with the run directory. On a transport
|
|
189
|
+
# failure the directory stays, and the receipt names it. It is mode 0700 under
|
|
190
|
+
# TMPDIR and holds exactly what it held while the run was in flight, so nothing
|
|
191
|
+
# is exposed that was not already; reclaiming it is the platform's temp-directory
|
|
192
|
+
# lifetime, as it is for every other run directory here.
|
|
193
|
+
PRESERVE_RUN_ROOT=0
|
|
194
|
+
cleanup() { [ "$PRESERVE_RUN_ROOT" = 1 ] || rm -rf "$RUN_ROOT"; }
|
|
183
195
|
trap cleanup EXIT
|
|
184
196
|
signal_inconclusive() {
|
|
185
197
|
emit_inconclusive codex_review_terminated operator_interrupt false
|
|
@@ -196,6 +208,142 @@ PROMPT_FILE="$RUN_ROOT/prompt.txt"
|
|
|
196
208
|
SCHEMA_FILE="$RUN_ROOT/schema.json"
|
|
197
209
|
RUN_WORKSPACE="$RUN_ROOT/workspace"
|
|
198
210
|
mkdir -p "$RUN_WORKSPACE"
|
|
211
|
+
# The reviewer runs from a private CODEX_HOME, not the user's. The user's home
|
|
212
|
+
# carries MCP servers -- their own, plus any an installed plugin contributes --
|
|
213
|
+
# and those servers run outside the CLI sandbox, so `--sandbox read-only` and
|
|
214
|
+
# `--disable shell_tool` do not reach them. A tool call completes before
|
|
215
|
+
# `audit_codex` can refuse the verdict, and a server auto-approves itself by
|
|
216
|
+
# declaring `readOnlyHint`, which the CLI trusts, so a tool that executes
|
|
217
|
+
# arbitrary code can be auto-approved while claiming to be read-only. Denying
|
|
218
|
+
# them without naming them was measured and does not work
|
|
219
|
+
# (`apps._default.default_tools_approval_mode` does not override the hint), and
|
|
220
|
+
# naming them cannot work either: an override under `mcp_servers` for a
|
|
221
|
+
# plugin-contributed server builds a transportless entry the CLI rejects
|
|
222
|
+
# outright. So this run gets a home that never had them.
|
|
223
|
+
#
|
|
224
|
+
# Model preferences are carried across explicitly, because this lane is
|
|
225
|
+
# contracted to review on the user's own default model and an empty home
|
|
226
|
+
# silently substitutes the CLI default. That carry-over is an allowlist, and
|
|
227
|
+
# deliberately not a denylist: a key this list has not heard of costs a
|
|
228
|
+
# preference, while a key a denylist has not heard of would let an executable
|
|
229
|
+
# server back in.
|
|
230
|
+
RUNTIME_HOME="$RUN_ROOT/codex-home"
|
|
231
|
+
mkdir -m 700 "$RUNTIME_HOME" \
|
|
232
|
+
|| die_inconclusive runtime_home_unavailable local_tool_failure false
|
|
233
|
+
AUTH_LINK_TARGET=""
|
|
234
|
+
if [ -e "$SOURCE_HOME/auth.json" ]; then
|
|
235
|
+
# A link, not a copy: the CLI refreshes the credential in place, and the
|
|
236
|
+
# rotated token has to land in the user's own file. The link is re-checked
|
|
237
|
+
# after the run, because a replaced link means the credential was written
|
|
238
|
+
# into this run directory instead.
|
|
239
|
+
AUTH_LINK_TARGET="$SOURCE_HOME/auth.json"
|
|
240
|
+
ln -s "$AUTH_LINK_TARGET" "$RUNTIME_HOME/auth.json" \
|
|
241
|
+
|| die_inconclusive runtime_home_auth_link_failed local_tool_failure false
|
|
242
|
+
fi
|
|
243
|
+
if [ -f "$SOURCE_HOME/config.toml" ]; then
|
|
244
|
+
python3 - "$SOURCE_HOME/config.toml" "$RUNTIME_HOME/config.toml" <<'PY_HOME_PREFERENCES' \
|
|
245
|
+
|| die_inconclusive codex_home_preferences_unreadable capability_missing true
|
|
246
|
+
import sys, tomllib
|
|
247
|
+
from pathlib import Path
|
|
248
|
+
|
|
249
|
+
# Model identity only, by KEY. `model_providers` is the exception worth naming:
|
|
250
|
+
# its value is a subtree this list does not inspect, so the allowlist bounds
|
|
251
|
+
# which keys travel, not everything that travels inside them. It is copied from
|
|
252
|
+
# the host's own configuration into a run-scoped home, so it grants a provider
|
|
253
|
+
# definition the host already had; narrowing it is a recorded follow-up.
|
|
254
|
+
# Nothing here can introduce a tool, a server, a hook, or a skill.
|
|
255
|
+
PREFERENCE_KEYS = (
|
|
256
|
+
"model",
|
|
257
|
+
"model_provider",
|
|
258
|
+
"model_providers",
|
|
259
|
+
"model_reasoning_effort",
|
|
260
|
+
"model_reasoning_summary",
|
|
261
|
+
"model_verbosity",
|
|
262
|
+
"service_tier",
|
|
263
|
+
)
|
|
264
|
+
try:
|
|
265
|
+
source = tomllib.loads(Path(sys.argv[1]).read_text(encoding="utf-8"))
|
|
266
|
+
except (OSError, UnicodeError, tomllib.TOMLDecodeError):
|
|
267
|
+
sys.exit(1)
|
|
268
|
+
if not isinstance(source, dict):
|
|
269
|
+
sys.exit(1)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
ESCAPES = {"\\": "\\\\", '"': '\\"', "\b": "\\b", "\t": "\\t",
|
|
273
|
+
"\n": "\\n", "\f": "\\f", "\r": "\\r"}
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def render_string(value):
|
|
277
|
+
# A basic TOML string cannot carry a literal newline or control character,
|
|
278
|
+
# and a key is a string too: an unquoted `proxy.v1` would silently become a
|
|
279
|
+
# dotted path and rewrite the provider map this run is supposed to copy.
|
|
280
|
+
out = []
|
|
281
|
+
for character in value:
|
|
282
|
+
if character in ESCAPES:
|
|
283
|
+
out.append(ESCAPES[character])
|
|
284
|
+
elif ord(character) < 0x20 or ord(character) == 0x7F:
|
|
285
|
+
out.append("\\u%04X" % ord(character))
|
|
286
|
+
else:
|
|
287
|
+
out.append(character)
|
|
288
|
+
return '"' + "".join(out) + '"'
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def render(value):
|
|
292
|
+
if isinstance(value, bool):
|
|
293
|
+
return "true" if value else "false"
|
|
294
|
+
if isinstance(value, (int, float)):
|
|
295
|
+
return repr(value)
|
|
296
|
+
if isinstance(value, str):
|
|
297
|
+
return render_string(value)
|
|
298
|
+
if isinstance(value, list):
|
|
299
|
+
return "[" + ", ".join(render(item) for item in value) + "]"
|
|
300
|
+
if isinstance(value, dict):
|
|
301
|
+
return "{" + ", ".join(
|
|
302
|
+
f"{render_string(key)} = {render(item)}" for key, item in value.items()
|
|
303
|
+
) + "}"
|
|
304
|
+
raise TypeError(value)
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
# A profile selects the model on many hosts, and the profile table itself is
|
|
308
|
+
# not copied: it can carry approval, sandbox, or server settings this run must
|
|
309
|
+
# not inherit. Resolve the selected profile's model identity into top-level
|
|
310
|
+
# keys instead, so a profile-configured host keeps its own model rather than
|
|
311
|
+
# silently falling back to the CLI default.
|
|
312
|
+
resolved = {key: source[key] for key in PREFERENCE_KEYS if key in source}
|
|
313
|
+
selected = source.get("profile")
|
|
314
|
+
if selected is not None:
|
|
315
|
+
# A selected profile that cannot be resolved is refused, not skipped:
|
|
316
|
+
# falling through would run the review on a different model than the host
|
|
317
|
+
# explicitly asked for, which is the substitution this carry-over exists to
|
|
318
|
+
# prevent.
|
|
319
|
+
profiles = source.get("profiles")
|
|
320
|
+
profile = profiles.get(selected) if isinstance(profiles, dict) and isinstance(selected, str) else None
|
|
321
|
+
if not isinstance(selected, str) or not selected or not isinstance(profile, dict):
|
|
322
|
+
sys.exit(1)
|
|
323
|
+
for key in PREFERENCE_KEYS:
|
|
324
|
+
if key in profile:
|
|
325
|
+
resolved[key] = profile[key]
|
|
326
|
+
lines = []
|
|
327
|
+
try:
|
|
328
|
+
for key in PREFERENCE_KEYS:
|
|
329
|
+
if key in resolved:
|
|
330
|
+
lines.append(f"{render_string(key)} = {render(resolved[key])}")
|
|
331
|
+
except TypeError:
|
|
332
|
+
sys.exit(1)
|
|
333
|
+
Path(sys.argv[2]).write_text("".join(line + "\n" for line in lines), encoding="utf-8")
|
|
334
|
+
PY_HOME_PREFERENCES
|
|
335
|
+
chmod 0600 "$RUNTIME_HOME/config.toml" 2>/dev/null || true
|
|
336
|
+
fi
|
|
337
|
+
if [ "$REVIEW_SKILL_COUNT" -gt 0 ]; then
|
|
338
|
+
# Copied, not linked: the CLI does not follow a symlinked skill directory,
|
|
339
|
+
# so a link here would silently cost the owner-skill binding.
|
|
340
|
+
mkdir -m 700 "$RUNTIME_HOME/skills" \
|
|
341
|
+
|| die_inconclusive runtime_home_unavailable local_tool_failure false
|
|
342
|
+
for review_skill in "${REVIEW_SKILLS[@]}"; do
|
|
343
|
+
cp -R "$INSTALLED_SKILL_REGISTRY_ROOT/$review_skill" "$RUNTIME_HOME/skills/$review_skill" \
|
|
344
|
+
|| die_inconclusive codex_installed_skill_unavailable capability_missing true
|
|
345
|
+
done
|
|
346
|
+
fi
|
|
199
347
|
MODEL=""
|
|
200
348
|
PROVIDER="openai"
|
|
201
349
|
FAMILY="openai"
|
|
@@ -261,34 +409,33 @@ import json, sys
|
|
|
261
409
|
values = [sys.argv[2], "--packet", sys.argv[1], "--sha256", sys.argv[3], "--allow-search"]
|
|
262
410
|
print('mcp_servers={code_review_packet={command=' + json.dumps(sys.executable)
|
|
263
411
|
+ ',args=[' + ','.join(json.dumps(value) for value in values)
|
|
264
|
-
+ '],enabled=true,enabled_tools=["read_packet","search_packet"]
|
|
412
|
+
+ '],enabled=true,enabled_tools=["read_packet","search_packet"]'
|
|
413
|
+
+ ',default_tools_approval_mode="approve"}}')
|
|
265
414
|
PY_MCP_CONFIG
|
|
266
415
|
)" || die_inconclusive packet_config_failed local_tool_failure false
|
|
267
|
-
#
|
|
268
|
-
#
|
|
416
|
+
# Inherited MCP servers are data, not a boundary. Disabling them by name was
|
|
417
|
+
# tried and cannot work: a plugin contributes its server outside `mcp_servers`,
|
|
418
|
+
# so `mcp_servers.<name>={enabled=false}` builds a transportless entry and the
|
|
419
|
+
# CLI refuses the whole configuration -- while leaving it enabled failed an
|
|
420
|
+
# exactly-one-server count. Either branch dead-ended the lane before inference.
|
|
421
|
+
# So this preflight verifies only that the frozen packet server is present and
|
|
422
|
+
# bound to the exact interpreter, script, packet and digest this run created.
|
|
423
|
+
#
|
|
424
|
+
# Accepted residual, measured rather than assumed: other servers stay enabled
|
|
425
|
+
# and CAN execute during a review. `audit_codex` refuses a verdict from any
|
|
426
|
+
# stream containing a foreign mcp_tool_call, but it runs afterwards -- the call
|
|
427
|
+
# has already completed, and a remote write or send cannot be undone by
|
|
428
|
+
# rejecting the verdict. Auto-approval is not a defence either: a server opts
|
|
429
|
+
# itself in by declaring `readOnlyHint` on a tool, which the CLI trusts, so a
|
|
430
|
+
# tool that executes arbitrary code can be auto-approved while claiming to be
|
|
431
|
+
# read-only. Two containment routes that name no server were measured and both
|
|
432
|
+
# failed: a global `apps._default.default_tools_approval_mode` did not override
|
|
433
|
+
# the hint, and `--disable plugins` would disable the reviewer's own installed
|
|
434
|
+
# skill registry, which ships as a plugin. The owner accepted this residual for
|
|
435
|
+
# this round; the route that would close it is a private CODEX_HOME seeded with
|
|
436
|
+
# auth and the registry only, as the Kimi lane already does.
|
|
269
437
|
CODEX_PACKET_CONFIG=(-c "$MCP_CONFIG" -c 'web_search="disabled"' -c 'approval_policy="never"')
|
|
270
|
-
CODEX_HOME="$
|
|
271
|
-
|| die_inconclusive codex_packet_tools_unavailable capability_missing true
|
|
272
|
-
MCP_CONFIG="$(python3 - "$RUN_ROOT/mcp.json" "$MCP_CONFIG" <<'PY_MCP_OVERRIDES'
|
|
273
|
-
import json, sys
|
|
274
|
-
from pathlib import Path
|
|
275
|
-
try:
|
|
276
|
-
rows = json.loads(Path(sys.argv[1]).read_text())
|
|
277
|
-
if not isinstance(rows, list):
|
|
278
|
-
raise ValueError()
|
|
279
|
-
disabled = []
|
|
280
|
-
for row in rows:
|
|
281
|
-
if not isinstance(row, dict) or not isinstance(row.get("name"), str) or not row["name"]:
|
|
282
|
-
raise ValueError()
|
|
283
|
-
if row["name"] != "code_review_packet":
|
|
284
|
-
disabled.append(json.dumps(row["name"]) + "={enabled=false}")
|
|
285
|
-
print(sys.argv[2][:-1] + "".join("," + entry for entry in disabled) + "}")
|
|
286
|
-
except (OSError, ValueError, TypeError):
|
|
287
|
-
sys.exit(1)
|
|
288
|
-
PY_MCP_OVERRIDES
|
|
289
|
-
)" || die_inconclusive codex_packet_tools_unavailable capability_missing true
|
|
290
|
-
CODEX_PACKET_CONFIG=(-c "$MCP_CONFIG" -c 'web_search="disabled"' -c 'approval_policy="never"')
|
|
291
|
-
CODEX_HOME="$SOURCE_HOME" timeout --kill-after=1s 5s "$CODEX_BIN_PATH" mcp list --json "${CODEX_PACKET_CONFIG[@]}" >"$RUN_ROOT/mcp.json" 2>"$STDERR_FILE" \
|
|
438
|
+
CODEX_HOME="$RUNTIME_HOME" timeout --kill-after=1s 5s "$CODEX_BIN_PATH" mcp list --json "${CODEX_PACKET_CONFIG[@]}" >"$RUN_ROOT/mcp.json" 2>"$STDERR_FILE" \
|
|
292
439
|
|| die_inconclusive codex_packet_tools_unavailable capability_missing true
|
|
293
440
|
python3 - "$RUN_ROOT/mcp.json" "$PACKET_FILE" "$PACKET_SERVER" "$PACKET_SHA256" <<'PY_MCP_CHECK' \
|
|
294
441
|
|| die_inconclusive codex_packet_tools_unavailable capability_missing true
|
|
@@ -298,12 +445,30 @@ try:
|
|
|
298
445
|
rows = json.loads(Path(sys.argv[1]).read_text())
|
|
299
446
|
if not isinstance(rows, list):
|
|
300
447
|
raise ValueError()
|
|
301
|
-
|
|
302
|
-
|
|
448
|
+
# Every row is validated before any filtering. Dropping the old enumeration
|
|
449
|
+
# also dropped its per-row name check, which let a malformed reply through
|
|
450
|
+
# whenever the malformed row happened to be disabled.
|
|
451
|
+
if any(
|
|
452
|
+
not isinstance(row, dict)
|
|
453
|
+
or not isinstance(row.get("name"), str)
|
|
454
|
+
or not row["name"]
|
|
455
|
+
for row in rows
|
|
456
|
+
):
|
|
457
|
+
raise ValueError()
|
|
458
|
+
# Under the private home this is an invariant the run establishes, not a
|
|
459
|
+
# bet on the user's configuration: nothing else was ever there to enable.
|
|
460
|
+
# A second enabled server means the home leaked, so refuse.
|
|
461
|
+
# Two predicates, not one: exactly one row carries the packet name
|
|
462
|
+
# anywhere in the reply, and exactly one row is enabled at all. Checking
|
|
463
|
+
# only the enabled set would accept a correctly bound row beside a disabled
|
|
464
|
+
# duplicate of the same name.
|
|
465
|
+
named = [row for row in rows if row.get("name") == "code_review_packet"]
|
|
466
|
+
active = [row for row in rows if row.get("enabled") is not False]
|
|
467
|
+
if len(named) != 1 or len(active) != 1 or active[0] is not named[0]:
|
|
303
468
|
raise ValueError()
|
|
304
469
|
row = active[0]
|
|
305
470
|
transport = row.get("transport", {})
|
|
306
|
-
if (row.get("
|
|
471
|
+
if (row.get("enabled") is not True
|
|
307
472
|
or transport.get("type") != "stdio" or transport.get("command") != sys.executable
|
|
308
473
|
or transport.get("args") != [sys.argv[3], "--packet", sys.argv[2], "--sha256", sys.argv[4], "--allow-search"]
|
|
309
474
|
or transport.get("env") or transport.get("env_vars") or transport.get("cwd")):
|
|
@@ -350,33 +515,114 @@ JSON
|
|
|
350
515
|
fi
|
|
351
516
|
|
|
352
517
|
run_started=$SECONDS
|
|
353
|
-
CMUX_CODEX_HOOKS_DISABLED=1 CODEX_HOME="$
|
|
518
|
+
CMUX_CODEX_HOOKS_DISABLED=1 CODEX_HOME="$RUNTIME_HOME" timeout --kill-after=1s "${TIMEOUT}s" "$CODEX_BIN_PATH" exec --disable hooks --disable shell_tool --sandbox read-only --ephemeral --skip-git-repo-check \
|
|
354
519
|
"${CODEX_PACKET_CONFIG[@]}" \
|
|
355
520
|
--json --output-schema "$SCHEMA_FILE" --output-last-message "$RESULT_FILE" \
|
|
356
521
|
-C "$RUN_WORKSPACE" - <"$PROMPT_FILE" >"$EVENTS" 2>"$STDERR_FILE"
|
|
357
522
|
run_rc=$?
|
|
358
523
|
run_elapsed=$((SECONDS - run_started))
|
|
524
|
+
if [ -n "$AUTH_LINK_TARGET" ]; then
|
|
525
|
+
[ -L "$RUNTIME_HOME/auth.json" ] \
|
|
526
|
+
&& [ "$(readlink "$RUNTIME_HOME/auth.json")" = "$AUTH_LINK_TARGET" ] \
|
|
527
|
+
|| die_inconclusive codex_runtime_home_credential_moved binding_mismatch false
|
|
528
|
+
fi
|
|
359
529
|
if [ "$run_rc" != 0 ]; then
|
|
530
|
+
# `codex exec --json` reports supply and credential failures as structured
|
|
531
|
+
# events on stdout, not on stderr, so a classifier reading only stderr sees a
|
|
532
|
+
# quota exhaustion as an unclassifiable failure and stops the reviewer lane
|
|
533
|
+
# instead of cascading.
|
|
534
|
+
#
|
|
535
|
+
# Only TOP-LEVEL error events are read. Model-authored content arrives nested
|
|
536
|
+
# under `item`, and the model quotes the packet, which is untrusted candidate
|
|
537
|
+
# data -- grepping the raw stream would let a reviewed diff pick the verdict
|
|
538
|
+
# for this lane by writing quota vocabulary into itself.
|
|
539
|
+
TRANSPORT_ERRORS="$RUN_ROOT/transport-errors.txt"
|
|
540
|
+
: >"$TRANSPORT_ERRORS"
|
|
541
|
+
python3 - "$EVENTS" >"$TRANSPORT_ERRORS" 2>/dev/null <<'PY_TRANSPORT_ERRORS'
|
|
542
|
+
import json, sys
|
|
543
|
+
from pathlib import Path
|
|
544
|
+
|
|
545
|
+
try:
|
|
546
|
+
lines = Path(sys.argv[1]).read_text(encoding="utf-8", errors="replace").splitlines()
|
|
547
|
+
except OSError:
|
|
548
|
+
sys.exit(0)
|
|
549
|
+
seen = set()
|
|
550
|
+
for line in lines:
|
|
551
|
+
try:
|
|
552
|
+
event = json.loads(line)
|
|
553
|
+
except ValueError:
|
|
554
|
+
continue
|
|
555
|
+
if not isinstance(event, dict):
|
|
556
|
+
continue
|
|
557
|
+
kind = event.get("type")
|
|
558
|
+
if not isinstance(kind, str) or not (kind == "error" or kind.endswith(".failed")):
|
|
559
|
+
continue
|
|
560
|
+
message = event.get("message")
|
|
561
|
+
if not isinstance(message, str):
|
|
562
|
+
nested = event.get("error")
|
|
563
|
+
message = nested.get("message") if isinstance(nested, dict) else None
|
|
564
|
+
if not isinstance(message, str) or not message:
|
|
565
|
+
continue
|
|
566
|
+
# A failing turn repeats the error event verbatim, and the diagnostic is
|
|
567
|
+
# bounded: relaying both would spend half the budget on one sentence.
|
|
568
|
+
message = " ".join(message.split())
|
|
569
|
+
if message not in seen:
|
|
570
|
+
seen.add(message)
|
|
571
|
+
print(message)
|
|
572
|
+
PY_TRANSPORT_ERRORS
|
|
573
|
+
PRESERVE_RUN_ROOT=1
|
|
574
|
+
# Physical paths on both sides, not the literal `$HOME` string: a home spelled
|
|
575
|
+
# with a trailing slash, or reached through a symlink, is the same directory
|
|
576
|
+
# and must elide the same way. Comparing the raw variable would put the
|
|
577
|
+
# username into a committed receipt on exactly those hosts.
|
|
578
|
+
TRANSPORT_RUN_DIR="$(cd "$RUN_ROOT" 2>/dev/null && pwd -P)" || TRANSPORT_RUN_DIR="$RUN_ROOT"
|
|
579
|
+
[ -n "$TRANSPORT_RUN_DIR" ] || TRANSPORT_RUN_DIR="$RUN_ROOT"
|
|
580
|
+
transport_home_real=""
|
|
581
|
+
if [ -n "${HOME:-}" ]; then
|
|
582
|
+
transport_home_real="$(cd "$HOME" 2>/dev/null && pwd -P)" || transport_home_real=""
|
|
583
|
+
fi
|
|
584
|
+
case "$transport_home_real" in
|
|
585
|
+
"" | */) transport_home_real="" ;;
|
|
586
|
+
esac
|
|
587
|
+
if [ -n "$transport_home_real" ]; then
|
|
588
|
+
case "$TRANSPORT_RUN_DIR" in
|
|
589
|
+
"$transport_home_real"/*)
|
|
590
|
+
TRANSPORT_RUN_DIR="~${TRANSPORT_RUN_DIR#"$transport_home_real"}" ;;
|
|
591
|
+
esac
|
|
592
|
+
fi
|
|
593
|
+
# The receipt carries NO text derived from the run. Eight review chains each
|
|
594
|
+
# found a different escape from a filter over that text -- an unlisted key
|
|
595
|
+
# name, an assignment form, URL userinfo, a password containing the separator,
|
|
596
|
+
# an escaped quote, an uppercase scheme -- because "nothing secret-shaped
|
|
597
|
+
# survives" is not a decidable property of free text, and an adversarial
|
|
598
|
+
# reviewer can always spell one more. So the free text is gone: what the
|
|
599
|
+
# transport said stays in the preserved run directory, and the receipt says
|
|
600
|
+
# where that is. The classifier still reads the extracted error messages
|
|
601
|
+
# above; those are matched against fixed patterns and never persisted.
|
|
602
|
+
TRANSPORT_DIAGNOSTIC="the transport output for this failure is in transport_run_dir"
|
|
603
|
+
|
|
360
604
|
if bash "$TIMEOUT_CLASSIFIER" "$run_rc" "$run_elapsed" "$TIMEOUT"; then
|
|
361
|
-
die_inconclusive codex_timeout timeout true "$run_rc"
|
|
605
|
+
die_inconclusive codex_timeout timeout true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
362
606
|
fi
|
|
363
607
|
case "$run_rc" in
|
|
364
|
-
129|130|137|143) die_inconclusive codex_process_interrupted operator_interrupt false "$run_rc" ;;
|
|
608
|
+
129|130|137|143) die_inconclusive codex_process_interrupted operator_interrupt false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR" ;;
|
|
365
609
|
esac
|
|
366
|
-
|
|
367
|
-
|
|
610
|
+
# `usage limit` is the wording the CLI actually uses for an exhausted account;
|
|
611
|
+
# none of the older patterns match it.
|
|
612
|
+
if grep -qiE '429|rate.?limit|quota|usage limit' "$STDERR_FILE" "$TRANSPORT_ERRORS"; then
|
|
613
|
+
die_inconclusive codex_quota quota true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
368
614
|
fi
|
|
369
|
-
if grep -qiE 'unauthori[sz]ed|authentication|login|api key' "$STDERR_FILE"; then
|
|
370
|
-
die_inconclusive codex_auth_unavailable provider_unavailable true "$run_rc"
|
|
615
|
+
if grep -qiE 'unauthori[sz]ed|authentication|login|api key' "$STDERR_FILE" "$TRANSPORT_ERRORS"; then
|
|
616
|
+
die_inconclusive codex_auth_unavailable provider_unavailable true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
371
617
|
fi
|
|
372
618
|
if [ ! -s "$EVENTS" ] && [ ! -s "$RESULT_FILE" ] \
|
|
373
619
|
&& grep -qiE 'failed to initialize in-process app-server client: Operation not permitted' "$STDERR_FILE"; then
|
|
374
620
|
if [ "$HOST_REMEDIATION_ATTEMPTED" -eq 1 ]; then
|
|
375
|
-
die_inconclusive codex_host_path_unavailable_after_host_retry host_path_unavailable_after_host_retry true "$run_rc"
|
|
621
|
+
die_inconclusive codex_host_path_unavailable_after_host_retry host_path_unavailable_after_host_retry true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
376
622
|
fi
|
|
377
|
-
die_inconclusive codex_host_path_unavailable host_path_unavailable false "$run_rc"
|
|
623
|
+
die_inconclusive codex_host_path_unavailable host_path_unavailable false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
378
624
|
fi
|
|
379
|
-
die_inconclusive codex_run_failed unknown_client_failure false "$run_rc"
|
|
625
|
+
die_inconclusive codex_run_failed unknown_client_failure false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
|
|
380
626
|
fi
|
|
381
627
|
|
|
382
628
|
python3 "$PARSER" --client codex --mode "$MODE" --implementer-family "$IMPL_FAMILY" \
|