@ccoalm/ccl-skills 0.15.4 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (16) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +15 -15
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +70 -0
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +286 -40
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +177 -89
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +396 -25
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +236 -1
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +1 -1
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +3 -0
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +20 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +56 -5
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +6 -0
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +68 -0
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +14 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/scenario-testing.md +1 -1
  15. package/dist/assets/release.json +17 -17
  16. package/package.json +1 -1
@@ -134,17 +134,16 @@ echo "code_review_skill_dir=$CODE_REVIEW_SKILL_DIR" >&2
134
134
  : "${REVIEW_CHAIN_ID:?set REVIEW_CHAIN_ID to a task-scoped chain id: letters, digits, dot, underscore, hyphen only}"
135
135
  : "${REVIEW_STAGE:?set REVIEW_STAGE to the stage this candidate is actually at: explore, build, or release}"
136
136
  : "${REVIEW_EVIDENCE_DIR:?set REVIEW_EVIDENCE_DIR to a durable directory you control for the per-round result rows}"
137
- # Exactly one frozen packet source: a packet you composed (REVIEW_DIFF_FILE, see the
138
- # packet-composition rules below) or a base ref (REVIEW_BASE). The gate rejects both.
139
- if [ -n "${REVIEW_DIFF_FILE:-}" ] && [ -n "${REVIEW_BASE:-}" ]; then
140
- echo "set exactly one of REVIEW_DIFF_FILE or REVIEW_BASE" >&2; exit 1
141
- elif [ -n "${REVIEW_DIFF_FILE:-}" ]; then
142
- PACKET_ARGS=(--diff-file "$REVIEW_DIFF_FILE")
143
- elif [ -n "${REVIEW_BASE:-}" ]; then
144
- PACKET_ARGS=(--base "$REVIEW_BASE")
145
- else
146
- echo "set exactly one of REVIEW_DIFF_FILE or REVIEW_BASE" >&2; exit 1
137
+ # REVIEW_BASE names the candidate, REVIEW_DIFF_FILE widens what the reviewer reads,
138
+ # and BOTH is a widened packet that must BEGIN with the candidate (composition
139
+ # rules: references/staged-review-contract.md). if-blocks, not `[ -n ... ] && ...`: a
140
+ # trailing false test returns non-zero and `set -e` would kill the caller.
141
+ if [ -z "${REVIEW_DIFF_FILE:-}" ] && [ -z "${REVIEW_BASE:-}" ]; then
142
+ echo "set REVIEW_BASE, REVIEW_DIFF_FILE, or both" >&2; exit 1
147
143
  fi
144
+ PACKET_ARGS=()
145
+ if [ -n "${REVIEW_BASE:-}" ]; then PACKET_ARGS+=(--base "$REVIEW_BASE"); fi
146
+ if [ -n "${REVIEW_DIFF_FILE:-}" ]; then PACKET_ARGS+=(--diff-file "$REVIEW_DIFF_FILE"); fi
148
147
  # REVIEW_RUN_DIR holds the raw round-1 result only for the chain handoff; the durable
149
148
  # per-round evidence is persisted to REVIEW_EVIDENCE_DIR, whose confidentiality you own.
150
149
  REVIEW_RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/review-run.XXXXXX")" || exit 1
@@ -199,10 +198,11 @@ bash "$CODE_REVIEW_SKILL_DIR/scripts/review_gate.sh" \
199
198
  >"$REVIEW_RUN_DIR/round2.json"
200
199
  require_tracked_result "$REVIEW_RUN_DIR/round2.json" challenge 2
201
200
  # Both rounds must bind the SAME candidate: the chain accepts older candidate hashes,
202
- # so a packet edited between rounds would otherwise be persisted as one coherent pair.
201
+ # so a candidate edited between rounds would otherwise be persisted as one coherent
202
+ # pair. Their packets may differ; each receipt records the packet it actually read.
203
203
  python3 -c 'import json,sys; a=json.load(open(sys.argv[1])); b=json.load(open(sys.argv[2])); h=a.get("candidate_sha256"); sys.exit(0 if h and h==b.get("candidate_sha256") else 1)' \
204
204
  "$ROUND1_RESULT_FILE" "$REVIEW_RUN_DIR/round2.json" \
205
- || { echo "round 2 reviewed a different packet than round 1; rerun the pair on one frozen candidate" >&2; exit 1; }
205
+ || { echo "round 2 bound a different candidate than round 1; rerun the pair on one frozen candidate" >&2; exit 1; }
206
206
  cp "$REVIEW_RUN_DIR/round2.json" "$EVIDENCE_RUN_DIR/round2-challenge.json" || exit 1
207
207
  cat "$EVIDENCE_RUN_DIR/round2-challenge.json"
208
208
  # A secret-free diff egresses to non-Claude reviewers automatically; add
@@ -216,10 +216,10 @@ Run the script by path while keeping `--cwd` pointed at the product repository u
216
216
  **The packet is the reviewer's whole world — compose it deliberately.** Review and challenge are built packet-bounded — Claude runs `--tools ""` with no `--add-dir`, and the other wrappers run in an isolated run workspace or a packet-only read surface. Treat the packet as the reviewer's whole world when deciding coverage: it is the only content bound by the packet hash and scanned before egress, so anything outside it is neither reliably visible to the reviewer nor covered by the verdict; a diff-only packet surfaces defects visible inside the changed lines and little else, and `--paths` only narrows it further. Whatever is absent from the packet is unreachable, not merely missed: a contradiction with an unchanged sibling clause, drift against a carrier outside the diff, or a silent weakening of upstream wording cannot be found by a reviewer who never saw the other side — that is the packet's shape, not the reviewer's weakness.
217
217
 
218
218
  - Codex permits frozen-packet read/search; see [tool boundaries](references/development-completion.md#review-tools).
219
- - To widen the packet, assemble it yourself and pass `--diff-file`: it replaces base-derived generation, is mutually exclusive with `--base`/`--paths`, and must name a regular file (no symlink or hardlink) holding text without NUL bytes. Worth adding beyond the diff — the canonical rule or contract text the changed lines must not contradict, the sibling clauses in the same file, the derived carriers that restate the change (commit message, MR/PR body), and the actual output of a gate or script under review. The gate hard-caps a packet at 200,000 bytes; split a larger candidate as described in the next bullet.
220
- - A verdict covers exactly the packet it was taken on, because the recorded packet hash is the reviewed identity. Within a packet, added context sits on top of the candidate diff and never in place of part of it. A candidate too large for one packet is split by file group or risk class into a partition that still covers the whole candidate — every part in some packet, none dropped — each partition's verdict recorded against its own packet hash, and the candidate-wide claim withheld until every partition is conclusive; one partition's `no blocking findings` is never a verdict on the landing candidate. Cross-partition contradictions are unreachable by construction, so repeat the shared canonical context in every partition's packet and review anything that spans partitions as its own packet.
219
+ - To widen the packet, assemble it yourself and pass `--diff-file`, plus `--base` whenever the round must bind a landing candidate ([composition rules](references/staged-review-contract.md#the-packet-and-the-candidate)). It must name a regular file (no symlink or hardlink) holding text without NUL bytes. The gate hard-caps a packet at 200,000 bytes; split a larger candidate as described in the next bullet.
220
+ - A verdict covers exactly the packet it was taken on; the receipt records `packet_sha256` for those bytes and `candidate_sha256` for the base-derived candidate that will land, equal unless the packet was widened. A candidate too large for one packet is split by file group or risk class into a partition that still covers the whole candidate — every part in some packet, none dropped — each partition's verdict recorded against its own packet hash, and the candidate-wide claim withheld until every partition is conclusive; one partition's `no blocking findings` is never a verdict on the landing candidate. Cross-partition contradictions are unreachable by construction, so repeat the shared canonical context in every partition's packet and review anything that spans partitions as its own packet.
221
221
  - Added context egresses to the selected reviewer exactly like the diff does, through the same credential tripwire — which catches machine-detectable secrets only. Paste rule text, carriers, and tool output; never paste credentials or material you would not send to that provider.
222
- - A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane rather than editing the candidate to satisfy it.
222
+ - A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane, keeping `--base` so it still binds the same candidate, rather than editing the candidate to satisfy it.
223
223
 
224
224
  When intentionally reviewing `code-review` itself, override the resolver from the ccl-skills repo under review before invoking the gate:
225
225
 
@@ -16,6 +16,36 @@ The budget is a ceiling, not a quota: after a clean or fully source-refuted trac
16
16
  `autonomous_review_allowed=false`; release/high-risk still requires at least one
17
17
  challenge before this early close is eligible.
18
18
 
19
+ ## What `complete` closes, and what it does not
20
+
21
+ `--mode complete` is the checkpoint for a chain whose findings were **shown to
22
+ be wrong**. Every original occurrence must carry a `source_refuted`
23
+ disposition; `unresolved`, `accepted_risk`, `accepted_tradeoff`, and
24
+ `needs_human_decision` are refused, and that refusal is deliberate. The gate
25
+ binds structure and provenance, never authority: it cannot tell a human
26
+ acceptance from an agent that labelled its own findings accepted, so it does
27
+ not let an acceptance close a machine checkpoint.
28
+
29
+ **A chain whose findings are accepted, out of scope, or input defects is not
30
+ stalled — it is simply not closed by this mode.** Such a round ends at its
31
+ `findings` result with a recorded disposition per occurrence, and the round's
32
+ own ledger carries the wider vocabulary. Do not read a refused `complete` as an
33
+ unfinished review; read it as "no refutation was claimed". Reporting the round
34
+ requires the dispositions, not a completion receipt.
35
+
36
+ Two mechanics that cost time when they are discovered by experiment:
37
+
38
+ - **`--stage` and `--risk-tag` must be passed to `complete`, not omitted.** The
39
+ binding predicate compares the prior rounds against the profile derived from
40
+ the arguments given here, so a risk-tagged chain checked without its tags
41
+ fails as an unbound candidate rather than as a mismatch.
42
+ - **A round that edits `skills/code-review/scripts/**` cannot bind its own
43
+ earlier rounds.** `review_controller_sha256` covers every `.py` and `.sh`
44
+ there, so any further edit to the harness mid-round changes the controller
45
+ identity and both chain succession and `complete` refuse the earlier
46
+ receipts. Land every harness edit first, then run review and challenge back
47
+ to back with nothing changed in between.
48
+
19
49
  ## Plan and owner binding
20
50
 
21
51
  The plan is optional for `review` and `challenge` and required for `complete`.
@@ -154,6 +184,46 @@ commit an in-scope path when Git should represent it; or compose complete
154
184
  `--diff-file` partitions when the candidate must be split. Never omit a path
155
185
  and report the remaining packet as the whole candidate.
156
186
 
187
+ ## The packet and the candidate
188
+
189
+ They are two objects. The **packet** is what the reviewer reads; the **candidate**
190
+ is what will land and what `review_ledger_binding.py` recomputes at merge time.
191
+ A receipt records both hashes.
192
+
193
+ They hold the same value when the packet came from `--base` alone. Pass
194
+ `--diff-file` **with** `--base`/`--paths` to widen what the reviewer reads while
195
+ the round still binds the landing candidate — the shape an evidence-gap finding
196
+ needs, since editing the candidate would answer an input defect with a candidate
197
+ change. `--diff-file` alone binds no landing; only the combined form rejects a
198
+ `--wording-only-proof-file`.
199
+
200
+ What makes the widened form safe is a **prefix requirement**: the packet begins
201
+ with the base-derived candidate, byte for byte, so nothing lands unread.
202
+
203
+ - **Append context after the candidate diff.** Putting anything before the
204
+ candidate fails, and that is not cosmetic: a packet preceding it with a decoy
205
+ diff would read as the change while the real candidate read as context.
206
+ Interleaving context inside the candidate fails, as does dropping any part of
207
+ it. The reviewer is told where the candidate ends — the profile carries
208
+ `candidate_bytes` and states that exactly the first N packet bytes land — so
209
+ appended hunks that continue or seem to revert the diff cannot pass as it.
210
+ - **Keep the packet file outside the repository, and put nothing else in the
211
+ tree while the rounds run.** The controller counts every untracked path into
212
+ the candidate; the binder counts only committed content minus the receipt JSON
213
+ a round adds. A packet file, a superseded round's receipt, or any scratch
214
+ artifact in the worktree therefore moves the candidate the rounds bind and the
215
+ binder never computes it — the mirror of committing a plain-text attestation
216
+ after the rounds. Bound evidence lands before the rounds, receipts after,
217
+ nothing else present.
218
+ - **Read the candidate identity, do not reconstruct it.** `--print-candidate`
219
+ is the authority: its base is a fork point, its paths carry the round's
220
+ exclusions, and it refuses an uncommitted tree.
221
+ - Worth adding beyond the diff — the canonical rule the changed lines must not
222
+ contradict, sibling clauses, the carriers restating the change, gate output.
223
+
224
+ Rounds in one chain agree on the **candidate**, not the packet, which lets a
225
+ later round read more than an earlier one.
226
+
157
227
  ## Proof-bound wording-only single review
158
228
 
159
229
  The wording-only exception is one untracked `review` with
@@ -16,7 +16,7 @@ MAX_PROMPT_BYTES=245000
16
16
  CHALLENGE_CLASSES="race conditions, data loss, security holes, auth bypass, lost or duplicated work, operational footguns"
17
17
 
18
18
  emit_inconclusive() {
19
- python3 - "$MODE" "$1" "${2:-invalid_input}" "${3:-false}" "${4:-}" <<'PY'
19
+ python3 - "$MODE" "$1" "${2:-invalid_input}" "${3:-false}" "${4:-}" "${5:-}" "${6:-}" <<'PY'
20
20
  import json, sys
21
21
  payload = {
22
22
  "reviewer": "codex",
@@ -31,6 +31,10 @@ payload = {
31
31
  }
32
32
  if sys.argv[5]:
33
33
  payload["transport_exit_code"] = int(sys.argv[5]) if sys.argv[5].isdigit() else sys.argv[5]
34
+ if sys.argv[6]:
35
+ payload["transport_diagnostic"] = sys.argv[6]
36
+ if sys.argv[7]:
37
+ payload["transport_run_dir"] = sys.argv[7]
34
38
  print(json.dumps(payload, ensure_ascii=False, separators=(",", ":")))
35
39
  PY
36
40
  }
@@ -179,7 +183,15 @@ if [ "$REVIEW_SKILL_COUNT" -gt 0 ]; then
179
183
  || die_inconclusive codex_installed_skill_binding_invalid binding_mismatch false
180
184
  fi
181
185
  RUN_ROOT="$(mktemp -d "${TMPDIR:-/tmp}/codex-review.XXXXXX")"
182
- cleanup() { rm -rf "$RUN_ROOT"; }
186
+ # A failure that deletes its own evidence is the defect this round started from:
187
+ # rounds 122 and 123 left six receipts and no account of why the lane failed,
188
+ # because both captured streams went out with the run directory. On a transport
189
+ # failure the directory stays, and the receipt names it. It is mode 0700 under
190
+ # TMPDIR and holds exactly what it held while the run was in flight, so nothing
191
+ # is exposed that was not already; reclaiming it is the platform's temp-directory
192
+ # lifetime, as it is for every other run directory here.
193
+ PRESERVE_RUN_ROOT=0
194
+ cleanup() { [ "$PRESERVE_RUN_ROOT" = 1 ] || rm -rf "$RUN_ROOT"; }
183
195
  trap cleanup EXIT
184
196
  signal_inconclusive() {
185
197
  emit_inconclusive codex_review_terminated operator_interrupt false
@@ -196,6 +208,142 @@ PROMPT_FILE="$RUN_ROOT/prompt.txt"
196
208
  SCHEMA_FILE="$RUN_ROOT/schema.json"
197
209
  RUN_WORKSPACE="$RUN_ROOT/workspace"
198
210
  mkdir -p "$RUN_WORKSPACE"
211
+ # The reviewer runs from a private CODEX_HOME, not the user's. The user's home
212
+ # carries MCP servers -- their own, plus any an installed plugin contributes --
213
+ # and those servers run outside the CLI sandbox, so `--sandbox read-only` and
214
+ # `--disable shell_tool` do not reach them. A tool call completes before
215
+ # `audit_codex` can refuse the verdict, and a server auto-approves itself by
216
+ # declaring `readOnlyHint`, which the CLI trusts, so a tool that executes
217
+ # arbitrary code can be auto-approved while claiming to be read-only. Denying
218
+ # them without naming them was measured and does not work
219
+ # (`apps._default.default_tools_approval_mode` does not override the hint), and
220
+ # naming them cannot work either: an override under `mcp_servers` for a
221
+ # plugin-contributed server builds a transportless entry the CLI rejects
222
+ # outright. So this run gets a home that never had them.
223
+ #
224
+ # Model preferences are carried across explicitly, because this lane is
225
+ # contracted to review on the user's own default model and an empty home
226
+ # silently substitutes the CLI default. That carry-over is an allowlist, and
227
+ # deliberately not a denylist: a key this list has not heard of costs a
228
+ # preference, while a key a denylist has not heard of would let an executable
229
+ # server back in.
230
+ RUNTIME_HOME="$RUN_ROOT/codex-home"
231
+ mkdir -m 700 "$RUNTIME_HOME" \
232
+ || die_inconclusive runtime_home_unavailable local_tool_failure false
233
+ AUTH_LINK_TARGET=""
234
+ if [ -e "$SOURCE_HOME/auth.json" ]; then
235
+ # A link, not a copy: the CLI refreshes the credential in place, and the
236
+ # rotated token has to land in the user's own file. The link is re-checked
237
+ # after the run, because a replaced link means the credential was written
238
+ # into this run directory instead.
239
+ AUTH_LINK_TARGET="$SOURCE_HOME/auth.json"
240
+ ln -s "$AUTH_LINK_TARGET" "$RUNTIME_HOME/auth.json" \
241
+ || die_inconclusive runtime_home_auth_link_failed local_tool_failure false
242
+ fi
243
+ if [ -f "$SOURCE_HOME/config.toml" ]; then
244
+ python3 - "$SOURCE_HOME/config.toml" "$RUNTIME_HOME/config.toml" <<'PY_HOME_PREFERENCES' \
245
+ || die_inconclusive codex_home_preferences_unreadable capability_missing true
246
+ import sys, tomllib
247
+ from pathlib import Path
248
+
249
+ # Model identity only, by KEY. `model_providers` is the exception worth naming:
250
+ # its value is a subtree this list does not inspect, so the allowlist bounds
251
+ # which keys travel, not everything that travels inside them. It is copied from
252
+ # the host's own configuration into a run-scoped home, so it grants a provider
253
+ # definition the host already had; narrowing it is a recorded follow-up.
254
+ # Nothing here can introduce a tool, a server, a hook, or a skill.
255
+ PREFERENCE_KEYS = (
256
+ "model",
257
+ "model_provider",
258
+ "model_providers",
259
+ "model_reasoning_effort",
260
+ "model_reasoning_summary",
261
+ "model_verbosity",
262
+ "service_tier",
263
+ )
264
+ try:
265
+ source = tomllib.loads(Path(sys.argv[1]).read_text(encoding="utf-8"))
266
+ except (OSError, UnicodeError, tomllib.TOMLDecodeError):
267
+ sys.exit(1)
268
+ if not isinstance(source, dict):
269
+ sys.exit(1)
270
+
271
+
272
+ ESCAPES = {"\\": "\\\\", '"': '\\"', "\b": "\\b", "\t": "\\t",
273
+ "\n": "\\n", "\f": "\\f", "\r": "\\r"}
274
+
275
+
276
+ def render_string(value):
277
+ # A basic TOML string cannot carry a literal newline or control character,
278
+ # and a key is a string too: an unquoted `proxy.v1` would silently become a
279
+ # dotted path and rewrite the provider map this run is supposed to copy.
280
+ out = []
281
+ for character in value:
282
+ if character in ESCAPES:
283
+ out.append(ESCAPES[character])
284
+ elif ord(character) < 0x20 or ord(character) == 0x7F:
285
+ out.append("\\u%04X" % ord(character))
286
+ else:
287
+ out.append(character)
288
+ return '"' + "".join(out) + '"'
289
+
290
+
291
+ def render(value):
292
+ if isinstance(value, bool):
293
+ return "true" if value else "false"
294
+ if isinstance(value, (int, float)):
295
+ return repr(value)
296
+ if isinstance(value, str):
297
+ return render_string(value)
298
+ if isinstance(value, list):
299
+ return "[" + ", ".join(render(item) for item in value) + "]"
300
+ if isinstance(value, dict):
301
+ return "{" + ", ".join(
302
+ f"{render_string(key)} = {render(item)}" for key, item in value.items()
303
+ ) + "}"
304
+ raise TypeError(value)
305
+
306
+
307
+ # A profile selects the model on many hosts, and the profile table itself is
308
+ # not copied: it can carry approval, sandbox, or server settings this run must
309
+ # not inherit. Resolve the selected profile's model identity into top-level
310
+ # keys instead, so a profile-configured host keeps its own model rather than
311
+ # silently falling back to the CLI default.
312
+ resolved = {key: source[key] for key in PREFERENCE_KEYS if key in source}
313
+ selected = source.get("profile")
314
+ if selected is not None:
315
+ # A selected profile that cannot be resolved is refused, not skipped:
316
+ # falling through would run the review on a different model than the host
317
+ # explicitly asked for, which is the substitution this carry-over exists to
318
+ # prevent.
319
+ profiles = source.get("profiles")
320
+ profile = profiles.get(selected) if isinstance(profiles, dict) and isinstance(selected, str) else None
321
+ if not isinstance(selected, str) or not selected or not isinstance(profile, dict):
322
+ sys.exit(1)
323
+ for key in PREFERENCE_KEYS:
324
+ if key in profile:
325
+ resolved[key] = profile[key]
326
+ lines = []
327
+ try:
328
+ for key in PREFERENCE_KEYS:
329
+ if key in resolved:
330
+ lines.append(f"{render_string(key)} = {render(resolved[key])}")
331
+ except TypeError:
332
+ sys.exit(1)
333
+ Path(sys.argv[2]).write_text("".join(line + "\n" for line in lines), encoding="utf-8")
334
+ PY_HOME_PREFERENCES
335
+ chmod 0600 "$RUNTIME_HOME/config.toml" 2>/dev/null || true
336
+ fi
337
+ if [ "$REVIEW_SKILL_COUNT" -gt 0 ]; then
338
+ # Copied, not linked: the CLI does not follow a symlinked skill directory,
339
+ # so a link here would silently cost the owner-skill binding.
340
+ mkdir -m 700 "$RUNTIME_HOME/skills" \
341
+ || die_inconclusive runtime_home_unavailable local_tool_failure false
342
+ for review_skill in "${REVIEW_SKILLS[@]}"; do
343
+ cp -R "$INSTALLED_SKILL_REGISTRY_ROOT/$review_skill" "$RUNTIME_HOME/skills/$review_skill" \
344
+ || die_inconclusive codex_installed_skill_unavailable capability_missing true
345
+ done
346
+ fi
199
347
  MODEL=""
200
348
  PROVIDER="openai"
201
349
  FAMILY="openai"
@@ -261,34 +409,33 @@ import json, sys
261
409
  values = [sys.argv[2], "--packet", sys.argv[1], "--sha256", sys.argv[3], "--allow-search"]
262
410
  print('mcp_servers={code_review_packet={command=' + json.dumps(sys.executable)
263
411
  + ',args=[' + ','.join(json.dumps(value) for value in values)
264
- + '],enabled=true,enabled_tools=["read_packet","search_packet"]}}')
412
+ + '],enabled=true,enabled_tools=["read_packet","search_packet"]'
413
+ + ',default_tools_approval_mode="approve"}}')
265
414
  PY_MCP_CONFIG
266
415
  )" || die_inconclusive packet_config_failed local_tool_failure false
267
- # TOML overrides merge server tables. Disable inherited servers for this run
268
- # without changing user configuration, then verify the effective public list.
416
+ # Inherited MCP servers are data, not a boundary. Disabling them by name was
417
+ # tried and cannot work: a plugin contributes its server outside `mcp_servers`,
418
+ # so `mcp_servers.<name>={enabled=false}` builds a transportless entry and the
419
+ # CLI refuses the whole configuration -- while leaving it enabled failed an
420
+ # exactly-one-server count. Either branch dead-ended the lane before inference.
421
+ # So this preflight verifies only that the frozen packet server is present and
422
+ # bound to the exact interpreter, script, packet and digest this run created.
423
+ #
424
+ # Accepted residual, measured rather than assumed: other servers stay enabled
425
+ # and CAN execute during a review. `audit_codex` refuses a verdict from any
426
+ # stream containing a foreign mcp_tool_call, but it runs afterwards -- the call
427
+ # has already completed, and a remote write or send cannot be undone by
428
+ # rejecting the verdict. Auto-approval is not a defence either: a server opts
429
+ # itself in by declaring `readOnlyHint` on a tool, which the CLI trusts, so a
430
+ # tool that executes arbitrary code can be auto-approved while claiming to be
431
+ # read-only. Two containment routes that name no server were measured and both
432
+ # failed: a global `apps._default.default_tools_approval_mode` did not override
433
+ # the hint, and `--disable plugins` would disable the reviewer's own installed
434
+ # skill registry, which ships as a plugin. The owner accepted this residual for
435
+ # this round; the route that would close it is a private CODEX_HOME seeded with
436
+ # auth and the registry only, as the Kimi lane already does.
269
437
  CODEX_PACKET_CONFIG=(-c "$MCP_CONFIG" -c 'web_search="disabled"' -c 'approval_policy="never"')
270
- CODEX_HOME="$SOURCE_HOME" timeout --kill-after=1s 5s "$CODEX_BIN_PATH" mcp list --json "${CODEX_PACKET_CONFIG[@]}" >"$RUN_ROOT/mcp.json" 2>"$STDERR_FILE" \
271
- || die_inconclusive codex_packet_tools_unavailable capability_missing true
272
- MCP_CONFIG="$(python3 - "$RUN_ROOT/mcp.json" "$MCP_CONFIG" <<'PY_MCP_OVERRIDES'
273
- import json, sys
274
- from pathlib import Path
275
- try:
276
- rows = json.loads(Path(sys.argv[1]).read_text())
277
- if not isinstance(rows, list):
278
- raise ValueError()
279
- disabled = []
280
- for row in rows:
281
- if not isinstance(row, dict) or not isinstance(row.get("name"), str) or not row["name"]:
282
- raise ValueError()
283
- if row["name"] != "code_review_packet":
284
- disabled.append(json.dumps(row["name"]) + "={enabled=false}")
285
- print(sys.argv[2][:-1] + "".join("," + entry for entry in disabled) + "}")
286
- except (OSError, ValueError, TypeError):
287
- sys.exit(1)
288
- PY_MCP_OVERRIDES
289
- )" || die_inconclusive codex_packet_tools_unavailable capability_missing true
290
- CODEX_PACKET_CONFIG=(-c "$MCP_CONFIG" -c 'web_search="disabled"' -c 'approval_policy="never"')
291
- CODEX_HOME="$SOURCE_HOME" timeout --kill-after=1s 5s "$CODEX_BIN_PATH" mcp list --json "${CODEX_PACKET_CONFIG[@]}" >"$RUN_ROOT/mcp.json" 2>"$STDERR_FILE" \
438
+ CODEX_HOME="$RUNTIME_HOME" timeout --kill-after=1s 5s "$CODEX_BIN_PATH" mcp list --json "${CODEX_PACKET_CONFIG[@]}" >"$RUN_ROOT/mcp.json" 2>"$STDERR_FILE" \
292
439
  || die_inconclusive codex_packet_tools_unavailable capability_missing true
293
440
  python3 - "$RUN_ROOT/mcp.json" "$PACKET_FILE" "$PACKET_SERVER" "$PACKET_SHA256" <<'PY_MCP_CHECK' \
294
441
  || die_inconclusive codex_packet_tools_unavailable capability_missing true
@@ -298,12 +445,30 @@ try:
298
445
  rows = json.loads(Path(sys.argv[1]).read_text())
299
446
  if not isinstance(rows, list):
300
447
  raise ValueError()
301
- active = [row for row in rows if isinstance(row, dict) and row.get("enabled") is not False]
302
- if len(active) != 1 or len(rows) != len([row for row in rows if isinstance(row, dict)]):
448
+ # Every row is validated before any filtering. Dropping the old enumeration
449
+ # also dropped its per-row name check, which let a malformed reply through
450
+ # whenever the malformed row happened to be disabled.
451
+ if any(
452
+ not isinstance(row, dict)
453
+ or not isinstance(row.get("name"), str)
454
+ or not row["name"]
455
+ for row in rows
456
+ ):
457
+ raise ValueError()
458
+ # Under the private home this is an invariant the run establishes, not a
459
+ # bet on the user's configuration: nothing else was ever there to enable.
460
+ # A second enabled server means the home leaked, so refuse.
461
+ # Two predicates, not one: exactly one row carries the packet name
462
+ # anywhere in the reply, and exactly one row is enabled at all. Checking
463
+ # only the enabled set would accept a correctly bound row beside a disabled
464
+ # duplicate of the same name.
465
+ named = [row for row in rows if row.get("name") == "code_review_packet"]
466
+ active = [row for row in rows if row.get("enabled") is not False]
467
+ if len(named) != 1 or len(active) != 1 or active[0] is not named[0]:
303
468
  raise ValueError()
304
469
  row = active[0]
305
470
  transport = row.get("transport", {})
306
- if (row.get("name") != "code_review_packet" or row.get("enabled") is not True
471
+ if (row.get("enabled") is not True
307
472
  or transport.get("type") != "stdio" or transport.get("command") != sys.executable
308
473
  or transport.get("args") != [sys.argv[3], "--packet", sys.argv[2], "--sha256", sys.argv[4], "--allow-search"]
309
474
  or transport.get("env") or transport.get("env_vars") or transport.get("cwd")):
@@ -350,33 +515,114 @@ JSON
350
515
  fi
351
516
 
352
517
  run_started=$SECONDS
353
- CMUX_CODEX_HOOKS_DISABLED=1 CODEX_HOME="$SOURCE_HOME" timeout --kill-after=1s "${TIMEOUT}s" "$CODEX_BIN_PATH" exec --disable hooks --disable shell_tool --sandbox read-only --ephemeral --skip-git-repo-check \
518
+ CMUX_CODEX_HOOKS_DISABLED=1 CODEX_HOME="$RUNTIME_HOME" timeout --kill-after=1s "${TIMEOUT}s" "$CODEX_BIN_PATH" exec --disable hooks --disable shell_tool --sandbox read-only --ephemeral --skip-git-repo-check \
354
519
  "${CODEX_PACKET_CONFIG[@]}" \
355
520
  --json --output-schema "$SCHEMA_FILE" --output-last-message "$RESULT_FILE" \
356
521
  -C "$RUN_WORKSPACE" - <"$PROMPT_FILE" >"$EVENTS" 2>"$STDERR_FILE"
357
522
  run_rc=$?
358
523
  run_elapsed=$((SECONDS - run_started))
524
+ if [ -n "$AUTH_LINK_TARGET" ]; then
525
+ [ -L "$RUNTIME_HOME/auth.json" ] \
526
+ && [ "$(readlink "$RUNTIME_HOME/auth.json")" = "$AUTH_LINK_TARGET" ] \
527
+ || die_inconclusive codex_runtime_home_credential_moved binding_mismatch false
528
+ fi
359
529
  if [ "$run_rc" != 0 ]; then
530
+ # `codex exec --json` reports supply and credential failures as structured
531
+ # events on stdout, not on stderr, so a classifier reading only stderr sees a
532
+ # quota exhaustion as an unclassifiable failure and stops the reviewer lane
533
+ # instead of cascading.
534
+ #
535
+ # Only TOP-LEVEL error events are read. Model-authored content arrives nested
536
+ # under `item`, and the model quotes the packet, which is untrusted candidate
537
+ # data -- grepping the raw stream would let a reviewed diff pick the verdict
538
+ # for this lane by writing quota vocabulary into itself.
539
+ TRANSPORT_ERRORS="$RUN_ROOT/transport-errors.txt"
540
+ : >"$TRANSPORT_ERRORS"
541
+ python3 - "$EVENTS" >"$TRANSPORT_ERRORS" 2>/dev/null <<'PY_TRANSPORT_ERRORS'
542
+ import json, sys
543
+ from pathlib import Path
544
+
545
+ try:
546
+ lines = Path(sys.argv[1]).read_text(encoding="utf-8", errors="replace").splitlines()
547
+ except OSError:
548
+ sys.exit(0)
549
+ seen = set()
550
+ for line in lines:
551
+ try:
552
+ event = json.loads(line)
553
+ except ValueError:
554
+ continue
555
+ if not isinstance(event, dict):
556
+ continue
557
+ kind = event.get("type")
558
+ if not isinstance(kind, str) or not (kind == "error" or kind.endswith(".failed")):
559
+ continue
560
+ message = event.get("message")
561
+ if not isinstance(message, str):
562
+ nested = event.get("error")
563
+ message = nested.get("message") if isinstance(nested, dict) else None
564
+ if not isinstance(message, str) or not message:
565
+ continue
566
+ # A failing turn repeats the error event verbatim, and the diagnostic is
567
+ # bounded: relaying both would spend half the budget on one sentence.
568
+ message = " ".join(message.split())
569
+ if message not in seen:
570
+ seen.add(message)
571
+ print(message)
572
+ PY_TRANSPORT_ERRORS
573
+ PRESERVE_RUN_ROOT=1
574
+ # Physical paths on both sides, not the literal `$HOME` string: a home spelled
575
+ # with a trailing slash, or reached through a symlink, is the same directory
576
+ # and must elide the same way. Comparing the raw variable would put the
577
+ # username into a committed receipt on exactly those hosts.
578
+ TRANSPORT_RUN_DIR="$(cd "$RUN_ROOT" 2>/dev/null && pwd -P)" || TRANSPORT_RUN_DIR="$RUN_ROOT"
579
+ [ -n "$TRANSPORT_RUN_DIR" ] || TRANSPORT_RUN_DIR="$RUN_ROOT"
580
+ transport_home_real=""
581
+ if [ -n "${HOME:-}" ]; then
582
+ transport_home_real="$(cd "$HOME" 2>/dev/null && pwd -P)" || transport_home_real=""
583
+ fi
584
+ case "$transport_home_real" in
585
+ "" | */) transport_home_real="" ;;
586
+ esac
587
+ if [ -n "$transport_home_real" ]; then
588
+ case "$TRANSPORT_RUN_DIR" in
589
+ "$transport_home_real"/*)
590
+ TRANSPORT_RUN_DIR="~${TRANSPORT_RUN_DIR#"$transport_home_real"}" ;;
591
+ esac
592
+ fi
593
+ # The receipt carries NO text derived from the run. Eight review chains each
594
+ # found a different escape from a filter over that text -- an unlisted key
595
+ # name, an assignment form, URL userinfo, a password containing the separator,
596
+ # an escaped quote, an uppercase scheme -- because "nothing secret-shaped
597
+ # survives" is not a decidable property of free text, and an adversarial
598
+ # reviewer can always spell one more. So the free text is gone: what the
599
+ # transport said stays in the preserved run directory, and the receipt says
600
+ # where that is. The classifier still reads the extracted error messages
601
+ # above; those are matched against fixed patterns and never persisted.
602
+ TRANSPORT_DIAGNOSTIC="the transport output for this failure is in transport_run_dir"
603
+
360
604
  if bash "$TIMEOUT_CLASSIFIER" "$run_rc" "$run_elapsed" "$TIMEOUT"; then
361
- die_inconclusive codex_timeout timeout true "$run_rc"
605
+ die_inconclusive codex_timeout timeout true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
362
606
  fi
363
607
  case "$run_rc" in
364
- 129|130|137|143) die_inconclusive codex_process_interrupted operator_interrupt false "$run_rc" ;;
608
+ 129|130|137|143) die_inconclusive codex_process_interrupted operator_interrupt false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR" ;;
365
609
  esac
366
- if grep -qiE '429|rate.?limit|quota' "$STDERR_FILE"; then
367
- die_inconclusive codex_quota quota true "$run_rc"
610
+ # `usage limit` is the wording the CLI actually uses for an exhausted account;
611
+ # none of the older patterns match it.
612
+ if grep -qiE '429|rate.?limit|quota|usage limit' "$STDERR_FILE" "$TRANSPORT_ERRORS"; then
613
+ die_inconclusive codex_quota quota true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
368
614
  fi
369
- if grep -qiE 'unauthori[sz]ed|authentication|login|api key' "$STDERR_FILE"; then
370
- die_inconclusive codex_auth_unavailable provider_unavailable true "$run_rc"
615
+ if grep -qiE 'unauthori[sz]ed|authentication|login|api key' "$STDERR_FILE" "$TRANSPORT_ERRORS"; then
616
+ die_inconclusive codex_auth_unavailable provider_unavailable true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
371
617
  fi
372
618
  if [ ! -s "$EVENTS" ] && [ ! -s "$RESULT_FILE" ] \
373
619
  && grep -qiE 'failed to initialize in-process app-server client: Operation not permitted' "$STDERR_FILE"; then
374
620
  if [ "$HOST_REMEDIATION_ATTEMPTED" -eq 1 ]; then
375
- die_inconclusive codex_host_path_unavailable_after_host_retry host_path_unavailable_after_host_retry true "$run_rc"
621
+ die_inconclusive codex_host_path_unavailable_after_host_retry host_path_unavailable_after_host_retry true "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
376
622
  fi
377
- die_inconclusive codex_host_path_unavailable host_path_unavailable false "$run_rc"
623
+ die_inconclusive codex_host_path_unavailable host_path_unavailable false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
378
624
  fi
379
- die_inconclusive codex_run_failed unknown_client_failure false "$run_rc"
625
+ die_inconclusive codex_run_failed unknown_client_failure false "$run_rc" "$TRANSPORT_DIAGNOSTIC" "$TRANSPORT_RUN_DIR"
380
626
  fi
381
627
 
382
628
  python3 "$PARSER" --client codex --mode "$MODE" --implementer-family "$IMPL_FAMILY" \