@ccoalm/ccl-skills 0.15.1 → 0.15.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +3 -1
  2. package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +1 -1
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +2 -2
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -5
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +26 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +37 -13
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +5 -2
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +77 -5
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_packet_mcp.py +98 -4
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_cli_review.py +48 -1
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +230 -16
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +165 -11
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_kimi_packet_mcp.py +143 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +572 -0
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +3 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +1 -1
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -0
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +1 -1
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +2 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +3 -1
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +2 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +1 -1
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +20 -11
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/refactoring-discipline.md +7 -1
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +1 -1
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +2 -2
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +17 -17
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +4 -4
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/resume-paused-delivery.md +3 -3
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +12 -0
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +83 -48
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +80 -2
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +3 -2
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +190 -2
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +106 -4
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +1 -1
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +3 -1
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +6 -6
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/writing-judgments.md +63 -0
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -1
  44. package/dist/assets/release.json +60 -50
  45. package/dist/codex-host.d.ts +1 -1
  46. package/dist/codex-host.js +39 -15
  47. package/dist/host-probe.d.ts +16 -0
  48. package/dist/host-probe.js +29 -5
  49. package/dist/operations.js +34 -8
  50. package/dist/unified.d.ts +1 -1
  51. package/dist/unified.js +11 -4
  52. package/package.json +1 -1
package/README.md CHANGED
@@ -23,7 +23,7 @@ Restart your CLI so it reloads the skills.
23
23
 
24
24
  `install` configures every host it detects. The package carries an immutable snapshot of the skills, agent context, plugin manifests, and runtime hooks, so installation needs no Git checkout.
25
25
 
26
- Requirements: Node.js 20 or later, macOS or Linux, and at least one host CLI — Claude Code, Codex 0.133.0 or later, or OpenCode.
26
+ Requirements: Node.js 20 or later, macOS or Linux, and at least one host CLI — Claude Code, Codex with working `plugin marketplace list` and `plugin list` commands, or OpenCode. Codex availability is checked through these commands rather than its version number.
27
27
 
28
28
  Run it without a global install:
29
29
 
@@ -62,6 +62,8 @@ ccl-skills uninstall --yes # remove host assets
62
62
 
63
63
  Limit any operation to one host with `--host claude`, `--host codex`, or `--host opencode`. Add `--json` for machine-readable output.
64
64
 
65
+ For `--host codex`, unreadable public plugin state returns exit `3` with `host-state-unknown`; a missing CLI or failed capability probe returns `4`. If that host failure occurs with a pending journal, recovery is deferred: exit `5` with `partial-journal` retains the journal and records `details.hostFailure`. Restore the CLI or readable public plugin state, then rerun the command. Other outcomes can share these exit codes, so inspect the JSON status as well.
66
+
65
67
  `update` and `uninstall` are previews unless `--yes` is supplied. `update --yes` first upgrades the global npm package to `@latest`, then asks the freshly installed CLI to refresh host assets. Set `CCL_SKILLS_SKIP_SELF_UPDATE=1` for an assets-only refresh; `--allow-downgrade` always uses the currently invoked package without installing `@latest` first.
66
68
 
67
69
  After `ccl-skills uninstall --yes`, remove the CLI package itself with `npm uninstall --global @ccoalm/ccl-skills` if it is no longer needed.
@@ -41,5 +41,5 @@
41
41
  - **完整优先**:能多花几分钟做完就别交半成品;但"完整"是把该做的做完,不是镀金或扩范围(详见 product-rd / feature-risk-router 的 gate)。
42
42
  - **持久件锚定(长/多阶段/委托/跨会话工作)**:锚到持久件、别只靠对话或临时任务卡——交付级 spec/plan → product-rd-workflow、委托进度 → multi-agent-delegation、技能/流程教训 → skill-extraction-workflow 的 source-register;更新/取代既有件,别复制(只提醒,不是第二个 plan 门,深度归 product-rd)。
43
43
  - **大文件/大技能分块读(读取易丢中段)**:单次读取**输出**超过 ~256 行 / 10KB 时,codex 等工具会头尾截断、丢中段([openai/codex#6426](https://github.com/openai/codex/issues/6426)),常有截断标记但极易忽略、某些场景无标记(无标记 ≠ 读全)。需要看全时(完整评审 / 下"没有 X"结论 / 加载技能照做)分块读(每块 < ~200 行**且** < 8KB)并确认**中段**已读到,别一次整文件读就当看全(定点 `sed -n 'Np'` 不受限)。写码/测试/评审同样适用,详见 skill-extraction blocked-source-read。(`project_doc_max_bytes` 只管 project-doc 预算、不影响工具输出截断,不是绕过手段。)
44
- - **委托三方模型评审前先自审到收敛(任何开发都适用)**:codex/Claude 等是**最后一层对抗性兜底,不是主要发现机制**。先做**可证明的实现者自检**到预期通过再交评审——想清楚影响结果的分支/路径/状态,自审 security/privacy/authority/数据丢失轴;别把未收敛的活丢过去,再拿 findings 当清单陷进 review→改→再 review 的循环。深浅按对应 owner/风险闸定,不无差别全量、也不给窄/未触发改动强加 product-rd self-review row。自审**不缩短/不软化**那道闸(仍是一次全量对抗性 pass,按 diff 给对抗 prompt、别附"确认我的结论"导向),findings 仍按门处置不可降级。详见 product-rd 验证门 + skill-extraction `dual-track-review-gate.md`(shared-skill dual-track 的 persisted row 与有效性规则按 extraction-quickstart)。
44
+ - **开发完成自动评审(含窄修复和测试代码)**:实现者先自检分支/失败路径及 security/privacy/authority/数据丢失风险,按 `testing-strategy` 完成适用测试,再自动调用 `code-review`,无需用户提醒;执行与收尾见 `skills/code-review/references/development-completion.md`。自审、读技能或说“下一步评审”都不算独立评审。按风险定深度;窄任务不额外套 product-rd self-review row,既有高风险/shared-skill gate 不降级。评审覆盖实际 diff,采用对抗问题,不要求确认实现者结论;findings 先核实再修复或有证据处置,避免循环追逐建议。用户明确跳过时记录 skipped;当前候选已有有效独立评审则复用。详见 product-rd 验证门 + skill-extraction `dual-track-review-gate.md`。
45
45
  </ccl-skills-routing>
@@ -5,7 +5,7 @@ description: Use when designing, implementing, reviewing, debugging, testing, or
5
5
 
6
6
  # App Cross-Platform Dev
7
7
 
8
- Use this skill for mobile app and cross-platform client engineering. It covers Flutter, React Native, native Android, and native iOS. It does not own mini-programs, React web, backend service design, product requirements, or visual design rules.
8
+ Use this skill for mobile app and cross-platform client engineering. It covers Flutter, React Native, native Android, and native iOS. It does not own mini-programs, React web, backend service design, product requirements, or visual design rules. After code/test edits, self-check and invoke `code-review` automatically before completion.
9
9
 
10
10
  ## Routing
11
11
 
@@ -15,7 +15,7 @@ Use this skill for mobile app and cross-platform client engineering. It covers F
15
15
  - Use `web-react-dev` for React web and browser-specific client work.
16
16
  - Use Go or Python backend skills for server contracts, persistence, queues, auth services, and API ownership.
17
17
  - Use `testing-strategy` to choose the test layer; return here for Flutter, React Native, Android, or iOS implementation details.
18
- - Use `test-artifact-management` when the ask is about generating structured test cases from a Feishu requirements doc or codebase and tracking them in Feishu Bitable before implementation begins.
18
+ - Use `test-artifact-management` for structured Feishu Bitable cases derived from Feishu requirements or code before implementation.
19
19
  - Use `defect-diagnosis` first for bugs, failed tests, flaky behavior, crashes, rendering regressions, build failures, or store/release symptoms.
20
20
  - For money, quota, permission, tenant/user data, high-impact AI, repeated submit, async finality, or support-traceable incidents, apply `product-rd-workflow` high-risk resilience gates before treating the app flow as complete.
21
21
 
@@ -1,11 +1,11 @@
1
1
  ---
2
2
  name: code-review
3
- description: Use when an implementation needs an independent CLI reviewer or adversarial challenger, routing across Claude Code, Kimi, OpenCode, or Codex according to local availability, model-family independence, and the user's client order; also covers Claude-only bounded consultation and requests such as code review, Claude review, Kimi review, OpenCode review, second opinion, 找茬, 唱反调, or 第二意见. Skip 某个改动/提交还有没有价值、值不值得修这类交付裁决 → product-rd-workflow:本技能执行评审、找缺陷与风险,不裁决交付价值。
3
+ description: Use automatically after changing code or executable tests, before completion or landing handoff, even without a user review request. Obtain independent CLI review/challenge across Claude Code, Kimi, OpenCode, or Codex by capability, model-family independence, and the user's client order; also covers Claude-only consultation, code review, Claude review, Kimi review, OpenCode review, second opinion, 找茬, 唱反调, or 第二意见. Skip 改动/提交值不值得修这类交付裁决 → product-rd-workflow;本技能执行评审,不裁决交付价值。
4
4
  ---
5
5
 
6
6
  # Code Review
7
7
 
8
- Use this skill from Claude, Codex, OpenCode, or another compatible agent to obtain an independent, attributable review or adversarial challenge. The job is not to delegate implementation. The caller supplies the model family that produced the candidate so the router can exclude same-family reviewers.
8
+ After changing code or executable tests, invoke this skill automatically before completion or landing handoff; follow `references/development-completion.md`. Compatible hosts obtain independent review/challenge here, never delegate implementation. The caller supplies the implementer's model family so the router excludes same-family reviewers.
9
9
 
10
10
  ## Modes
11
11
 
@@ -13,7 +13,7 @@ Choose the smallest useful mode:
13
13
 
14
14
  - **Review mode**: normal pre-merge or plan review. Find blocking or materially misleading issues.
15
15
  - **Challenge mode**: adversarial pass modeled after `codex challenge`. Try to break the diff or decision by finding production failure paths.
16
- - **Complete mode**: local exact-candidate deep-self-review checkpoint after a passed final round. It invokes no reviewer and grants no human or merge authority.
16
+ - **Complete mode**: local exact-candidate checkpoint after passing review or bound source refutations under the staged contract. It invokes no reviewer and grants no merge authority.
17
17
  - **Consult mode**: ask Claude a bounded question when no diff or plan review is needed.
18
18
 
19
19
  Use challenge mode when the user asks for "challenge", "poke holes", "try to break it", "adversarial", or when the change touches money, permissions, privacy, compliance, tenant/user data, production rollout, high-impact AI, architecture, economics, or IA.
@@ -74,7 +74,7 @@ Positive challenge capacity opens it at index 1; budget zero is untracked.
74
74
  The sole release/high-risk budget-zero exception is a controller-proved
75
75
  `markdown-punctuation-only` review: it requires `wording_only_boundary`, permits
76
76
  no `complete`, and rejects an author assertion alone (recipe:
77
- `references/staged-review-contract.md`). After a clean tracked
77
+ `references/staged-review-contract.md`). After a clean/source-refuted tracked
78
78
  challenge, `complete` may close early and preserve unused rounds. Every result
79
79
  exposes controller-owned `self_review_gate`; an outstanding checkpoint blocks
80
80
  only external review or completion, not implementation or tests. Even a passed
@@ -215,10 +215,11 @@ Run the script by path while keeping `--cwd` pointed at the product repository u
215
215
 
216
216
  **The packet is the reviewer's whole world — compose it deliberately.** Review and challenge are built packet-bounded — Claude runs `--tools ""` with no `--add-dir`, and the other wrappers run in an isolated run workspace or a packet-only read surface. Treat the packet as the reviewer's whole world when deciding coverage: it is the only content bound by the packet hash and scanned before egress, so anything outside it is neither reliably visible to the reviewer nor covered by the verdict; a diff-only packet surfaces defects visible inside the changed lines and little else, and `--paths` only narrows it further. Whatever is absent from the packet is unreachable, not merely missed: a contradiction with an unchanged sibling clause, drift against a carrier outside the diff, or a silent weakening of upstream wording cannot be found by a reviewer who never saw the other side — that is the packet's shape, not the reviewer's weakness.
217
217
 
218
+ - Codex permits frozen-packet read/search; see [tool boundaries](references/development-completion.md#review-tools).
218
219
  - To widen the packet, assemble it yourself and pass `--diff-file`: it replaces base-derived generation, is mutually exclusive with `--base`/`--paths`, and must name a regular file (no symlink or hardlink) holding text without NUL bytes. Worth adding beyond the diff — the canonical rule or contract text the changed lines must not contradict, the sibling clauses in the same file, the derived carriers that restate the change (commit message, MR/PR body), and the actual output of a gate or script under review. The gate hard-caps a packet at 200,000 bytes; split a larger candidate as described in the next bullet.
219
220
  - A verdict covers exactly the packet it was taken on, because the recorded packet hash is the reviewed identity. Within a packet, added context sits on top of the candidate diff and never in place of part of it. A candidate too large for one packet is split by file group or risk class into a partition that still covers the whole candidate — every part in some packet, none dropped — each partition's verdict recorded against its own packet hash, and the candidate-wide claim withheld until every partition is conclusive; one partition's `no blocking findings` is never a verdict on the landing candidate. Cross-partition contradictions are unreachable by construction, so repeat the shared canonical context in every partition's packet and review anything that spans partitions as its own packet.
220
221
  - Added context egresses to the selected reviewer exactly like the diff does, through the same credential tripwire — which catches machine-detectable secrets only. Paste rule text, carriers, and tool output; never paste credentials or material you would not send to that provider.
221
- - A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane rather than editing the candidate to satisfy it. A reviewer reporting that the input is insufficient to judge the change is reporting an input defect — add the missing context and rerun that lane, do not edit the candidate to satisfy it.
222
+ - A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane rather than editing the candidate to satisfy it.
222
223
 
223
224
  When intentionally reviewing `code-review` itself, override the resolver from the ccl-skills repo under review before invoking the gate:
224
225
 
@@ -0,0 +1,26 @@
1
+ # Automatic review after development
2
+
3
+ This transition applies across implementation owners, including narrow fixes and executable-test changes. Do not wait for the user to request review. Use the actual diff to classify the work: an implementation diff triggers this transition regardless of the task label. Read-only investigation, status answers and design-only work retain their owning workflow; they do not acquire an implementation-review requirement merely by using a development skill.
4
+
5
+ ## Before invoking review
6
+
7
+ 1. Recover the current task, actual diff, scope and authorization. Changing the implementation or scope reopens this check; an earlier plan review cannot discharge review of the resulting code.
8
+ 2. Finish proportionate implementer self-review and affected verification. A failed quality check calls for available in-scope diagnosis and cleanup under [refactoring discipline](../../product-rd-workflow/references/refactoring-discipline.md#responding-to-quality-gates); preserve behavior and readability, rerun the check, and escalate only a remaining real blocker. Use `testing-strategy` to select tests: changed named test properties require the killing-mutation walk on disposable or restore-guarded resources; when the same contract has two implementations or paths, use differential/equivalence checks with bounded, asserted known differences. Record a concrete applicability or unavailable-evidence reason when a test family does not run. Do not force a full mutation framework or differential suite onto an unrelated change.
9
+ 3. Reuse a terminal independent review only when it covers the current candidate and satisfies the applicable owner gate, reviewer independence and required depth. Record the receipt location and its candidate identifier (commit or diff/packet digest), then compare with the current candidate using the owning gate's binding rules; HEAD alone cannot cover uncommitted edits. A different or missing candidate identifier cannot discharge review. A loaded skill, self-review, planned command, unfinished handle or unverified prose claim is not that evidence. A native subagent result counts only when the owning gate accepts its independence and evidence; it never silently substitutes for a required CLI receipt.
10
+ 4. An explicit user instruction to skip review controls this task: record `skipped`, not `passed`, and preserve any separate landing restrictions. Record its original wording and current scope; a superseded or unrelated instruction is not a skip for this task. Do not ask for confirmation of ordinary review already within the authorized development task. User client restrictions and existing confidentiality boundaries still control reviewer selection; capability matters, not a numeric CLI or skill version.
11
+
12
+ ## Invoke and finish
13
+
14
+ When no valid current review discharges the requirement, invoke `scripts/review_gate.sh` from this skill's actual installed/source directory with the current candidate, real implementer family, user client order and applicable risk tags. Follow the entrypoint's script contract; narrow work may use its derived-default plan. Use ordinary review for ordinary development; challenge and additional owner gates apply when triggered. Do not inflate a narrow repair into a product-design or shared-skill review ceremony.
15
+
16
+ Invoke the reviewer in the same turn once self-checks are ready. Await an existing handle to its terminal result; do not stop at “review next,” start a duplicate process, or present timeout, invalid output or authentication failure as pass. Handle operational failures using the existing bounded recovery rules; a stopped reviewer lane does not stop safe independent work or authorize completion.
17
+
18
+ Verify each finding against the actual call path and evidence. Fix confirmed defects and rerun affected checks; record evidenced rejection, deferral or acceptance under the owning gate. Pre-existing issues and optional suggestions do not automatically expand the task. After a tracked review and challenge on the unchanged candidate, when every finding is source-refuted, run the local disposition completion path in [the staged contract](staged-review-contract.md#mechanical-self-review-gate). Keep the raw findings; do not rerun merely to obtain zero findings or reset a review budget. Budget exhaustion limits reviewer calls, so finish available disposition, self-review and validation work in the same turn. A changed candidate still needs the owning gate's renewed review.
19
+
20
+ Before completion or landing handoff, report the actual diff classification and a concrete reason if review is inapplicable. Support that classification with the change-inspection command and result, including untracked implementation files. Report the actual review outcome and candidate it covers, relevant tests and their results, skips, unresolved findings and remaining restrictions. A failed or missing required review leaves review/completion pending. Review does not grant permission to commit, push, merge, publish or deploy.
21
+
22
+ ## Review tools
23
+
24
+ Packet-only allows read-only tools over the frozen material. Codex exposes pathless `read_packet` and `search_packet`; the parser checks returned content against the same packet and requires completed calls. Each server read checks the packet hash. Shell execution is disabled by effective capability, and inherited MCP servers are disabled only for this invocation. Unsupported capability falls back without a CLI version requirement. A read-only sandbox alone does not disable commands.
25
+
26
+ Installed owner skills supply review lenses, not permission to execute development workflows or fetch references outside the packet. Add missing source context to the packet through the entrypoint's packet-composition contract. Read/search does not expand the verdict's coverage beyond its recorded packet.
@@ -8,10 +8,10 @@ The controller has three modes:
8
8
 
9
9
  Explore/build may configure `challenge_budget=0..4`; release/high-risk requires
10
10
  at least one challenge unless the exact candidate qualifies for the
11
- proof-bound wording-only single-review exception below. The initial review consumes Agent round 1, so total
12
- Agent-autonomous external review is at most five rounds. Human-requested review
13
- is outside this budget and must be attributed by the consuming trusted platform.
14
- The budget is a ceiling, not a quota: after a clean tracked challenge, local
11
+ proof-bound wording-only single-review exception below. The initial review consumes chain round 1;
12
+ each bounded chain uses at most five rounds. Necessary task-scoped review after a
13
+ checkpoint inherits existing task authority; attribute explicit human round requests separately.
14
+ The budget is a ceiling, not a quota: after a clean or fully source-refuted tracked challenge, local
15
15
  `complete` may close the chain early. It preserves unused-round count but sets
16
16
  `autonomous_review_allowed=false`; release/high-risk still requires at least one
17
17
  challenge before this early close is eligible.
@@ -283,7 +283,7 @@ Multi-round Agent automation supplies `review_chain_id`, a contiguous
283
283
  `autonomous_review_index` in `1..5`, and every earlier result through ordered
284
284
  `--prior-review-result-file` arguments. Prior rounds may contain findings and
285
285
  older candidate hashes; they remain consumed. Candidate edits, commits, plan
286
- refreshes, mode changes, and renamed invocations do not reset Agent authority.
286
+ refreshes, mode changes, and renamed invocations never erase spending or broaden task authority.
287
287
  An initial `review` with positive challenge capacity must start this chain at
288
288
  index 1; an untracked initial review is single-round and therefore uses budget 0.
289
289
 
@@ -324,7 +324,7 @@ Envelope `schema_version` is `3`; a legacy `2` envelope predates the recorded
324
324
  scope and is rejected, which requires restarting an in-flight chain. Every
325
325
  prior round must retain the same controller digest, owner-selection source,
326
326
  selected owner names, and selected-owner digest. Missing, substituted,
327
- inconclusive, reordered, renamed-chain, or fourth-round input fails before any
327
+ inconclusive, reordered, renamed-chain, or over-budget input fails before any
328
328
  provider runs. Scope drift returns `review_scope_changed` and requires deep
329
329
  self-review plus explicit task reframing; it does not silently create a new
330
330
  Agent budget. An untracked challenge is one-off advisory evidence; it cannot
@@ -332,8 +332,8 @@ enter a later Agent round or satisfy the local completion checkpoint.
332
332
 
333
333
  Two consequences follow from those stable bindings and must be planned for before round 1:
334
334
 
335
- - The selected-owner digest hashes each selected owner package's current working tree, and owners derive from the candidate's own paths — so a candidate edit inside any selected owner package invalidates every prior receipt and the next tracked round fails `review_chain_invalid`. For a self-hosted candidate (a skill-repo diff editing the package that owns it) that is nearly every applied fix — one confined to files outside every selected owner drifts only the candidate hash and may continue in-chain: "do not reset Agent authority" promises no continuation, and the in-chain tolerance for older candidate hashes is reachable only while the fix stays outside its selected owners.
336
- - A chain restarted after such a break re-enters the same cumulative Agent budget and must never be counted as fresh authority; the bounded restart recipe for the extraction lane (batched dispositions, cross-chain round accounting, full-context first packet, terminal disposition at the cap) is owned by the extraction workflow's dual-track gate reference.
335
+ - The selected-owner digest hashes each selected owner package's current working tree, and owners derive from the candidate's own paths — so a candidate edit inside any selected owner package invalidates every prior receipt and the next tracked round fails `review_chain_invalid`. For a self-hosted candidate (a skill-repo diff editing the package that owns it) that is nearly every applied fix — one confined to files outside every selected owner drifts only the candidate hash and may continue in-chain: in-chain tolerance for older candidate hashes applies only while fixes stay outside selected owners. Necessary recovery uses fresh bindings after the task checkpoint below.
336
+ - A chain restarted after such a break never erases cumulative spending or task history. An existing task includes necessary fixes, tests and review by default. At exhaustion, first disposition findings from source, run deep self-review and tests, and change the failed method or add missing evidence before another necessary bounded sequence. Record `continuation_basis=existing-task-scope`, the original authority reference and scope, the reason and changed method/evidence, cumulative rounds, and old/new sequence links in the caller-owned task artifact. Preserve every earlier receipt, focus and disposition; do not add this field to CLI arguments or runtime receipts. Existing per-chain and consuming-owner sequence bounds still apply; the extraction recipe remains in its dual-track gate reference. No repeated calls solely to obtain an empty verdict, invented human round requests, history reset or ignored user cost/round/stop limit is allowed.
337
337
 
338
338
  The controller is stateless and prevents accidental/cooperative resets only. A
339
339
  trusted host or platform must retain the ledger when hostile local callers are in
@@ -370,10 +370,34 @@ allowing refreshed self-review conclusions and evidence, and is the Agent path t
370
370
  round index, prior-result hashes, and prior challenge focuses. It is not a human
371
371
  waiver or merge authorization.
372
372
 
373
+ For an unchanged candidate with a conclusive tracked review and challenge,
374
+ `complete` also accepts `--finding-dispositions-file`. Supply every earlier raw
375
+ receipt with ordered `--prior-review-result-file` arguments and the final one
376
+ with `--completion-review-result-file`. The UTF-8 JSON has `schema_version: 1`,
377
+ `candidate_sha256`, ordered `review_result_sha256` hashes including the final
378
+ receipt, and `dispositions`. Each disposition contains `receipt_sha256`, the
379
+ canonical-JSON `finding_sha256`, `disposition: "source_refuted"`, and a non-empty
380
+ `evidence` array naming the first-hand source or failure-path counter-evidence.
381
+ Every original finding occurrence must appear exactly once. Within one receipt,
382
+ findings with identical canonical content share one hash-pair identity in
383
+ first-seen order; retain every raw entry unchanged. Different content or a
384
+ different receipt remains a separate occurrence. Missing history,
385
+ changed candidates, inconclusive results, open findings and risk acceptance do
386
+ not qualify. A code fix with changed bytes requires renewed review.
387
+
388
+ This local checkpoint records `completion_basis=source_refuted_findings`, the
389
+ dispositions digest and resolved occurrence bindings; original external
390
+ findings remain unchanged. A clean external result uses `external_pass`.
391
+ Validation proves binding and coverage, not the truth of an evidence statement:
392
+ the implementer must trace the cited source, and the judgment remains open to
393
+ challenge. No budget is refreshed and no model is called. At the review cap,
394
+ finish this local work instead of requesting another round solely to obtain an
395
+ empty verdict. Unresolved findings still block completion, not independent work.
396
+
373
397
  ## Human and failure boundary
374
398
 
375
- Only an external authenticated platform action may prove human request, stop,
376
- resume, waiver, commit, or merge authority. A `review_waiver` clears only the
399
+ Existing task scope covers necessary continuation; exhaustion alone creates no new grant.
400
+ Only an external authenticated platform action may prove new human request, stop, resume, waiver, commit, or merge authority. A `review_waiver` clears only the
377
401
  review-process gate. A distinct exact-candidate `merge_authorization` is the
378
402
  human's final decision: CI may keep running and reporting every failed/pending
379
403
  gate, but none may block that authorized merge. Report
@@ -381,9 +405,9 @@ gate, but none may block that authorized merge. Report
381
405
 
382
406
  Provider/input/integrity failures stop that reviewer lane, not the whole task.
383
407
  Their stable action is `stop_reviewer_lane`, never the ambiguous `stop`.
384
- Budget exhaustion likewise stops only automatic reviewer calls. Continue local
385
- fixes, self-review, tests, and independent work; park only decision-dependent
386
- work. Enter `awaiting_human` only when no independent runnable work remains.
408
+ Budget exhaustion triggers the method/authority checkpoint, not an automatic user handoff. Check legacy
409
+ `human_decision_required` / `continuation_authorization_required` against existing task scope first.
410
+ Continue necessary bounded review; enter `awaiting_human` only for a genuine missing decision, explicit user limit or authority outside that scope.
387
411
 
388
412
  The current result envelope is schema 3. The generic per-invocation `--timeout`
389
413
  keeps its 600-second default and accepts 5..1200 seconds; direct wrappers clamp
@@ -5,8 +5,11 @@ parsers for Claude, Kimi, OpenCode, and Codex.
5
5
 
6
6
  Rules:
7
7
 
8
- - Preserve no-tools posture and structured output validation. A malformed,
9
- timeout, or inconclusive wrapper result is not a pass.
8
+ - Preserve the packet boundary and structured output validation. A wrapper may
9
+ expose pathless read/search tools over its frozen, hash-bound packet; verify
10
+ their arguments, returned bytes and completed lifecycle. This does not permit
11
+ arbitrary commands or workspace access. A malformed, timeout, or inconclusive
12
+ wrapper result is not a pass.
10
13
  - **Never pin the parser to a CLI version's vocabulary.** `parse_probe_result.py`
11
14
  gates on *shape*, not on field/value names: the isolation proof is the exact
12
15
  `tools` allowlist plus the tool_use scan, which no init field can bypass.
@@ -140,10 +140,15 @@ CODEX_EXEC_HELP="$(timeout --kill-after=1s 5s "$CODEX_BIN_PATH" exec --disable h
140
140
  # supported lifecycle state: `features list` still prints removed keys, and a
141
141
  # removed or unknown-state row means `--disable hooks` may be a silent no-op
142
142
  # that lets user-trusted hooks run during packet-only review.
143
- CODEX_FEATURES_LIST="$(timeout --kill-after=1s 5s "$CODEX_BIN_PATH" features list 2>/dev/null)" \
143
+ CODEX_FEATURES_LIST="$(timeout --kill-after=1s 5s "$CODEX_BIN_PATH" features list --disable hooks --disable shell_tool 2>/dev/null)" \
144
144
  || die_inconclusive codex_hook_disable_unavailable capability_missing true
145
- grep -Eq '^hooks[[:space:]]+(stable|under development|experimental)([[:space:]]|$)' <<<"$CODEX_FEATURES_LIST" \
145
+ grep -Eq '^hooks[[:space:]]+(stable|under development|experimental)[[:space:]]+false[[:space:]]*$' <<<"$CODEX_FEATURES_LIST" \
146
146
  || die_inconclusive codex_hook_disable_unavailable capability_missing true
147
+ # A read-only sandbox still exposes command execution. Disable the actual
148
+ # shell surface, including its unified-exec implementation, before inference.
149
+ # Check effective capability rather than a CLI release number or flag parsing.
150
+ grep -Eq '^shell_tool[[:space:]]+(stable|under development|experimental)[[:space:]]+false[[:space:]]*$' <<<"$CODEX_FEATURES_LIST" \
151
+ || die_inconclusive codex_shell_disable_unavailable capability_missing true
147
152
  if [ -n "${CODEX_HOME:-}" ]; then
148
153
  SOURCE_HOME="$CODEX_HOME"
149
154
  else
@@ -240,6 +245,72 @@ fi
240
245
  DIFF_TEXT="$(cat "$DIFF_FILE" || exit 1; printf '\001')" \
241
246
  || die_inconclusive diff_read_failed local_tool_failure false
242
247
  DIFF_TEXT="${DIFF_TEXT%$'\001'}"
248
+ PACKET_FILE="$RUN_ROOT/packet.txt"
249
+ PACKET_SERVER="$SCRIPT_DIR/kimi_packet_mcp.py"
250
+ [ -f "$PACKET_SERVER" ] && [ -r "$PACKET_SERVER" ] && [ ! -L "$PACKET_SERVER" ] \
251
+ || die_inconclusive packet_server_missing local_tool_failure false
252
+ printf '%s' "$DIFF_TEXT" >"$PACKET_FILE"
253
+ PACKET_SHA256="$(python3 - "$PACKET_FILE" <<'PY_HASH'
254
+ import hashlib, sys
255
+ from pathlib import Path
256
+ print(hashlib.sha256(Path(sys.argv[1]).read_bytes()).hexdigest())
257
+ PY_HASH
258
+ )" || die_inconclusive packet_hash_failed local_tool_failure false
259
+ MCP_CONFIG="$(python3 - "$PACKET_FILE" "$PACKET_SERVER" "$PACKET_SHA256" <<'PY_MCP_CONFIG'
260
+ import json, sys
261
+ values = [sys.argv[2], "--packet", sys.argv[1], "--sha256", sys.argv[3], "--allow-search"]
262
+ print('mcp_servers={code_review_packet={command=' + json.dumps(sys.executable)
263
+ + ',args=[' + ','.join(json.dumps(value) for value in values)
264
+ + '],enabled=true,enabled_tools=["read_packet","search_packet"]}}')
265
+ PY_MCP_CONFIG
266
+ )" || die_inconclusive packet_config_failed local_tool_failure false
267
+ # TOML overrides merge server tables. Disable inherited servers for this run
268
+ # without changing user configuration, then verify the effective public list.
269
+ CODEX_PACKET_CONFIG=(-c "$MCP_CONFIG" -c 'web_search="disabled"' -c 'approval_policy="never"')
270
+ CODEX_HOME="$SOURCE_HOME" timeout --kill-after=1s 5s "$CODEX_BIN_PATH" mcp list --json "${CODEX_PACKET_CONFIG[@]}" >"$RUN_ROOT/mcp.json" 2>"$STDERR_FILE" \
271
+ || die_inconclusive codex_packet_tools_unavailable capability_missing true
272
+ MCP_CONFIG="$(python3 - "$RUN_ROOT/mcp.json" "$MCP_CONFIG" <<'PY_MCP_OVERRIDES'
273
+ import json, sys
274
+ from pathlib import Path
275
+ try:
276
+ rows = json.loads(Path(sys.argv[1]).read_text())
277
+ if not isinstance(rows, list):
278
+ raise ValueError()
279
+ disabled = []
280
+ for row in rows:
281
+ if not isinstance(row, dict) or not isinstance(row.get("name"), str) or not row["name"]:
282
+ raise ValueError()
283
+ if row["name"] != "code_review_packet":
284
+ disabled.append(json.dumps(row["name"]) + "={enabled=false}")
285
+ print(sys.argv[2][:-1] + "".join("," + entry for entry in disabled) + "}")
286
+ except (OSError, ValueError, TypeError):
287
+ sys.exit(1)
288
+ PY_MCP_OVERRIDES
289
+ )" || die_inconclusive codex_packet_tools_unavailable capability_missing true
290
+ CODEX_PACKET_CONFIG=(-c "$MCP_CONFIG" -c 'web_search="disabled"' -c 'approval_policy="never"')
291
+ CODEX_HOME="$SOURCE_HOME" timeout --kill-after=1s 5s "$CODEX_BIN_PATH" mcp list --json "${CODEX_PACKET_CONFIG[@]}" >"$RUN_ROOT/mcp.json" 2>"$STDERR_FILE" \
292
+ || die_inconclusive codex_packet_tools_unavailable capability_missing true
293
+ python3 - "$RUN_ROOT/mcp.json" "$PACKET_FILE" "$PACKET_SERVER" "$PACKET_SHA256" <<'PY_MCP_CHECK' \
294
+ || die_inconclusive codex_packet_tools_unavailable capability_missing true
295
+ import json, sys
296
+ from pathlib import Path
297
+ try:
298
+ rows = json.loads(Path(sys.argv[1]).read_text())
299
+ if not isinstance(rows, list):
300
+ raise ValueError()
301
+ active = [row for row in rows if isinstance(row, dict) and row.get("enabled") is not False]
302
+ if len(active) != 1 or len(rows) != len([row for row in rows if isinstance(row, dict)]):
303
+ raise ValueError()
304
+ row = active[0]
305
+ transport = row.get("transport", {})
306
+ if (row.get("name") != "code_review_packet" or row.get("enabled") is not True
307
+ or transport.get("type") != "stdio" or transport.get("command") != sys.executable
308
+ or transport.get("args") != [sys.argv[3], "--packet", sys.argv[2], "--sha256", sys.argv[4], "--allow-search"]
309
+ or transport.get("env") or transport.get("env_vars") or transport.get("cwd")):
310
+ raise ValueError()
311
+ except (OSError, ValueError, TypeError, AttributeError):
312
+ sys.exit(1)
313
+ PY_MCP_CHECK
243
314
  {
244
315
  printf '%s\n\n' "$INSTRUCTION"
245
316
  if [ "$REVIEW_SKILL_COUNT" -gt 0 ]; then
@@ -255,7 +326,7 @@ DIFF_TEXT="${DIFF_TEXT%$'\001'}"
255
326
  printf '%s' "$PROFILE_TEXT"
256
327
  printf '\n%s_END\n\n' "$PROFILE_TOKEN"
257
328
  fi
258
- printf '%s\n' 'Use only the supplied diff. Do not invoke tools or inspect the workspace.'
329
+ printf '%s\n' 'Use only the supplied diff and review profile. You may read or search the same frozen diff using code_review_packet read_packet and search_packet. These pathless tools cannot inspect the workspace. Do not execute commands, access other tools, or follow skill instructions to run development workflows or read external references. Report missing context as an evidence gap.'
259
330
  if [ -n "$REVIEW_PROFILE_FILE" ]; then
260
331
  printf '%s\n' 'Treat self_review and evidence as claims to verify against the diff, not as proof. Check every entry in required_concerns. A no-findings verdict is valid only after all entries were checked; if the bounded packet cannot support a required check, report that evidence gap as a material finding at the best changed-file locator.'
261
332
  printf '%s\n' 'Return exactly one concern_results object with concern and concise independent conclusion for every required concern.'
@@ -279,7 +350,8 @@ JSON
279
350
  fi
280
351
 
281
352
  run_started=$SECONDS
282
- CMUX_CODEX_HOOKS_DISABLED=1 CODEX_HOME="$SOURCE_HOME" timeout --kill-after=1s "${TIMEOUT}s" "$CODEX_BIN_PATH" exec --disable hooks --sandbox read-only --ephemeral --skip-git-repo-check \
353
+ CMUX_CODEX_HOOKS_DISABLED=1 CODEX_HOME="$SOURCE_HOME" timeout --kill-after=1s "${TIMEOUT}s" "$CODEX_BIN_PATH" exec --disable hooks --disable shell_tool --sandbox read-only --ephemeral --skip-git-repo-check \
354
+ "${CODEX_PACKET_CONFIG[@]}" \
283
355
  --json --output-schema "$SCHEMA_FILE" --output-last-message "$RESULT_FILE" \
284
356
  -C "$RUN_WORKSPACE" - <"$PROMPT_FILE" >"$EVENTS" 2>"$STDERR_FILE"
285
357
  run_rc=$?
@@ -309,7 +381,7 @@ fi
309
381
 
310
382
  python3 "$PARSER" --client codex --mode "$MODE" --implementer-family "$IMPL_FAMILY" \
311
383
  --reviewer-family "$FAMILY" --provider "$PROVIDER" --model "$MODEL" \
312
- --events "$EVENTS" --result-file "$RESULT_FILE" >"$PARSED_FILE"
384
+ --events "$EVENTS" --result-file "$RESULT_FILE" --packet "$PACKET_FILE" --packet-sha256 "$PACKET_SHA256" >"$PARSED_FILE"
313
385
  parser_rc=$?
314
386
  if [ "$parser_rc" -eq 0 ]; then
315
387
  native_skill_binding="not_requested"
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env python3
2
- """Expose one SHA-256-bound review packet through a pathless stdio MCP tool."""
2
+ """Expose one SHA-256-bound review packet through pathless stdio MCP tools."""
3
3
 
4
4
  from __future__ import annotations
5
5
 
@@ -12,8 +12,11 @@ from typing import Any
12
12
 
13
13
 
14
14
  TOOL_NAME = "read_packet"
15
+ SEARCH_TOOL_NAME = "search_packet"
15
16
  MAX_PAGE_BYTES = 48_000
16
17
  MAX_CHUNK_BYTES = 46_000
18
+ MAX_QUERY_BYTES = 1_024
19
+ MAX_SEARCH_RESULTS = 100
17
20
 
18
21
 
19
22
  def packet_bytes(path: Path, expected_sha256: str) -> bytes:
@@ -57,6 +60,84 @@ def tool_result(text: str, *, error: bool = False) -> dict[str, Any]:
57
60
  return {"content": [{"type": "text", "text": text}], "isError": error}
58
61
 
59
62
 
63
+ def search_tool_definition() -> dict[str, Any]:
64
+ return {
65
+ "name": SEARCH_TOOL_NAME,
66
+ "description": (
67
+ "Find non-overlapping, case-sensitive literal substrings in the one "
68
+ "controller-frozen review packet. No regex or filesystem paths. "
69
+ "Query is at most 1024 UTF-8 bytes; byte_offset must be a UTF-8 "
70
+ "boundary. Returns byte offsets and 1-based line numbers, with "
71
+ "next_byte_offset for more matches or null when exhausted. "
72
+ "Use read_packet at a match offset to inspect its context."
73
+ ),
74
+ "inputSchema": {
75
+ "type": "object",
76
+ "properties": {
77
+ "query": {"type": "string", "minLength": 1, "maxLength": MAX_QUERY_BYTES},
78
+ "byte_offset": {"type": "integer", "minimum": 0},
79
+ "limit": {"type": "integer", "minimum": 1, "maximum": MAX_SEARCH_RESULTS},
80
+ },
81
+ "required": ["query", "byte_offset", "limit"],
82
+ "additionalProperties": False,
83
+ },
84
+ }
85
+
86
+
87
+ def search_packet(path: Path, expected_sha256: str, arguments: Any) -> dict[str, Any]:
88
+ if not isinstance(arguments, dict) or set(arguments) != {"query", "byte_offset", "limit"}:
89
+ return tool_result("invalid packet search arguments", error=True)
90
+ query = arguments.get("query")
91
+ byte_offset = arguments.get("byte_offset")
92
+ limit = arguments.get("limit")
93
+ if (
94
+ not isinstance(query, str)
95
+ or not isinstance(byte_offset, int)
96
+ or isinstance(byte_offset, bool)
97
+ or byte_offset < 0
98
+ or not isinstance(limit, int)
99
+ or isinstance(limit, bool)
100
+ or not 1 <= limit <= MAX_SEARCH_RESULTS
101
+ ):
102
+ return tool_result("invalid packet search arguments", error=True)
103
+ try:
104
+ needle = query.encode("utf-8")
105
+ except UnicodeEncodeError:
106
+ return tool_result("invalid packet search arguments", error=True)
107
+ if not 1 <= len(needle) <= MAX_QUERY_BYTES:
108
+ return tool_result("invalid packet search arguments", error=True)
109
+ try:
110
+ data = packet_bytes(path, expected_sha256)
111
+ except (OSError, ValueError):
112
+ return tool_result("packet binding changed", error=True)
113
+ if byte_offset > len(data):
114
+ return tool_result("packet search starts beyond end", error=True)
115
+ try:
116
+ data[:byte_offset].decode("utf-8")
117
+ except UnicodeDecodeError:
118
+ return tool_result("packet search starts inside a UTF-8 character", error=True)
119
+
120
+ matches = []
121
+ cursor = byte_offset
122
+ while len(matches) < limit:
123
+ found = data.find(needle, cursor)
124
+ if found < 0:
125
+ break
126
+ matches.append({"byte_offset": found, "line": data.count(b"\n", 0, found) + 1})
127
+ cursor = found + len(needle)
128
+ rendered = json.dumps(
129
+ {
130
+ "matches": matches,
131
+ "next_byte_offset": cursor if data.find(needle, cursor) >= 0 else None,
132
+ "total_bytes": len(data),
133
+ },
134
+ separators=(",", ":"),
135
+ )
136
+ if len(rendered.encode("utf-8")) > MAX_PAGE_BYTES:
137
+ return tool_result("packet search exceeds result bound", error=True)
138
+ return tool_result(rendered)
139
+
140
+
60
141
  def read_chunk(path: Path, expected_sha256: str, arguments: Any) -> dict[str, Any]:
61
142
  if not isinstance(arguments, dict) or set(arguments) != {"byte_offset", "max_bytes"}:
62
143
  return tool_result("invalid packet chunk arguments", error=True)
@@ -119,7 +200,9 @@ def response(request_id: Any, *, result: Any = None, error: Any = None) -> dict[
119
200
  return payload
120
201
 
121
202
 
122
- def handle(message: Any, path: Path, expected_sha256: str) -> dict[str, Any] | None:
203
+ def handle(
204
+ message: Any, path: Path, expected_sha256: str, *, allow_search: bool = False
205
+ ) -> dict[str, Any] | None:
123
206
  if not isinstance(message, dict) or message.get("jsonrpc") != "2.0":
124
207
  return response(None, error={"code": -32600, "message": "Invalid Request"})
125
208
  request_id = message.get("id")
@@ -140,9 +223,17 @@ def handle(message: Any, path: Path, expected_sha256: str) -> dict[str, Any] | N
140
223
  if method == "ping":
141
224
  return response(request_id, result={})
142
225
  if method == "tools/list":
143
- return response(request_id, result={"tools": [tool_definition()]})
226
+ definitions = [tool_definition()]
227
+ if allow_search:
228
+ definitions.append(search_tool_definition())
229
+ return response(request_id, result={"tools": definitions})
144
230
  if method == "tools/call":
145
231
  params = message.get("params")
232
+ if isinstance(params, dict) and allow_search and params.get("name") == SEARCH_TOOL_NAME:
233
+ return response(
234
+ request_id,
235
+ result=search_packet(path, expected_sha256, params.get("arguments")),
236
+ )
146
237
  if not isinstance(params, dict) or params.get("name") != TOOL_NAME:
147
238
  return response(request_id, result=tool_result("unknown tool", error=True))
148
239
  return response(
@@ -156,6 +247,9 @@ def main() -> int:
156
247
  parser = argparse.ArgumentParser(description=__doc__)
157
248
  parser.add_argument("--packet", required=True)
158
249
  parser.add_argument("--sha256", required=True)
250
+ parser.add_argument(
251
+ "--allow-search", action="store_true", help="also expose bounded literal packet search"
252
+ )
159
253
  args = parser.parse_args()
160
254
  if not len(args.sha256) == 64 or any(ch not in "0123456789abcdef" for ch in args.sha256):
161
255
  parser.error("--sha256 must be a lowercase SHA-256 digest")
@@ -169,7 +263,7 @@ def main() -> int:
169
263
  for raw_line in sys.stdin:
170
264
  try:
171
265
  message = json.loads(raw_line)
172
- payload = handle(message, path, args.sha256)
266
+ payload = handle(message, path, args.sha256, allow_search=args.allow_search)
173
267
  except (json.JSONDecodeError, UnicodeError):
174
268
  payload = response(None, error={"code": -32700, "message": "Parse error"})
175
269
  if payload is not None:
@@ -10,6 +10,7 @@ import re
10
10
  from typing import Any
11
11
 
12
12
  from concern_excerpt import bounded_reason_detail, concern_fields
13
+ import kimi_packet_mcp as packet_tools
13
14
  from kimi_packet_mcp import MAX_CHUNK_BYTES
14
15
 
15
16
 
@@ -652,6 +653,18 @@ def audit_codex(
652
653
  def invalid(reason: str) -> dict[str, Any]:
653
654
  return invalid_model_output(args, reason, "\n".join(concern_fragments))
654
655
 
656
+ packet_path = getattr(args, "packet", None)
657
+ packet_sha256 = getattr(args, "packet_sha256", None)
658
+ if packet_path or packet_sha256:
659
+ try:
660
+ if not packet_path or not packet_sha256:
661
+ raise ValueError()
662
+ packet_tools.packet_bytes(Path(packet_path), packet_sha256)
663
+ except (OSError, ValueError):
664
+ return inconclusive(args, "Codex packet binding changed", "binding_mismatch", False)
665
+ pending_packet_calls: dict[str, tuple[str, Any]] = {}
666
+ completed_packet_calls: set[str] = set()
667
+
655
668
  for event in events:
656
669
  event_type = event.get("type")
657
670
  if event_type == "turn.completed":
@@ -678,6 +691,39 @@ def audit_codex(
678
691
  attempted_tool=event_type,
679
692
  )
680
693
  item_type = item.get("type")
694
+ if item_type == "mcp_tool_call" and packet_path and packet_sha256:
695
+ tool = item.get("tool")
696
+ if item.get("server") != "code_review_packet" or tool not in {"read_packet", "search_packet"}:
697
+ return inconclusive(args, "Codex attempted a tool outside the frozen packet", "tool_boundary_violation", False)
698
+ reader = packet_tools.read_chunk if tool == "read_packet" else packet_tools.search_packet
699
+ expected = reader(Path(packet_path), packet_sha256, item.get("arguments"))
700
+ content = expected["content"]
701
+ if expected.get("isError") and content[0]["text"] == "packet binding changed":
702
+ return inconclusive(args, "Codex packet binding changed", "binding_mismatch", False)
703
+ if expected.get("isError") and content[0]["text"].startswith("invalid packet"):
704
+ return inconclusive(args, "Codex supplied arguments outside the packet tool schema", "tool_boundary_violation", False)
705
+ call_id = item.get("id")
706
+ signature = (tool, item.get("arguments"))
707
+ if (not isinstance(call_id, str) or not call_id or call_id in completed_packet_calls
708
+ or (call_id in pending_packet_calls and pending_packet_calls[call_id] != signature)):
709
+ return inconclusive(args, "Codex packet tool lifecycle is unverifiable", "transport_unverifiable", False)
710
+ if event_type != "item.completed":
711
+ if item.get("status") != "in_progress":
712
+ return inconclusive(args, "Codex packet tool lifecycle is unverifiable", "transport_unverifiable", False)
713
+ pending_packet_calls[call_id] = signature
714
+ continue
715
+ result = item.get("result")
716
+ if not isinstance(result, dict):
717
+ return invalid("Codex packet tool returned no verifiable result")
718
+ if (result.get("content") != content or result.get("structured_content") not in (None, {})
719
+ or result.get("structuredContent") not in (None, {})):
720
+ return inconclusive(args, "Codex packet tool result did not match frozen bytes", "binding_mismatch", False)
721
+ allowed_statuses = {"completed", "failed"} if expected.get("isError") else {"completed"}
722
+ if item.get("status") not in allowed_statuses or (item.get("error") and not expected.get("isError")):
723
+ return invalid("Codex packet tool did not complete successfully")
724
+ pending_packet_calls.pop(call_id, None)
725
+ completed_packet_calls.add(call_id)
726
+ continue
681
727
  if item_type == "agent_message":
682
728
  item_text = item.get("text")
683
729
  if isinstance(item_text, str) and item_text.strip():
@@ -729,7 +775,7 @@ def audit_codex(
729
775
  )
730
776
  if event_type == "item.completed" and item_type == "agent_message":
731
777
  saw_completed_agent_message = True
732
- if not saw_completed_agent_message or not saw_completed_turn:
778
+ if pending_packet_calls or not saw_completed_agent_message or not saw_completed_turn:
733
779
  return inconclusive(
734
780
  args,
735
781
  "Codex event stream lacked positive completion evidence",
@@ -926,6 +972,7 @@ def main() -> int:
926
972
  parser.add_argument("--model", default="")
927
973
  parser.add_argument("--events", required=True)
928
974
  parser.add_argument("--packet")
975
+ parser.add_argument("--packet-sha256")
929
976
  parser.add_argument("--packet-delivery", choices=("mcp", "inline"), default="mcp")
930
977
  parser.add_argument("--packet-receipt", default="")
931
978
  parser.add_argument("--result-file")