@ccoalm/ccl-skills 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/README.md +2 -2
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +8 -7
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +1 -1
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +16 -17
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +1 -1
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +195 -7
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/timeout-auth-and-capabilities.md +3 -3
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +13 -5
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +9 -3
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh +9 -3
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/normalize_review_timeout.sh +22 -0
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/opencode_review.sh +9 -3
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +1540 -129
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +8 -3
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +76 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +1858 -3
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +789 -0
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/update_review_plan_intent.py +513 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +4 -1
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -1
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +11 -10
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +64 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/agents/openai.yaml +4 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +72 -0
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/runtime-and-project-contract.md +58 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/source-map.md +41 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/verification-diagnostics-and-security.md +63 -0
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +8 -10
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-routing-and-readiness.md +10 -14
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/verify-developer-experience.md +1 -1
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/SKILL.md +135 -86
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/behavioral-aesthetic-logic.md +66 -80
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/delivery-contract.md +275 -0
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-execution-checklist.md +88 -214
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-impl-naming-and-versioning.md +2 -2
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-intake-and-acceptance.md +10 -8
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +4 -5
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md +112 -95
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/frontend-code-evidence-map.md +30 -21
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/interaction-design-patterns.md +22 -3
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/layout-recipes-and-screenshot-acceptance.md +20 -17
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-project-token-consistency.md +7 -9
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-stack-strategy.md +14 -10
  44. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/operational-processing-workflows.md +2 -0
  45. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +1 -1
  46. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-lifecycle-acceptance-and-iteration.md +9 -6
  47. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-surface-patterns.md +3 -0
  48. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/source-map.md +37 -10
  49. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +7 -1
  50. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +8 -5
  51. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-design-development.md +16 -5
  52. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/visual-craft.md +4 -2
  53. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +4 -1
  54. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +4 -4
  55. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +95 -5
  56. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +11 -9
  57. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/r0-leakage-audit.md +102 -0
  58. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +54 -0
  59. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +8 -0
  60. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +6 -6
  61. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +4 -3
  62. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +69 -2
  63. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +22 -0
  64. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +49 -4
  65. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/obligation-ledger.py +2748 -0
  66. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +20 -5
  67. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/shared_git_surface_gate.py +1142 -0
  68. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +17 -0
  69. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh +41 -4
  70. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ci_checkout_ref_binding.sh +120 -0
  71. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_domain_scan_terms.sh +82 -8
  72. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +336 -0
  73. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh +82 -10
  74. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh +1416 -0
  75. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_obligation_ledger_repo_audit.sh +57 -0
  76. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +141 -4
  77. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_pointer_integrity.sh +3 -1
  78. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh +1696 -0
  79. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_delivery_contract.sh +2117 -0
  80. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_loading_budget.sh +316 -0
  81. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +1176 -0
  82. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_cross_refs.sh +31 -1
  83. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate-skill.sh +9 -4
  84. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +980 -0
  85. package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +8 -6
  86. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +8 -7
  87. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/client-runtime-test-matrices.md +10 -2
  88. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +6 -5
  89. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/complex-workspace-patterns.md +1 -1
  90. package/dist/assets/release.json +175 -70
  91. package/package.json +1 -1
package/README.md CHANGED
@@ -1,12 +1,12 @@
1
1
  # @ccoalm/ccl-skills
2
2
 
3
3
  [![npm version](https://img.shields.io/npm/v/@ccoalm/ccl-skills)](https://www.npmjs.com/package/@ccoalm/ccl-skills)
4
- [![downloads](https://img.shields.io/npm/dm/@ccoalm/ccl-skills)](https://www.npmjs.com/package/@ccoalm/ccl-skills)
4
+ [![downloads](https://img.shields.io/npm/dt/@ccoalm/ccl-skills)](https://www.npmjs.com/package/@ccoalm/ccl-skills)
5
5
  [![license](https://img.shields.io/npm/l/@ccoalm/ccl-skills)](https://github.com/ccoalm/ccl-skills/blob/main/LICENSE)
6
6
 
7
7
  **Reusable workflows that help coding agents plan, build, test, review, and release software.**
8
8
 
9
- Not a prompt pack — a routed delivery system. 32 skills cover the whole lifecycle, and a routing layer reads what you asked for and hands it to the skill that owns that deliverable, so you never look one up.
9
+ Not a prompt pack — a routed delivery system. 33 skills cover the whole lifecycle, and a routing layer reads what you asked for and hands it to the skill that owns that deliverable, so you never look one up.
10
10
 
11
11
  Ask it to fix a bug and `defect-diagnosis` takes over: reproduce from first-hand failure evidence, isolate the cause, verify the fix, leave a regression test behind. Ask for a feature and `product-rd-workflow` routes it through requirement shaping, risk gates, implementation, and release. Ask it to touch code at all and `worktree-isolation` puts the work on its own branch first. Every skill also states when *not* to use it and which one to use instead — that is what keeps the routing sharp.
12
12
 
@@ -21,7 +21,7 @@ Use this skill for mobile app and cross-platform client engineering. It covers F
21
21
 
22
22
  ## Core Workflow
23
23
 
24
- Before editing app code, native project files, platform configs, styles, assets, or tests, complete enough analysis and planning for the change to be reviewable. Scale the plan to risk: a simple low-risk single-screen change can use a short inline plan; multi-platform, user-visible, API-visible, native-capability, release/store, bug-fix, branch/MR, unclear-risk, or high-risk work needs explicit task split, design checkpoint, target-device matrix, acceptance checks, verification commands, rollback or stop conditions, and named handoffs to design, testing, backend, release, or diagnosis skills before edits.
24
+ Before editing app code, native project files, platform configs, styles, assets, or tests, complete enough analysis and planning for the change to be reviewable. Scale the plan to risk: a simple low-risk single-screen change can use a short inline plan; multi-platform, API-visible, native-capability, release/store, bug-fix, branch/MR, unclear-risk, or high-risk work needs explicit task split, target-device matrix, acceptance checks, verification commands, rollback or stop conditions, and named handoffs to testing, backend, release, or diagnosis skills before edits. Runtime-visible work additionally consumes the canonical UI/UX delivery contract's Design brief and Test selection Phase 0 before the first implementation edit.
25
25
 
26
26
  Repo-local agent contracts (`AGENTS.md` at the repo root and in source directories) are part of the delivery contract: when a change moves a stable boundary, generated surface, workflow, or directory-local rule, update the nearest contract in the same MR and keep coverage in sync per `product-rd-workflow`'s spec / repo-contract sync gate.
27
27
 
@@ -37,8 +37,8 @@ When checking a cross-platform app against team standards, split conformance int
37
37
  - Identify shared behavior versus platform-specific behavior; list which platforms must be changed and verified.
38
38
  - Read local wrappers first: Flutter/Gradle/Xcode scripts, package managers, CI jobs, test targets, flavors/schemes, and generated files.
39
39
  - If a design exists, map each visible state and interaction to code ownership before implementing.
40
- - For any visible UI change, map the design checkpoint to code ownership: aesthetic hierarchy/density, interaction flow, behavioral feedback, user psychology, safe-area/keyboard/orientation adaptation, and device screenshot acceptance. Also record `product-ui-ux-design`'s implementation-owner checkpoint before the first edit — its field list (design/stack/test owners, entry-rule evidence, rendered/device evidence status) and copy-only path are authoritative there; load the named owner skills rather than only naming them, and treat a completion claim without `captured/verified` rendered evidence as incomplete — an explicitly accepted gap closes the slice only as `pre-runtime-test ready` / handoff, never as complete/done.
41
- - For UI/UX redesign slices meeting `product-ui-ux-design`'s page-slice trigger conditions — that gate's trigger list is authoritative and must be checked, not paraphrased, whenever a screen/surface change could be a redesign, restyle, new-style declaration, structural/visual-system change, continuation, or redesigned-surface reference — apply its cross-stack page-slice gate before framework mechanics: RED-first focused assertion, IA regrouping by user intent/consequence, behavior-contract preservation, state matrix, rendered evidence, and the design verdict (`accepted` / `rejected` / `pending`; missing = `pending`, and `design-rejected` blocks complete/MR-ready/normal/draft MR per the **Rejected-surface rule**). Translate examples from other stacks into the target app runtime instead of copying Flutter-, Android-, iOS-, or Web-specific commands as the method itself.
40
+ - For every visible UI change, load `../product-ui-ux-design/references/delivery-contract.md` and consume either its full Design brief + Phase 0 or its valid low-risk copy-only record + lightweight Phase 0 before coding. The lightweight path checks semantics, accessible name, localization, rendered extent, and target render without inventing unrelated matrices; risk-bearing copy uses the full path. For full slices, map structure, state/adaptation matrices, behavior and criteria to shared/platform owners; record build target, device, safe area, keyboard, orientation, text scale, input/lifecycle, and preserved behavior. For a native WebView, this skill owns the container/bridge member while `web-react-dev` or the actual web-content owner supplies a separate content member; neither render closes the other layer.
41
+ - Before the first implementation edit, add the canonical `client_entry` defined there: local rule identifier or short quote and implementation decision, target surface/runtime, planned run/capture command, and behavior that must remain unchanged.
42
42
 
43
43
  3. Define the client boundary before coding.
44
44
  - Screens, routes, tabs, navigation stack, deep links, and back behavior.
@@ -146,12 +146,13 @@ When checking a cross-platform app against team standards, split conformance int
146
146
  3. iOS(Swift/ObjC):`grep -rn "import <module>\|<ClassName>" ios/`;Xcode 的 dead-code-stripping 报告;Storyboard / XIB 引用 grep `.storyboard` 与 `.xib`
147
147
  4. 边界:跨端项目(KMM / 桥接代码)每层独立判,删一层不代表另一层也死;反射 / runtime 注解(Android)/ Objective-C runtime(iOS)grep 抓不到,需运行时验证
148
148
  - Inspect the screen in a real device, emulator, simulator, preview, or captured screenshot for any visible UI change.
149
- - For UI/UX redesign evidence, include the target device/form factor, safe-area, keyboard, orientation, dynamic type/text scale, loading/empty/error/final states, and a screenshot or equivalent rendered artifact; mark each dimension covered or `N/A` with a one-line reason. `N/A` is valid only when the reason names a verifiable structural fact, explains why that fact makes the dimension unreachable or unchanged for this slice, and includes a checkable pointer such as a file path, config key, or commit that resolves at review time. Also record the design verdict (`accepted` / `rejected` / `pending`) from a user or named independent design-owner review per `product-ui-ux-design`'s page-slice gate; a missing or absent verdict counts as `pending` and blocks complete/MR-ready exactly like a missing dimension — the screenshot alone is never acceptance. Persist evidence artifacts where reviewers can access them using sanitized/test accounts and redacting tokens, PII, credentials, private paths, and raw personal data; remove temporary smoke files or scripts before commit unless the repo intentionally owns them.
149
+ - For UI/UX redesign evidence, include the target device/form factor, safe-area, keyboard, orientation, dynamic type/text scale, loading/empty/error/final states, and a screenshot or equivalent rendered artifact; mark each dimension covered or `N/A` with a one-line reason. `N/A` is valid only when the reason names a verifiable structural fact, explains why that fact makes the dimension unreachable or unchanged for this slice, and includes a checkable pointer such as a file path, config key, or commit that resolves at review time. Persist evidence artifacts where reviewers can access them using sanitized/test accounts and redacting tokens, PII, credentials, private paths, and raw personal data; remove temporary smoke files or scripts before commit unless the repo intentionally owns them.
150
+ - Return the complete canonical client-record member defined in `../product-ui-ux-design/references/delivery-contract.md` for testing Phase 1 and the design verdict. The member includes its applied rule/decision, affected files/components, preserved behavior, exact command, immutable candidate binding, producer member/version actually exercised, artifacts, tested build/device targets, dimensions/states/input and lifecycle modes, criterion-mapped observations, coverage boundary, and gaps. A screenshot proves only the captured host/member states; it cannot close an unbound producer member. `testing-strategy` records aggregate sufficiency before the design owner records the candidate-bound verdict.
150
151
  - In approval-sensitive runtimes, do not create a new one-off smoke file for every screenshot or slice. Prefer a repo-owned smoke harness, existing integration test, debug route, fixture flag, deep link, or already-created slice harness; if a temporary harness is genuinely unavoidable, reuse one stable harness for the whole batch, mutate it minimally, and explicitly delete it before commit (a throwaway harness is evidence scaffolding, not shippable code; verify it is gone in the pre-commit diff) unless the repo intentionally owns it. Repeated edit-approval prompts from throwaway helper files are an execution defect, not normal evidence collection.
151
152
  - Capture Android evidence as approval-friendly single commands. In environments where command-prefix approval is used, do not combine `adb` with shell pipes, `>`, `>>`, command substitution, `&&`, or local filtering in the same command; those forms are split or de-scoped by the approval layer and can re-trigger prompts even when `adb` itself is approved. Use sanitized/test fixture screens, then run three standalone commands: `adb -s <serial> shell screencap -p "/sdcard/<artifact>.png"`, `adb -s <serial> pull "/sdcard/<artifact>.png" "<repo-evidence-path>/"`, and `adb -s <serial> shell rm "/sdcard/<artifact>.png"` in cleanup even when pull fails or the run aborts. Generate `<artifact>` from a safe basename alphabet (no spaces, globs, or path separators) and quote the `/sdcard` path in every command so `rm` targets exactly the file you created, and create the destination with `mkdir -p <repo-evidence-path>` before pulling. Never write credential, PII, account, or otherwise sensitive real-user screens to `/sdcard`; redact or fixture first. If the environment explicitly pre-approves shell redirection for `adb`, `adb -s <serial> exec-out screencap -p > <repo-evidence-path>/<artifact>.png` is acceptable and avoids device-side storage, but the transcript must record that redirection is approved for that command shape. Either way, validate the captured file is a non-empty valid PNG before relying on it: a failed pull, or an empty/zero-byte `exec-out` stream, means there is no host artifact — retry rather than counting it as evidence.
152
- - A mobile screenshot that proves the app rendered is not automatically design acceptance evidence. If the design owner or page-slice review rejects the captured surface for weak hierarchy, blank-feeling composition, generic patched layout, clipped safe-area/keyboard behavior, or mismatch with the design checkpoint, route back to `product-ui-ux-design`'s **Rejected-surface rule** and re-capture after redesign; the app implementation waits on that design verdict instead of self-adjudicating taste. A technically successful but visually failed smoke leaves the slice `design-rejected` — blocking complete, MR-ready, and any normal or draft MR until the re-rendered surface is accepted, not only the completion claim.
153
+ - A missing or absent design verdict remains `pending` and blocks the canonical contract's `complete`, MR-ready, merge-ready, and normal MR/handoff readiness states; a mobile screenshot proves only that the app rendered and is never design acceptance. If a hierarchy, composition, behavior, safe-area/keyboard, or brief criterion fails, return that failure instead of self-adjudicating taste; the design owner applies `delivery-contract.md`'s rejected-candidate path and requires fresh runtime evidence for the revised candidate.
153
154
  - For platform enablement, verify each newly added platform with a real platform build and an install/run smoke on the relevant emulator or simulator. Adding the host project or passing one platform does not complete the slice; record the platform targets, commands, and screenshot or runtime evidence.
154
- - For mobile runtime changes, device/emulator/simulator smoke is a completion gate when lower layers cannot prove the behavior. This includes changes to native capabilities, WebView/native bridge callbacks, foreground/background recovery, orientation/safe-area/keyboard behavior, storage/session restore, route/deep-link behavior, upload/media flows, permissions, and rendered loading/error/final states. If the runner is missing, first attempt normal setup; if still unavailable, stop at `pre-runtime-test ready` or `blocked` and name the owner, attempted commands, residual risk, and next unblock action. `pre-runtime-test ready` is handoff-only, not merge-ready, release-ready, or complete.
155
+ - For mobile runtime changes, device/emulator/simulator smoke is a completion gate when lower layers cannot prove the behavior. This includes changes to native capabilities, WebView/native bridge callbacks, foreground/background recovery, orientation/safe-area/keyboard behavior, storage/session restore, route/deep-link behavior, upload/media flows, permissions, and rendered loading/error/final states. If the runner is missing, first attempt normal setup; if still unavailable, stop at `pre-runtime-test-ready` or `blocked` and name the owner, attempted commands, residual risk, and next unblock action. `pre-runtime-test-ready` is handoff-only, not merge-ready, release-ready, or complete.
155
156
  - For emulator readiness scripts, do not trust process launch or boot log text alone. Confirm the runner sees the target device, then verify direct device state such as `adb devices -l`, `adb -s <serial> get-state`, and platform boot readiness before calling Android ready; keep the serial in the output so later E2E commands target the same device.
156
157
  - If the app or app-hosted H5 CI only builds or deploys packages, treat that as structural release evidence, not functional proof. Interaction, permission, lifecycle, upload/media, and recovery changes still need focused assertions plus rendered device or host-container smoke.
157
158
  - Check accessibility labels, dynamic type/text scale, focus order, screen reader reachability, contrast, and touch targets.
@@ -175,7 +176,7 @@ When checking a cross-platform app against team standards, split conformance int
175
176
  - Do not treat a mobile API client as done until empty response, invalid JSON, non-2xx envelope, auth expiry, network failure, cancellation, and backend error message extraction are covered at the client or screen boundary when relevant.
176
177
  - Do not scatter backend enum/string literals through app screens, deep links, native bridge payload handling, storage, analytics, or tests. Centralize finite-value parsing, display labels, defaults, and unknown-value behavior at the API/client-domain boundary, and keep raw literals only in clearly named boundary conversion tests that cover every known external value plus unknown/default behavior. Migrate existing non-boundary test raw literals for that value in the same pull request or mark each remaining use with `finite-value-debt: <task-ref> <owner> <deadline> <reason>`, even when the current slice does not introduce a new mapper.
177
178
  - Do not debug app failures from code inspection alone when a runnable reproduction, trace, screenshot, or device log can be collected.
178
- - Do not claim a mobile client fix is complete without naming the platform targets that were verified. If required device/emulator/simulator smoke is unavailable after remediation, the status is `pre-runtime-test ready` or `blocked`, not complete.
179
+ - Do not claim a mobile client fix is complete without naming the platform targets that were verified. If required device/emulator/simulator smoke is unavailable after remediation, the status is `pre-runtime-test-ready` or `blocked`, not complete.
179
180
 
180
181
  ## Reference Loading
181
182
 
@@ -74,4 +74,4 @@ When the mobile app must talk to internal-CA / enterprise-CA / self-signed hosts
74
74
  - **Debug-only instrumentation must be structurally excluded from every real-user build, and the exclusion verified on the exact shipped artifact — a runtime flag is not the guard.** The class is broad: debug bridges, in-app state/inspection servers, dev overlays/menus, network inspectors, remote WebView debugging, verbose token/PII logging, mock-auth or auth-bypass hooks, and any QA state-export or runtime-introspection surface — including ones with no visible UI that are still reachable via deep link, push action, JS bridge message, or custom URL scheme. Any of these is an attack surface and data-exposure risk (an in-app state server can bind a port and dump app state on a real user's device) if it reaches a build that can touch real users, real auth, or real PII. **Scope by distribution channel + data sensitivity, not a single `isProduction`/`isDebug` enum**: the gate covers App Store / Play production AND TestFlight, enterprise / ad-hoc, dogfood, Play internal/closed testing, and staged rollout — all carry real users (same channel-not-enum boundary as **TLS Trust And Certificate Pinning** above).
75
75
  - **Exclude at the dependency/target level, per real-user variant — not just a runtime flag, and not just the variant named `release`.** An `if (isDebug)` check still compiles the code — and any bound port/server — into the binary; even `kDebugMode` / `__DEV__` only strip the Dart/JS call path while a native plugin/module/platform-channel stays linked. The primary guarantee is that the debug dependency/target is absent from the release classpath / target membership / build flavor of *every variant that reaches real users*, verified per variant: iOS `#if DEBUG` plus Debug-only target membership — CocoaPods can scope a pod to Debug configurations (`:configurations => ['Debug']`), but SwiftPM has no build-configuration-scoped *package dependency*, so a debug package must be confined to a debug-only target/scheme and left unlinked by every real-user target rather than gated by a `Package.swift` condition; also confirm no real-user scheme (TestFlight / enterprise / staging) inherits `DEBUG` in `SWIFT_ACTIVE_COMPILATION_CONDITIONS` / `GCC_PREPROCESSOR_DEFINITIONS`, or `#if DEBUG` code compiles straight into it; Android a `debugImplementation` / debug source-set dependency, NOT `implementation` plus a runtime check (a manifest `ContentProvider`/initializer or reflection keep-rule runs it anyway) — and note a `debug` / `qaDebug` buildType shipped to dogfood/internal testing IS a real-user build that includes those deps, so real-user channels must use release-class variants with debug deps excluded; Flutter/RN remove the package from the release flavor at the platform level (a native plugin autolinks/registers into the binary via podspec/Gradle/manifest unless excluded from the release target, even when the Dart/JS guard is stripped).
76
76
  - **Verify on the exact final signed/packaged upload artifact per channel and variant, and treat a symbol/string grep as a backstop, not proof.** A single `nm -j` / `strings` / dexdump / grep is defeated by R8 / Hermes / LTO renaming (`DevInspector` → `a.b.C`), dynamic loading (`dlopen`, Android dynamic feature, RN/Flutter OTA bundle, remote WebView JS), and non-code surfaces (debug assets, URL schemes, manifest providers/services, plist keys, debug entitlements). *Discover* the inventory from the dependency graph, manifest/plist, entitlements, and route/deep-link/bridge registries — not a remembered list, since an unlisted transitive SDK initializer or debug bridge handler is exactly the gap — then for each assert absence across binary, dex/native libs, assets/resources, manifest/plist, entitlements (`get-task-allow` is debug-only; Local Network / ATS / cleartext exceptions are security-sensitive plist/capability config that need production justification, not necessarily debug), and the OTA/dynamic-load channels reachable by that release (a passing artifact digest does NOT bind a debug bundle pushed later via CodePush/Expo/remote WebView JS, so real-user OTA/remote channels must not serve debug payloads). Run it on the post-signing artifact you actually upload (record its digest), across every shipped platform / scheme / flavor / extension, never on an intermediate `assembleRelease`. (`__DEV__` / `kDebugMode` elimination is also minifier-dependent and has silently regressed historically, so the source guard alone is never proof.)
77
- - **Make it a release-blocking evidence row, not MR prose** (otherwise the gate is decorative — a team ships from local Xcode/Fastlane, self-marks "checked", or never adds the CI job). The row is machine-generated, tied to the uploaded artifact digest + variant, and verified by someone other than the author; it records the digest, distribution channel, variant matrix, and check-command output, and the release/upload lane fails closed on a missing row. **Risk acceptance cannot waive structural exclusion** of debug-only instrumentation in any real-user / real-auth / real-PII build: an exception may only reclassify the channel as non-real-user / no-real-data, or apply to a production-intended diagnostic that ships its own auth + redaction controls (at which point it is no longer debug-only instrumentation) — otherwise the release stays blocked, an unbounded "accepted" is not a valid disposition. Guidance cannot force a given repo's pipeline to fail closed — that mechanical enforcement is the release-pipeline owner's job (route to `platform-release-engineering`); absent the row the status is not release-ready, the same way missing device smoke leaves a slice `pre-runtime-test ready`. A manual removal/teardown flow is a convenience path for a one-off exit, never the mechanism that keeps debug code out of real-user builds.
77
+ - **Make it a release-blocking evidence row, not MR prose** (otherwise the gate is decorative — a team ships from local Xcode/Fastlane, self-marks "checked", or never adds the CI job). The row is machine-generated, tied to the uploaded artifact digest + variant, and verified by someone other than the author; it records the digest, distribution channel, variant matrix, and check-command output, and the release/upload lane fails closed on a missing row. **Risk acceptance cannot waive structural exclusion** of debug-only instrumentation in any real-user / real-auth / real-PII build: an exception may only reclassify the channel as non-real-user / no-real-data, or apply to a production-intended diagnostic that ships its own auth + redaction controls (at which point it is no longer debug-only instrumentation) — otherwise the release stays blocked, an unbounded "accepted" is not a valid disposition. Guidance cannot force a given repo's pipeline to fail closed — that mechanical enforcement is the release-pipeline owner's job (route to `platform-release-engineering`); absent the row the status is not release-ready, the same way missing device smoke leaves a slice `pre-runtime-test-ready`. A manual removal/teardown flow is a convenience path for a one-off exit, never the mechanism that keeps debug code out of real-user builds.
@@ -66,28 +66,27 @@ credentials only; broad semantic confidentiality stays operator-owned per the
66
66
  Diff Confidentiality section. Consult stays on `claude_review.sh`. Load the
67
67
  staged-contract and client-routing references below for details.
68
68
 
69
- Current contract: the plan is optional for review/challenge (a derived default
70
- is stamped `review_plan_source=derived-default`) but required for `complete`
71
- mode, and gate output is schema 3. Agent automation gets the
72
- initial review plus at most four challenges, all in one retained task-level chain.
73
- Any initial review with positive challenge capacity must create that chain at
74
- Agent index 1; an untracked initial review is single-round with budget 0.
75
- The budget is a ceiling: after a clean tracked challenge, `complete` may close
76
- the chain early, preserve unused-round count, and disable further Agent review.
77
- Every result exposes controller-owned `self_review_gate`; an outstanding checkpoint
78
- blocks only another external review and/or a completion claim, never productive
79
- implementation or tests. A passed final external round remains gated until local
80
- `--mode complete` binds the exact result and current deep-self-review plan. Human
81
- review/stop/waiver/commit/merge authority is external and cannot be asserted by a
82
- repository file, CLI flag, environment variable, or model output. Load the staged
83
- contract before invoking the gate. Terminal wrapper/controller failures emit
84
- `stop_reviewer_lane`; they never instruct the host to stop the whole task.
69
+ Current contract: review/challenge may use a stamped
70
+ `review_plan_source=derived-default`; `complete` requires a plan and output uses
71
+ schema 3. Automation retains one chain: one review plus at most four challenges.
72
+ Positive challenge capacity opens it at Agent index 1; budget zero is untracked.
73
+ The sole release/high-risk budget-zero exception is a controller-proved
74
+ `markdown-punctuation-only` review: it requires `wording_only_boundary`, permits
75
+ no `complete`, and rejects an author assertion alone (recipe:
76
+ `references/staged-review-contract.md`). After a clean tracked
77
+ challenge, `complete` may close early and preserve unused rounds. Every result
78
+ exposes controller-owned `self_review_gate`; an outstanding checkpoint blocks
79
+ only external review or completion, not implementation or tests. Even a passed
80
+ final external round needs local `--mode complete` binding the exact result and
81
+ current deep-self-review plan. Human review/stop/waiver/commit/merge authority
82
+ cannot come from repository content, flags, environment, or model output.
83
+ Terminal failures emit `stop_reviewer_lane`; they never stop the whole task.
85
84
 
86
85
  The shared skill never pins a provider or model. Kimi inherits the user's local default model and is uniformly classified as the Moonshot family; Codex likewise inherits its local default model and is uniformly classified as the OpenAI family. Neither client reads or configures provider/model subdivisions. OpenCode invokes `ccl-review` without `--model`: if the user configured `agent.ccl-review.model`, that one model is used; otherwise OpenCode resolves its normal default. The exported OpenCode session must attribute the actual provider/model, and an unmapped or same-family result is rejected before another client is considered. There is no built-in Kimi/DeepSeek provider chain and no one-shot model override to type.
87
86
 
88
87
  Client discovery is separate from model ownership. For Kimi, the wrapper accepts an absolute executable `KIMI_BIN`; otherwise it checks `PATH`, then `$KIMI_CODE_HOME/bin/kimi` when configured, then the standard `~/.kimi-code/bin/kimi` location. The resolved executable must be absolute and executable before the wrapper changes directory. This avoids treating a non-interactive shell's narrower `PATH` as proof that Kimi is not installed, honors a custom Kimi home, and avoids relative-PATH drift in the temporary workspace; it does not change the user's model/provider settings. An executable PATH selection keeps normal shell precedence; use `KIMI_BIN` to bypass a broken executable shim explicitly.
89
88
 
90
- The Claude provider wrapper implements help inspection, built-in tool availability restriction, safe-mode customization isolation, strict empty MCP configuration, empty user/project/local setting sources, CLAUDE.md and auto-memory disabling, repository directory scoping with `--add-dir`, consult dirty-worktree fail-closed checks, prompt-only consult for pasted evidence, temp prompt file permissions and cleanup, Python subprocess capture without putting diff content on argv, structured JSON parsing, schema validation, timeout handling, auth false-negative classification, and the two-invocation cap. Runs without selected owner skills disable skills/commands. Owner-aware runs load the already-installed CCL skill plugin through `--plugin-dir` and explicitly invoke each selected skill; they do not manufacture a selected-only plugin or claim that unselected installed skills are unavailable. Claude's own built-in identifiers and all skills present in the controller-supplied registry may remain visible in stream init; only selected owners are frozen-hash verified and counted as explicit invocation. The parser rejects entries outside that registered surface or duplicate identifiers and keeps tools/MCP isolated. For review/challenge it also accepts the orchestrator's frozen `--diff-file` and emits stable `reason_code`, `fallback_eligible`, and `next_action` fields on every inconclusive path. Every mode captures Claude with structured stream JSON and adds `--json-schema` when the CLI advertises it; `parse_review_json.py` unwraps the result event's structured payload and rejects any envelope whose `is_error`, `subtype`, `api_error_status`, or `permission_denials` fields signal a non-clean run. `classify_envelope.py` reads those same structured fields to classify auth/quota/permission/error failures, so raw-text matching on the CLI's prose is only a fallback for the case where no JSON envelope was produced. On an auth false negative the wrapper returns `auth_path_unavailable`; a host rerun of the same gate adds `--host-remediation-attempted`. `--timeout` controls each formal invocation and must be an integer of at least 5 seconds; values above 600 are clamped to 600. The wrapper does not make a separate behavior-probe request. The formal invocation's own stream-init must declare exactly the expected tool set: packet-only modes allow only Claude's internal `StructuredOutput` tool when schema output is enabled, while repository consult additionally allows `Read,Grep,Glob`. Unexpected tool use, MCP inheritance, or schema drift fails closed. `--allowedTools` is a permission auto-approval rule, not an availability restriction, and is never used as the sandbox. The wrapper deliberately never runs `claude auth status` through command substitution, because that path can itself report a false logged-out state on this machine.
89
+ The Claude provider wrapper implements help inspection, built-in tool availability restriction, safe-mode customization isolation, strict empty MCP configuration, empty user/project/local setting sources, CLAUDE.md and auto-memory disabling, repository directory scoping with `--add-dir`, consult dirty-worktree fail-closed checks, prompt-only consult for pasted evidence, temp prompt file permissions and cleanup, Python subprocess capture without putting diff content on argv, structured JSON parsing, schema validation, timeout handling, auth false-negative classification, and the two-invocation cap. Runs without selected owner skills disable skills/commands. Owner-aware runs load the already-installed CCL skill plugin through `--plugin-dir` and explicitly invoke each selected skill; they do not manufacture a selected-only plugin or claim that unselected installed skills are unavailable. Claude's own built-in identifiers and all skills present in the controller-supplied registry may remain visible in stream init; only selected owners are frozen-hash verified and counted as explicit invocation. The parser rejects entries outside that registered surface or duplicate identifiers and keeps tools/MCP isolated. For review/challenge it also accepts the orchestrator's frozen `--diff-file` and emits stable `reason_code`, `fallback_eligible`, and `next_action` fields on every inconclusive path. Every mode captures Claude with structured stream JSON and adds `--json-schema` when the CLI advertises it; `parse_review_json.py` unwraps the result event's structured payload and rejects any envelope whose `is_error`, `subtype`, `api_error_status`, or `permission_denials` fields signal a non-clean run. `classify_envelope.py` reads those same structured fields to classify auth/quota/permission/error failures, so raw-text matching on the CLI's prose is only a fallback for the case where no JSON envelope was produced. On an auth false negative the wrapper returns `auth_path_unavailable`; a host rerun of the same gate adds `--host-remediation-attempted`. `--timeout` controls each formal invocation and must be an integer of at least 5 seconds; values above 1200 are clamped to 1200. The wrapper does not make a separate behavior-probe request. The formal invocation's own stream-init must declare exactly the expected tool set: packet-only modes allow only Claude's internal `StructuredOutput` tool when schema output is enabled, while repository consult additionally allows `Read,Grep,Glob`. Unexpected tool use, MCP inheritance, or schema drift fails closed. `--allowedTools` is a permission auto-approval rule, not an availability restriction, and is never used as the sandbox. The wrapper deliberately never runs `claude auth status` through command substitution, because that path can itself report a false logged-out state on this machine.
91
90
 
92
91
  Owner-aware Claude init must declare the `ccl-skills` plugin. Current public
93
92
  init output does not enumerate selected plugin skills or commands, so the wrapper
@@ -345,7 +345,7 @@ line echoes the packet-specific receipt and the remaining text matches the
345
345
  review contract. Pre-inference setup and the capability probe are charged
346
346
  against the controller-granted lane budget, and the probe itself is capped at
347
347
  60 seconds. MCP review receives
348
- the remaining budget up to the wrapper's existing 600-second ceiling; inline
348
+ the remaining budget up to the wrapper's 1200-second ceiling; inline
349
349
  review is additionally capped at 120 seconds because its bounded prompt remains
350
350
  visible in process argv. Both
351
351
  use a one-second forced-kill grace; a deadline timeout may cascade, while an
@@ -7,7 +7,8 @@ The controller has three modes:
7
7
  - `complete`: a local deep-self-review checkpoint that calls no reviewer.
8
8
 
9
9
  Explore/build may configure `challenge_budget=0..4`; release/high-risk requires
10
- at least one challenge. The initial review consumes Agent round 1, so total
10
+ at least one challenge unless the exact candidate qualifies for the
11
+ proof-bound wording-only single-review exception below. The initial review consumes Agent round 1, so total
11
12
  Agent-autonomous external review is at most five rounds. Human-requested review
12
13
  is outside this budget and must be attributed by the consuming trusted platform.
13
14
  The budget is a ceiling, not a quota: after a clean tracked challenge, local
@@ -29,6 +30,48 @@ safety; build adds failure paths, tests, and compatibility; release adds rollout
29
30
  and operations. High-risk input raises depth to release and adds
30
31
  `high_risk_boundary`.
31
32
 
33
+ The serialized plan is at most 32,000 bytes and `intent` is 8..4,000
34
+ characters. Those are validation limits, not permission for a caller to slice a
35
+ longer value into shape: the gate can validate only the final value it receives
36
+ and cannot detect that a caller discarded the newest scope transition first.
37
+ Call `scripts/update_review_plan_intent.py` to mutate an existing plan. It checks
38
+ an optional expected SHA-256, opens bounded inputs without following links,
39
+ writes atomically, preserves ordinary POSIX permission bits, rejects duplicate
40
+ JSON object keys before mutation, and leaves the plan byte-identical on
41
+ validation or overflow failure. When the plan producer first
42
+ chooses the stable core, it persists one reserved evidence row with
43
+ `id=review-plan-intent-stable-core-v1` and result
44
+ `chars=<character-count>;sha256=<UTF-8-SHA-256>`. On overflow, pass that exact
45
+ core and the latest transition as separate files: the helper requires both the
46
+ character count and digest to match the persisted identity, requires latest to
47
+ be absent from the current intent, and joins them with one fixed blank line.
48
+ Before replacing the visible intent, it archives the exact removed suffix in a
49
+ reserved, ordered base64 evidence group. That group's manifest binds the prior
50
+ intent character count and SHA-256, core length, suffix byte count and SHA-256,
51
+ encoding, and part count; if the plan or evidence-row cap cannot hold the group,
52
+ compaction fails without changing the file. These rows preserve bytes but do not
53
+ classify their meaning. A matching arbitrary prefix is therefore insufficient. Existing plans without
54
+ the reserved row remain appendable but cannot compact until their producer has
55
+ explicitly established the identity; the helper returns
56
+ `intent_core_identity_missing` instead of guessing. The identity proves the
57
+ recorded text boundary, not semantic correctness or protection from a caller
58
+ that rewrites the whole plan outside the helper. Keep semantic dispositions in
59
+ ordinary evidence rows and ordered prior-review results; the byte archive is a
60
+ loss-prevention carrier, not a substitute. Never use prefix/tail truncation as
61
+ compaction. One final line ending is treated as a file delimiter; other outer
62
+ whitespace is rejected.
63
+
64
+ The expected digest is a stale-at-open guard, not a lock: callers serialize
65
+ writers and provide a trusted, stable parent directory for the read-check-replace
66
+ interval. Plans are caller-owned, singly linked regular files; ACLs, extended
67
+ attributes, ownership, and special mode bits are outside this replacement
68
+ contract. A post-rename directory-sync failure reports
69
+ `plan_committed_durability_unknown` with the new digest: re-read before deciding
70
+ whether to retry, because the target has already changed. A stdout pipe that
71
+ closes before the success receipt is delivered likewise reports
72
+ `plan_committed_receipt_lost` on stderr with a nonzero exit: the update is
73
+ committed, only the receipt was lost, so re-read the plan instead of retrying.
74
+
32
75
  Each self-review row may name a direct sibling skill. Omission selects the
33
76
  `code-review` baseline. The controller also derives owners deterministically
34
77
  from changed `skills/<name>/` paths, test paths, and supported source extensions.
@@ -97,6 +140,145 @@ or MCP servers make the plugin ineligible for this bounded review lane.
97
140
  Wrappers keep an explicit selected-owner count instead of testing empty Bash
98
141
  arrays under `set -u`, preserving the no-owner lane on Bash 3.2.
99
142
 
143
+ ## Base-derived packet input boundary
144
+
145
+ `--base` freezes the tracked diff plus every non-ignored untracked path in
146
+ scope. Exact-candidate binding means the controller never skips an untracked
147
+ path or replaces its contents with a placeholder. An untracked path containing
148
+ a Unicode control or line-separator character, symlink, hardlink, other
149
+ non-regular file, NUL-bearing file, non-UTF-8 file, or file over 200,000 bytes
150
+ makes the whole lane fail before provider execution with
151
+ `reason_code=invalid_input`. The same result applies when rendered untracked
152
+ content or the combined tracked-plus-untracked packet exceeds 200,000 bytes.
153
+
154
+ Move an out-of-scope path outside the candidate or add a correct ignore rule;
155
+ commit an in-scope path when Git should represent it; or compose complete
156
+ `--diff-file` partitions when the candidate must be split. Never omit a path
157
+ and report the remaining packet as the whole candidate.
158
+
159
+ ## Proof-bound wording-only single review
160
+
161
+ The wording-only exception is one untracked `review` with
162
+ `challenge_budget=0`; it is not a chain, challenge, or `complete` checkpoint.
163
+ Supply `--wording-only-proof-file` to bind the exception to the exact packet.
164
+ Without that proof, an explore/build budget-zero review remains an ordinary
165
+ single review and cannot be recorded as the wording-only exception;
166
+ release/high-risk budget zero fails before inference.
167
+
168
+ At release depth, including depth raised by a high-risk tag, only a
169
+ controller-proved `markdown-punctuation-only` check may use this exception.
170
+ `markdown-token-replacement` remains available for explore/build budget-zero
171
+ review, but it cannot waive the release/high-risk challenge: byte-exact token
172
+ replacement does not prove that the old and new tokens have the same meaning.
173
+
174
+ The proof is a single-link regular UTF-8 JSON file of at most 16,000 bytes:
175
+
176
+ ```json
177
+ {"schema_version":1,"candidate_sha256":"<packet-sha256>","check":{"kind":"markdown-punctuation-only"}}
178
+ ```
179
+
180
+ The other fixed check is
181
+ `markdown-token-replacement`, whose `check` also contains `old_token`,
182
+ `new_token`, and integer `expected_count` (1..100). The controller never trusts
183
+ a caller-supplied pass result. It reparses the frozen packet and derives the
184
+ status, files, changed-line count, replacement count, and scope SHA-256.
185
+
186
+ The accepted packet is deliberately narrow: a canonical full-context unified
187
+ Git diff, LF-terminated, at most 200,000 bytes, changing existing regular
188
+ Markdown files inside exactly one existing non-linked skill package. Every
189
+ file's first hunk starts at line 1 so frontmatter is inspectable. Adds,
190
+ deletes, renames, multi-skill changes, frontmatter or `description` edits,
191
+ non-regular Git modes, custom/compact packets, extra context outside the diff,
192
+ and files without a final newline fail closed. `markdown-punctuation-only`
193
+ accepts only one-for-one plain-prose line replacements whose non-punctuation
194
+ characters remain identical; numeric tokens must additionally survive
195
+ byte-for-byte (deleting the dot in `5.5` is a threshold change, not
196
+ punctuation), and a question mark may not be added or removed (a statement
197
+ turned into a question is a meaning change). Lines must start at column zero
198
+ and contain prose; line adds/deletes, Markdown headings, lists, block quotes,
199
+ links, tables, inline code, fenced or indented code, and raw HTML `pre`/`code`
200
+ containers fail closed. `markdown-token-replacement` requires every changed
201
+ line pair to differ only by the named whole-token replacement, with the exact
202
+ total count, and rejects packets whose changed lines touch a Markdown or HTML
203
+ code container.
204
+
205
+ This recipe produces the exact packet and proof without a second parser or a
206
+ pretend verifier command. Set `WORDING_KIND=markdown-punctuation-only`, or set
207
+ `WORDING_KIND=markdown-token-replacement` plus `WORDING_OLD`, `WORDING_NEW`, and
208
+ `WORDING_COUNT`:
209
+
210
+ ```bash
211
+ : "${CODE_REVIEW_SKILL_DIR:?set the installed code-review skill directory}"
212
+ : "${REPO_ROOT:?set the absolute repository root}"
213
+ : "${REVIEW_BASE:?set the exact base ref}"
214
+ : "${SKILL_NAME:?set the one existing skill package name}"
215
+ : "${REVIEW_STAGE:?set explore, build, or release}"
216
+ : "${IMPLEMENTER_FAMILY:?set the implementer model family}"
217
+ : "${REVIEW_PLAN_FILE:?set the absolute review-plan JSON path}"
218
+ : "${REVIEW_EVIDENCE_DIR:?set an existing durable private evidence directory}"
219
+ : "${WORDING_KIND:?set one supported wording-only check kind}"
220
+
221
+ umask 077
222
+ WORDING_RUN_DIR="$(mktemp -d "$REVIEW_EVIDENCE_DIR/wording-review.XXXXXX")" || exit 1
223
+ WORDING_DIFF="$WORDING_RUN_DIR/candidate.diff"
224
+ WORDING_PROOF="$WORDING_RUN_DIR/proof.json"
225
+ WORDING_RESULT="$WORDING_RUN_DIR/review.json"
226
+
227
+ git -C "$REPO_ROOT" diff --no-color --no-ext-diff --no-textconv --full-index \
228
+ --src-prefix=a/ --dst-prefix=b/ --unified=1000000 \
229
+ "$REVIEW_BASE" -- "skills/$SKILL_NAME" >"$WORDING_DIFF" || exit 1
230
+
231
+ python3 - "$WORDING_DIFF" "$WORDING_PROOF" "$WORDING_KIND" \
232
+ "${WORDING_OLD:-}" "${WORDING_NEW:-}" "${WORDING_COUNT:-0}" <<'PY'
233
+ import hashlib
234
+ import json
235
+ import sys
236
+ from pathlib import Path
237
+
238
+ diff_path, proof_path = map(Path, sys.argv[1:3])
239
+ kind, old, new, count = sys.argv[3:]
240
+ check = {"kind": kind}
241
+ if kind == "markdown-token-replacement":
242
+ check.update(old_token=old, new_token=new, expected_count=int(count))
243
+ elif kind != "markdown-punctuation-only":
244
+ raise SystemExit("unsupported WORDING_KIND")
245
+ payload = {
246
+ "schema_version": 1,
247
+ "candidate_sha256": hashlib.sha256(diff_path.read_bytes()).hexdigest(),
248
+ "check": check,
249
+ }
250
+ proof_path.write_text(
251
+ json.dumps(payload, ensure_ascii=False, separators=(",", ":")) + "\n",
252
+ encoding="utf-8",
253
+ )
254
+ PY
255
+
256
+ WORDING_RISK_ARGS=()
257
+ for tag in ${REVIEW_RISK_TAGS:-}; do WORDING_RISK_ARGS+=(--risk-tag "$tag"); done
258
+ if ! bash "$CODE_REVIEW_SKILL_DIR/scripts/review_gate.sh" \
259
+ --mode review --stage "$REVIEW_STAGE" --challenge-budget 0 \
260
+ --cwd "$REPO_ROOT" --diff-file "$WORDING_DIFF" \
261
+ --review-plan-file "$REVIEW_PLAN_FILE" \
262
+ --wording-only-proof-file "$WORDING_PROOF" \
263
+ ${WORDING_RISK_ARGS[@]+"${WORDING_RISK_ARGS[@]}"} \
264
+ --implementer-family "$IMPLEMENTER_FAMILY" >"$WORDING_RESULT"; then
265
+ cat "$WORDING_RESULT" >&2
266
+ exit 1
267
+ fi
268
+ cat "$WORDING_RESULT"
269
+ ```
270
+
271
+ A valid result carries `wording_only_proof_sha256`, controller-derived
272
+ `wording_only_scope.status=passed`, and a reviewed
273
+ `wording_only_boundary` concern. That concern independently confirms the edit
274
+ changes no trigger, scope, routing, validation, acceptance, rule, threshold,
275
+ boundary, frontmatter, description, or other meaning. If it is missing,
276
+ inconclusive, or reports a possible semantic change, the wording-only exception
277
+ does not apply: use the normal challenge and behavioral-evidence path. Any
278
+ candidate edit regenerates the packet and proof and requires a new review.
279
+ Keep the diff, proof, and result together; a digest whose source artifact was
280
+ deleted is not independently auditable evidence.
281
+
100
282
  ## Agent review chain
101
283
 
102
284
  Multi-round Agent automation supplies `review_chain_id`, a contiguous
@@ -110,11 +292,13 @@ index 1; an untracked initial review is single-round and therefore uses budget 0
110
292
  The chain binds task scope, candidate identity per round, result hashes, mode,
111
293
  status, challenge focus, controller, and selected owners. The opaque
112
294
  `review_scope_sha256` always hashes normalized intent, acceptance, stage/depth,
113
- risk tags, and challenge budget; chain identity is validated separately. Every
114
- result also carries the canonical `review_scope` object that digest is taken
115
- over — intent and acceptance appear only as `intent_sha256` /
116
- `acceptance_sha256`, never as raw plan text, because results are logged and
117
- archived independently of the plan. A prior result is accepted only when its
295
+ risk tags, challenge budget, and the wording-only proof/scope digests (both
296
+ `null` outside that exception); chain identity is validated separately. This
297
+ prevents deleting or nulling the top-level wording-only fields from reclassifying
298
+ that receipt as a normal completion input. Every result also carries the
299
+ canonical `review_scope` object that digest is taken over — intent and
300
+ acceptance appear only as `intent_sha256` / `acceptance_sha256`, never as raw
301
+ plan text, because results are logged and archived independently of the plan. A prior result is accepted only when its
118
302
  recorded `review_scope` reproduces its own `review_scope_sha256` and that digest
119
303
  matches the current scope, so copying a digest onto a differently-scoped result
120
304
  no longer passes. This binding proves internal consistency, not authority: prior
@@ -180,7 +364,11 @@ Budget exhaustion likewise stops only automatic reviewer calls. Continue local
180
364
  fixes, self-review, tests, and independent work; park only decision-dependent
181
365
  work. Enter `awaiting_human` only when no independent runnable work remains.
182
366
 
183
- The current result envelope is schema 3. The controller bounds cumulative reviewer-lane execution with `--total-timeout`
367
+ The current result envelope is schema 3. The generic per-invocation `--timeout`
368
+ keeps its 600-second default and accepts 5..1200 seconds; direct wrappers clamp
369
+ higher decimal values, including values beyond shell integer range, to 1200.
370
+ Wrapper-internal sub-mode limits such as Kimi inline mode's 120-second cap stay
371
+ separate. The controller bounds cumulative reviewer-lane execution with `--total-timeout`
184
372
  (default 2400 seconds, accepted range 5..3600). Setup time reduces the budget,
185
373
  and git preflight subprocesses share its deadline; direct filesystem reads
186
374
  remain subject to the host's outer timeout. Each client receives the smaller of the requested per-invocation
@@ -38,13 +38,13 @@ cannot observe or authenticate it. Hosts that need mechanical enforcement must
38
38
  provide their own trusted execution adapter; this repository preserves the
39
39
  contract and evidence shape without claiming that enforcement.
40
40
 
41
- The wrapper default timeout is 600 seconds per formal invocation, matching the
42
- accepted maximum. Measured single-lane costs on this repository's own review
41
+ The wrapper default timeout remains 600 seconds per formal invocation; the
42
+ generic accepted maximum is 1200 seconds. Measured single-lane costs on this repository's own review
43
43
  diffs were roughly 89, 247, and 419 seconds, so a smaller default silently
44
44
  converts a working reviewer into an inconclusive timeout; because a timeout is
45
45
  cascade-eligible, spending wall-clock is recoverable while a premature timeout
46
46
  quietly loses that lane's verdict. Lower it explicitly with `--timeout` when a
47
- fast bound matters more than lane coverage. It makes no separate model behavior-probe request. Challenge makes one invocation; review and consult may make up to two only through their existing bounded result-recovery paths. The wrapper rejects timeout values below 5 seconds and clamps values above 600 seconds to 600 seconds. The bound is per invocation, not per wrapper run: an outer timeout must cover `timeout` for challenge and up to `2 * timeout` for review or consult, and must never kill the wrapper and then treat the killed output as success. With defaults, allow about ten minutes for challenge and up to twenty minutes for review or consult. If the command is still running but silent, poll the same execution handle again.
47
+ fast bound matters more than lane coverage. It makes no separate model behavior-probe request. Challenge makes one invocation; review and consult may make up to two only through their existing bounded result-recovery paths. The wrapper rejects timeout values below 5 seconds and normalizes every larger decimal, including values beyond shell integer range, to at most 1200 seconds. Wrapper-internal sub-mode limits remain separate; Kimi inline review still caps argv-exposed work at 120 seconds. The bound is per invocation, not per wrapper run: an outer timeout must cover `timeout` for challenge and up to `2 * timeout` for review or consult, and must never kill the wrapper and then treat the killed output as success. With defaults, allow about ten minutes for challenge and up to twenty minutes for review or consult. If the command is still running but silent, poll the same execution handle again.
48
48
 
49
49
  The controller separately enforces a cumulative reviewer-lane budget.
50
50
  `--total-timeout` defaults to 2400 seconds and accepts 5 to 3600. Its clock
@@ -230,13 +230,22 @@ PY_DIFF
230
230
  diff_stat="Frozen review packet: $(printf '%s' "$diff_body" | wc -c | tr -d '[:space:]') bytes"
231
231
  fi
232
232
 
233
- if ! [[ "$timeout_s" =~ ^[1-9][0-9]*$ ]] || [ "$timeout_s" -lt 5 ]; then
233
+ script_dir="$(cd "$(dirname "$0")" && pwd -P)"
234
+ [ -f "$script_dir/normalize_review_timeout.sh" ] \
235
+ && [ -r "$script_dir/normalize_review_timeout.sh" ] \
236
+ && [ ! -L "$script_dir/normalize_review_timeout.sh" ] || {
237
+ emit_inconclusive_payload "Claude review helper missing: normalize_review_timeout.sh" local_tool_failure false stop_reviewer_lane
238
+ exit 2
239
+ }
240
+ # shellcheck source=normalize_review_timeout.sh
241
+ . "$script_dir/normalize_review_timeout.sh"
242
+ requested_timeout_s="$timeout_s"
243
+ if ! timeout_s="$(normalize_review_timeout "$timeout_s")"; then
234
244
  emit_inconclusive_payload "--timeout must be an integer of at least 5 seconds" invalid_input false stop_reviewer_lane
235
245
  exit 2
236
246
  fi
237
- if [ "$timeout_s" -gt 600 ]; then
238
- printf 'claude_review.sh: clamping --timeout %s to 600 seconds\n' "$timeout_s" >&2
239
- timeout_s=600
247
+ if [ "$timeout_s" != "$requested_timeout_s" ]; then
248
+ printf 'claude_review.sh: clamping --timeout %s to 1200 seconds\n' "$requested_timeout_s" >&2
240
249
  fi
241
250
  wrapper_started=$SECONDS
242
251
  if [ "$mode" = "consult" ] && [ -z "${extra//[[:space:]]/}" ]; then
@@ -266,7 +275,6 @@ if [ -z "$repo_root" ]; then
266
275
  exit 2
267
276
  fi
268
277
 
269
- script_dir="$(cd "$(dirname "$0")" && pwd -P)"
270
278
  harness_root="$(cd "$script_dir/.." && pwd -P)"
271
279
  repo_root_real="$(cd "$repo_root" && pwd -P)"
272
280
  harness_root_real="$harness_root"
@@ -52,15 +52,21 @@ while [ "$#" -gt 0 ]; do
52
52
  done
53
53
 
54
54
  case "$MODE" in review|challenge) ;; *) die_inconclusive bad_mode ;; esac
55
- [[ "$TIMEOUT" =~ ^[1-9][0-9]*$ ]] && [ "$TIMEOUT" -ge 5 ] || die_inconclusive invalid_timeout invalid_input false
56
- [ "$TIMEOUT" -le 600 ] || TIMEOUT=600
55
+ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
56
+ [ -f "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
57
+ && [ -r "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
58
+ && [ ! -L "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
59
+ || die_inconclusive timeout_normalizer_missing local_tool_failure false
60
+ # shellcheck source=normalize_review_timeout.sh
61
+ . "$SCRIPT_DIR/normalize_review_timeout.sh"
62
+ TIMEOUT="$(normalize_review_timeout "$TIMEOUT")" \
63
+ || die_inconclusive invalid_timeout invalid_input false
57
64
  [ -n "$IMPL_FAMILY" ] || die_inconclusive implementer_family_required
58
65
  [ -f "$DIFF_FILE" ] && [ -r "$DIFF_FILE" ] && [ ! -L "$DIFF_FILE" ] || die_inconclusive invalid_diff_file
59
66
  if [ -n "$REVIEW_PROFILE_FILE" ]; then
60
67
  [ -f "$REVIEW_PROFILE_FILE" ] && [ -r "$REVIEW_PROFILE_FILE" ] && [ ! -L "$REVIEW_PROFILE_FILE" ] \
61
68
  || die_inconclusive invalid_review_profile_file
62
69
  fi
63
- SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
64
70
  PARSER="$SCRIPT_DIR/parse_cli_review.py"
65
71
  TIMEOUT_CLASSIFIER="$SCRIPT_DIR/classify_timeout_exit.sh"
66
72
  SKILL_VERIFIER="$SCRIPT_DIR/verify_native_skill_binding.py"
@@ -60,10 +60,17 @@ while [ "$#" -gt 0 ]; do
60
60
  done
61
61
 
62
62
  case "$MODE" in review|challenge) ;; *) die_inconclusive bad_mode ;; esac
63
- [[ "$TIMEOUT" =~ ^[1-9][0-9]*$ ]] && [ "$TIMEOUT" -ge 5 ] || die_inconclusive invalid_timeout invalid_input false
63
+ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
64
+ [ -f "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
65
+ && [ -r "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
66
+ && [ ! -L "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
67
+ || die_inconclusive timeout_normalizer_missing local_tool_failure false
68
+ # shellcheck source=normalize_review_timeout.sh
69
+ . "$SCRIPT_DIR/normalize_review_timeout.sh"
64
70
  # Keep every invocation, including inline argv exposure, inside the controller's
65
71
  # documented direct-client ceiling.
66
- [ "$TIMEOUT" -le 600 ] || TIMEOUT=600
72
+ TIMEOUT="$(normalize_review_timeout "$TIMEOUT")" \
73
+ || die_inconclusive invalid_timeout invalid_input false
67
74
  lane_budget_started=$SECONDS
68
75
  [ "$REVIEW_SKILL_COUNT" -eq 0 ] || for review_skill in "${REVIEW_SKILLS[@]}"; do
69
76
  # The controller and binding verifier use the same package-name grammar.
@@ -86,7 +93,6 @@ if [ -n "$REVIEW_PROFILE_FILE" ]; then
86
93
  || die_inconclusive invalid_review_profile_file
87
94
  fi
88
95
 
89
- SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
90
96
  PARSER="$SCRIPT_DIR/parse_cli_review.py"
91
97
  PACKET_MCP_SERVER="$SCRIPT_DIR/kimi_packet_mcp.py"
92
98
  TIMEOUT_CLASSIFIER="$SCRIPT_DIR/classify_timeout_exit.sh"
@@ -0,0 +1,22 @@
1
+ #!/usr/bin/env bash
2
+
3
+ # Print the normalized per-invocation timeout. The input is compared by digit
4
+ # count before any arithmetic so an arbitrarily long decimal cannot overflow
5
+ # Bash's integer parser.
6
+ normalize_review_timeout() {
7
+ local value="${1:-}" digits
8
+
9
+ case "$value" in
10
+ ""|0*|*[!0-9]*) return 2 ;;
11
+ 1|2|3|4) return 2 ;;
12
+ esac
13
+
14
+ digits=${#value}
15
+ if [ "$digits" -gt 4 ] || {
16
+ [ "$digits" -eq 4 ] && [ "$value" -gt 1200 ]
17
+ }; then
18
+ printf '%s\n' 1200
19
+ else
20
+ printf '%s\n' "$value"
21
+ fi
22
+ }
@@ -168,8 +168,15 @@ command -v jq >/dev/null 2>&1 || die_inconclusive "jq_not_installed" local_tool_
168
168
  command -v timeout >/dev/null 2>&1 || die_inconclusive "timeout_not_installed" local_tool_failure false
169
169
  [ -n "$IMPL_FAMILY" ] || die_inconclusive "implementer_family_required"
170
170
  case "$MODE" in review|challenge) ;; *) die_inconclusive "bad_mode";; esac
171
- [[ "$TIMEOUT" =~ ^[1-9][0-9]*$ ]] && [ "$TIMEOUT" -ge 5 ] || die_inconclusive "invalid_timeout" invalid_input false
172
- if [ "$TIMEOUT" -gt 600 ]; then TIMEOUT=600; fi
171
+ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
172
+ [ -f "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
173
+ && [ -r "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
174
+ && [ ! -L "$SCRIPT_DIR/normalize_review_timeout.sh" ] \
175
+ || die_inconclusive "timeout_normalizer_missing" local_tool_failure false
176
+ # shellcheck source=normalize_review_timeout.sh
177
+ . "$SCRIPT_DIR/normalize_review_timeout.sh"
178
+ TIMEOUT="$(normalize_review_timeout "$TIMEOUT")" \
179
+ || die_inconclusive "invalid_timeout" invalid_input false
173
180
  if [ "$REVIEW_SKILL_COUNT" -gt 0 ]; then
174
181
  for review_skill in "${REVIEW_SKILLS[@]}"; do
175
182
  # The controller and skill-package verifier already require this exact name
@@ -213,7 +220,6 @@ if [ -n "$DIFF_FILE" ]; then
213
220
  [ "$diff_link_count" -le 1 ] || die_inconclusive "diff_file_hardlink_rejected" invalid_input false
214
221
  fi
215
222
 
216
- SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
217
223
  PARSER="$SCRIPT_DIR/parse_opencode_review.py"
218
224
  SKILL_VERIFIER="$SCRIPT_DIR/verify_native_skill_binding.py"
219
225
  [ -f "$PARSER" ] || die_inconclusive "parser_missing"