@ccoalm/ccl-skills 0.18.4 → 0.18.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-policy.md +3 -2
  2. package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +11 -9
  3. package/dist/assets/marketplace/plugins/ccl-skills/hooks/host-input.py +99 -10
  4. package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-review-covers-head.sh +34 -9
  5. package/dist/assets/marketplace/plugins/ccl-skills/hooks/skill-extraction-gate-stop.sh +14 -3
  6. package/dist/assets/marketplace/plugins/ccl-skills/hooks/skill-loading.py +39 -2
  7. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_host_input.py +11 -0
  8. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_proposed_next.py +114 -3
  9. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_review_covers_head.sh +19 -3
  10. package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_skill_loading.py +77 -0
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +12 -2
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -1
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh +65 -11
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +4 -1
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +158 -6
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +39 -1
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +18 -0
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +2 -1
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +2 -2
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +1 -1
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +1 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +1 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +10 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +2 -2
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +4 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +23 -1
  27. package/dist/assets/release.json +29 -29
  28. package/package.json +1 -1
@@ -4863,6 +4863,24 @@ out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.s
4863
4863
  check "a bare --diff-file review records no receipt" \
4864
4864
  '[ "$rc" = 0 ] && json_fields "$out" status=passed && [ ! -e "$contract_repo/.git/ccl-code-review" ]'
4865
4865
 
4866
+ # The completion checkpoint is what the pull-request reminder reads as disposed:
4867
+ # on the same whole-worktree candidate it replaces the review's receipt with a
4868
+ # passed one; a checkpoint that fails leaves the receipt as it was.
4869
+ reset_case passed unavailable unavailable
4870
+ out="$(run_contract_gate --mode review)"; rc=$?
4871
+ printf '%s\n' "$out" >"$WORK/contract-completion-review.json"
4872
+ check "the review before a completion checkpoint records its own receipt" \
4873
+ '[ "$rc" = 0 ] && [ "$(jq -r .mode "$receipt_file")" = review ]'
4874
+ reset_case passed unavailable unavailable
4875
+ out="$(run_contract_gate --mode complete --completion-review-result-file "$WORK/contract-completion-review.json")"; rc=$?
4876
+ check "a successful completion checkpoint records a passed receipt for the HEAD it bound" \
4877
+ '[ "$rc" = 0 ] && json_fields "$out" mode=complete status=passed && [ "$(jq -r .mode "$receipt_file")" = complete ] && [ "$(jq -r .status "$receipt_file")" = passed ] && [ "$(jq -r .head "$receipt_file")" = "$(git -C "$contract_repo" rev-parse HEAD)" ]'
4878
+ cp "$receipt_file" "$WORK/receipt-before-failed-completion"
4879
+ reset_case passed unavailable unavailable
4880
+ out="$(run_contract_gate --mode complete --review-plan-file "$WORK/changed-review-plan.json" --completion-review-result-file "$WORK/contract-completion-review.json")"; rc=$?
4881
+ check "a failed completion checkpoint leaves the receipt untouched" \
4882
+ '[ "$rc" = 2 ] && cmp -s "$receipt_file" "$WORK/receipt-before-failed-completion"'
4883
+
4866
4884
  swap_out="$(PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY' 2>&1
4867
4885
  import os, sys, review_gate
4868
4886
  root = os.path.realpath(os.path.join(sys.argv[1], "gitdir-swap"))
@@ -99,7 +99,8 @@ At the start of the next turn, recover intent in this order:
99
99
 
100
100
  On hosts providing a current final message, `proposed-next-stop.sh` returns one bounded Stop reminder when the assistant declares a non-status `proposed-next:` action. Recheck the active request: execute a runnable, already-authorized action in the same turn; otherwise preserve explicit stop, planning-only and status-only scope, or state the concrete decision/resource/authority blocker. Missing labels with observable delivery evidence retain their formatting reminder. A status-only marker without another action declaration, quoted example, complete machine artifact, unsupported payload or host `stop_hook_active` retry does not trigger a continuation reminder.
101
101
 
102
- - Do not request continuation for `blocked:` with a concrete explanation or `none` with a dash-separated status explanation. Mixed status/action markers still require reconciliation.
102
+ - A stop that waits on the user gets one bounded decision recheck instead: a `blocked:` handoff, a `none` explanation naming an approval, confirmation, decision or resource wait, or a last prose line asking permission to continue, either before any handoff label or, without a label, after observable delivery or edits. The recheck names the real blockers (missing credentials or authority; a fact unavailable from local evidence; an action the safety rules gate, such as destructive or irreversible work without recovery, production or customer data, or merge or publication outside the goal; overturning an established user direction; a material product tradeoff the evidence cannot settle) and returns security self-review, owner-skill, approach, test and naming choices and the next in-scope step to the agent. A real blocker survives it by restating `blocked:` after independent work is finished.
103
+ - Do not request continuation for `none — status only` or another `none` status explanation, and never treat `blocked:` as a continuation request; the decision recheck above is separate. Mixed status/action markers still require reconciliation.
103
104
 
104
105
  The hook recognizes declarations, not authorization or actual task completion, and cannot force the model to follow through. OpenCode idle does not expose the required final-message evidence; its Stop behavior remains unverified.
105
106
 
@@ -50,7 +50,7 @@ Turn observed experience into reusable skills without business-specific details.
50
50
  - **A retrospective over a LARGE multi-batch / multi-phase session has a second axis beyond the per-delivery chain: distinct lesson-TYPE axes that must each be covered or explicitly marked `no-new-lesson` — (a) per-artifact CONTENT lessons (the specific bug / contract value / domain rule; for research/writing/design programs this axis is the METHOD/CRAFT — how the work was done well), (b) PROGRAM/PROCESS lessons (how the multi-batch effort was structured and driven), (c) WORKFLOW/META lessons (did the retro or extraction itself recur shallow, under-trigger, or stop at the most salient content lesson), and (d) SUSTAIN lessons (what went RIGHT and how the next run reuses it — counts only with mechanism + non-luck evidence + owner routing; **axes (a)-craft and (d) read from the produced-artifact class — an enumeration driven by correction turns cannot reach them and will come back falsely empty**). Landing only the loudest content lesson and declaring the session "fully summarized / 复盘完成" is incomplete.** "LARGE" is not a vibe — it fires when the session already carries a coverage/program structure: a source register or named batch-progress standard was applied, OR the work spanned multiple explicit phases/batches/verticals. A user re-ask after a "done" claim = same-scope correction signal — classify first; never manufacture a lesson. Per-axis detail, re-ask classification, DO-CONFIRM card, `covered-through` watermark: `references/source-to-skill-extraction.md` (Task Retrospective Extraction).
51
51
  - **A long operational delivery session also needs a separate non-lesson delivery-state axis (in addition to the content/program/meta lesson axes above).** When the session changed operational delivery state across multiple repositories, branches, MRs, pipelines, releases, or deployable artifacts, the source register must carry that axis — changed artifact set, branch/worktree state, remote/MR state, CI or local verification state, cancelled/retried pipeline state, unresolved risks, and the next concrete action — before "whole-session retro complete" is claimed; closeout records either the axis rows' locator (sanitized labels in the shared landing, real per-repo evidence in scratch/private archive) or `artifact/status axis: not-applicable` with a reason. If required rows are absent, the retro can be reported only as `interim`, even when the extracted lesson text is correct. The row-family fields and closeout-evidence forms: `references/source-to-skill-extraction.md` (Task Retrospective Extraction).
52
52
  - A blocked verification item is not closed by naming the blockage. Before marking a test, device, browser, service, credential, or environment layer unavailable, attempt the normal remediation path for that layer — restart the client daemon, run the documented fallback (full rung list and sandbox-denial triage: the next ladder). Only record `unavailable` after remediation fails, with command evidence, residual risk, and the next concrete unblock action.
53
- - **A landed CONCLUSION is a hypothesis until an operation that could have falsified it has been run** — both a claim that a tool / capability / lane is unavailable, impossible, or must permanently fail-closed ("fail-closed is the safe default" does not waive the in-env attempt) and a DIAGNOSIS of why an observed failure happened — a search hit proves the text EXISTS, not that it RAN on the path that failed, so run the falsifying operation first — exercise the suspected mechanism on the failing path for an observation only IT predicts, or build a paired control differing in exactly ONE variable — **only within existing sandbox/permission, non-destructive, synthetic-target, and credential-safety boundaries** (never unsafe mutation, prod/live credentials, secret-bearing state, or a permission-boundary bypass — trading this rule for the security/authority/data-loss axis). Where no safe attempt is available after remediation the record is `pending` with remediation and residual risk — never `unavailable`, `fail-closed`, or a stated cause — and an unfalsified cause is `hypothesis`, kept off shared surfaces, because withdrawing a landed cause costs more than testing it. **When REVIEWING a change that asserts impossibility/unavailability or rests on a diagnosis, independently run the same falsification attempt before accepting it** — an inherited "it can't be done" or "this is why it broke" is hypothesis-grade (see the named-convention primary-source re-verify rule). Both forms and failure shapes: `references/validation-and-landing.md` (Behavioral Validation).
53
+ - **A landed CONCLUSION is a hypothesis until an operation that could have falsified it has been run** — a claim that a tool / capability / lane is unavailable, impossible, or must permanently fail-closed ("fail-closed is the safe default" does not waive the in-env attempt), a DIAGNOSIS of why an observed failure happened, and a **MEASUREMENT reported as a finding about the subject** — run the falsifying operation first — exercise the suspected mechanism on the failing path for an observation only IT predicts, or build a paired control differing in exactly ONE variable — **only within existing sandbox/permission, non-destructive, synthetic-target, and credential-safety boundaries** (never unsafe mutation, prod/live credentials, secret-bearing state, or a permission-boundary bypass — trading this rule for the security/authority/data-loss axis). Where no safe attempt is available after remediation the record is `pending` with remediation and residual risk — never `unavailable`, `fail-closed`, or a stated cause — and an unfalsified cause is `hypothesis`, kept off shared surfaces, because withdrawing a landed cause costs more than testing it. **When REVIEWING a change that asserts impossibility/unavailability, rests on a diagnosis, or reports a measurement, independently run the same falsification first** — an inherited "it can't be done", "this is why it broke" or "the tool measured N" is hypothesis-grade (see the named-convention primary-source re-verify rule). Forms and failure shapes: `references/validation-and-landing.md` (Behavioral Validation).
54
54
  - A blocked source read is not closed by naming the blockage. If Figma, code, document, API, or repository reads time out, return partial output, or fail transport, switch to a smaller or different read strategy before extracting rules — and when the source is **missing rather than unreadable**, change WHERE you enumerate instead. Both ladders: `references/source-to-skill-extraction.md#blocked-verification-and-source-read-remediation`. Failed or timed-out reads do not count as coverage.
55
55
  - **Large reads can lose the middle with no reliable signal — the trigger is read-OUTPUT size, so chunk proactively.** A read whose OUTPUT exceeds ~256 lines / ~10 KiB can be silently head+tail truncated (no marker guaranteed), so a single `cat`/whole-file read does not count as coverage even when it returns no error. Whenever you need a **complete** view — whole-file coverage, a no-findings/absence claim, or a load-bearing section read — chunk it under **both ~200 lines AND ~8 KiB** and confirm a mid-file section was ingested. Detail: `references/source-to-skill-extraction.md#read-in-chunks-large-reads-lose-the-middle`.
56
56
  - Think across the full delivery lifecycle before editing: product intent, design/UX, implementation, debugging, test strategy, launch acceptance, iteration feedback, team onboarding, and normal users without source access. A rule that improves only one slice while leaving another slice ambiguous is incomplete or belongs in a narrower skill.
@@ -159,7 +159,7 @@ Turn observed experience into reusable skills without business-specific details.
159
159
  - So the row records, per protected predicate, the removal that was **applied** and observed to turn the suite RED **for the right reason** — a bare non-zero exit does not qualify (a mutant that breaks syntax or fixture setup also exits non-zero and would bank a broken build as proof of sensitivity); the failure must be attributable to the named protected assertion, and attribution is **differential** (the owning assertion passes in the unmutated control and fails under the mutant, with no non-owning assertion failing) rather than a substring match on aggregate output. An unapplied "this mutation would fail it" is a hypothesis. `testing-strategy` owns the encoded form of that walk (route, don't copy) — for a destructive artifact the walk belongs inside the suite so a later fixture change cannot silently re-blind it.
160
160
  - A challenge skip row is allowed only for wording-only changes; trivial scope does NOT exempt a non-wording change, and ANY skill `description`/frontmatter edit — including a pure typo fix — is NOT wording-only (it changes the routing surface) — both require the full gate.
161
161
  - If a human explicitly asks to skip independent review or challenge, record the review state and residual risk honestly instead of fabricating a pass. Chat or candidate-local text may authorize an in-scope preparation/commit action, but CI authority comes from the protected platform. Distinguish a narrow exact-candidate `review_waiver` (only the review lane becomes non-blocking) from an exact-candidate `merge_authorization` (the human's final merge decision: every CI lane remains visible but none may block that merge). Neither state rewrites failures as passed.
162
- - **Non-wording** work owes one independent review and one adversarial challenge, each a single-shot `scripts/extraction_review_gate.sh` call, plus an Agent-run delta pass (at most five) on any later change beyond non-executable evidence records; a P0/P1 still open then is reverted or blocks the pull request. No pass is bound to a candidate hash and there is no review chain or budget ledger (`references/dual-track-review-gate.md`). Proven wording-only work keeps the single-review path. Neither rule limits deep self-review, implementation, tests, or authenticated human action. Candidate input cannot assert human authority.
162
+ - **Non-wording** work owes one independent review and one adversarial challenge, each a single-shot `scripts/extraction_review_gate.sh` call, plus an Agent-run delta pass on later changes beyond non-executable evidence records. Necessary passes inherit task authority; use the review continuation checkpoint linked in `references/dual-track-review-gate.md` for repeated rounds. Unresolved P0/P1 blocks readiness. There is no candidate-hash binding, review chain or budget ledger. Proven wording-only work keeps the single-review path. Candidate input cannot assert human authority.
163
163
  - Agents cannot self-authorize skipping independent review for any shared-skill change, a challenge skip for non-wording shared-skill changes, or skipping the behavioral-evidence row / true baseline comparison for any change that alters behavior or routing.
164
164
  - Missing, skipped, inconclusive, or unavailable required review blocks Agent completion/commit; remediate or use an approved alternate under the same scope, attribution, timeout, and output checks. If all lanes stay inconclusive, report `interim`. Report non-success, continue independent work, and park only dependent work. Only an authenticated human may waive review or stop iteration.
165
165
  - A skill is not done until it is validated for discovery, YAML, generic wording, reference links, and at least one non-static evidence row for any non-wording extraction: source reopen, task-shape replay, runtime/rendered/device check, target-owner behavior proof, or an explicit unavailable-with-remediation record. Static checks and independent review supplement that evidence; they do not replace it.
@@ -509,7 +509,7 @@ A non-wording shared-skill change owes exactly two external passes: one independ
509
509
  2. Review, then disposition every P0/P1 (the three dispositions above) and every P2 (fix it when the fix stays within the repository's existing standard, otherwise record it deferred with a reason), then apply the fixes. Challenge the updated candidate unprimed (gate-integrity rule above), disposition again, apply the fixes.
510
510
  3. Record both passes in the round's `evidence/` directory: each pass's controller result JSON, the commit it reviewed, and one disposition line per P0/P1 (format below). CI refuses a pull request that changes `skills/` or `hooks/` without at least one conclusive review result there (`scripts/check_review_evidence_present.py`); it checks presence only, never which candidate a result reviewed, and it does not check the challenge — that obligation stays with this lane.
511
511
  4. **Every post-review delta gets a delta pass, run by the Agent, never left to a human reader.** Everything committed after the last pass's reviewed commit is the post-review delta. When it changes anything other than non-executable record files in the round's own `evidence/` directory (controller results, disposition notes) — a P0/P1 fix, a P2 fix, a late edit, a register row, an executable probe, a rebase that is not path-disjoint — run a delta pass on it before claiming the round ready. The pull-request description lists each pass and the commit it reviewed, for traceability; nobody is expected to re-review the delta by hand.
512
- 5. **A delta pass reviews only the delta.** Its packet is the delta from the reviewed commit — pass `--base <reviewed commit>`, which binds it to the worktree and records the local receipt the pull-request hook reads — plus, for a fix, the original finding verbatim as an open item, asking for any P0/P1 in that delta — never a fix-claim (gate-integrity rule above). A new P0/P1 in the delta is fixed and gets one more delta pass. After five delta passes a still-open P0/P1 is not a human decision: revert the change that introduced it, or mark the pull request blocked and do not report it ready. Only an exact rollback of that change to a previously accepted state — the base or a version a pass reviewed — with its dependent changes owes no further pass; any other deletion leaves the pull request blocked. A delta pass never re-reviews unchanged content and never voids an earlier pass. Any pass uses the same adversarial framing; a softer prompt after fixes defeats it. P2/P3 findings owe a disposition (step 2), not a pass of their own.
512
+ 5. **A delta pass reviews only the delta.** Its packet is the delta from the reviewed commit — pass `--base <reviewed commit>`, which binds it to the worktree and records the local receipt the pull-request hook reads — plus, for a fix, the original finding verbatim as an open item, asking for any P0/P1 in that delta — never a fix-claim (gate-integrity rule above). A new P0/P1 in the delta is fixed and gets one more delta pass. After five delta passes, or earlier when findings recur without progress, apply the [review continuation checkpoint](../../code-review/references/development-completion.md#review-continuation-checkpoint): necessary passes inherit task authority; explicit user limits and real permission boundaries remain binding. Unresolved P0/P1 or an unreviewed delta still blocks readiness. Only an exact rollback to a previously accepted state — the base or a version a pass reviewed — with its dependent changes owes no further pass; any other deletion owes its delta pass. A delta pass never re-reviews unchanged content and never voids an earlier pass. Any pass uses the same adversarial framing; a softer prompt after fixes defeats it. P2/P3 findings owe a disposition (step 2), not a pass of their own.
513
513
 
514
514
  A rebase owes nothing only when it is path-disjoint: `git diff --name-only <old base> <new base>` shares no path with the candidate's changed files. When the target's new commits touched a file the candidate also touches — with or without a textual conflict — the combination was never reviewed, so the delta pass covers those files.
515
515
 
@@ -94,7 +94,7 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
94
94
  - Run deterministic checks and implementer self-review first, and record what each proves before invoking review/challenge (this self-review-before-review ordering applies to every non-wording shared-skill change the dual-track table requires review for, not only the rows that look high-risk): `git diff --check` proves whitespace/conflict-marker hygiene only; validators prove schema/link/routing invariants; leakage/sanitization scans prove only their configured patterns; scope checks must name the changed files or expected file set; the self-review row is conclusive only when each required field is non-empty (acceptance criteria, changed-file scope, edge/failure paths, known residual risks) and the changed-file scope equals the candidate diff's changed-file set, or explicitly explains any excluded generated/irrelevant file. Persist it before the review/challenge run in a fresh, non-overwritten task-evidence path outside the candidate diff, pass that exact file as the gate's review plan, and retain the gate result that binds its profile hash; do not edit the candidate merely to record self-review or review outcome, because that creates self-referential candidate churn. A candidate-local row is appropriate only when the row itself is a substantive deliverable under review. A plain in-place-editable MR description or scratch log is not ordering proof unless its edit history is retrievable and checked; a backfilled row is invalid and forces a rerun. If the candidate diff changes after the row is saved — a file added/removed OR the content of any listed file materially changed — refresh the row; a changed candidate does not by itself owe another external pass (the extraction review lane in `references/dual-track-review-gate.md` decides which passes are owed). Changing only the external self-review record refreshes the profile binding; it does not by itself invalidate implementation tests or the candidate packet. A missing field, "ok" placeholder, mismatched scope, or unprovable ordering makes the row inconclusive. Do not spend LLM review rounds on issues a script or implementer-side checklist can decide. If the independent pass is the first place basic scope, contract, privacy, or test issues surface, apply those findings to the diff, close the self-review gap, and rerun the deterministic gates before rerunning review/challenge; the process-defect repair is in addition to resolving the findings, not a way to discard or downgrade them.
95
95
  - Review pass: persist the complete self-review row and encode it in the review plan. For a **non-wording** lane, resolve the repository-owned `scripts/extraction_review_gate.sh` once per pass; never substitute the generic controller, scan writable plugin roots, or pass chain or budget options (the wrapper refuses them). For a strictly proven **wording-only** lane, use the generic `code-review` proof-bound single-review recipe in `code-review/references/staged-review-contract.md` and record `challenge: not-required`; require its controller-derived wording scope plus the independent `wording_only_boundary` confirmation. The gate, not this page, decides whether the wording-only single review is legal, and it may still demand the review-plus-challenge pair. Take all controller options from that runnable recipe, supplying the actual stage and exact candidate rather than an example default. Read the packet-composition rules in `references/dual-track-review-gate.md` first. Require conclusive JSON, selected-client attribution, packet/profile binding, family exclusion, and wrapper runtime evidence. When the host returns a live execution handle (`session_id`, `cell_id`, or equivalent), keep polling that exact handle until terminal exit; empty current output is progress, not a verdict, and no replacement/fallback reviewer may start while the original process is live. The result row records handle type, an opaque host transcript/tool-call reference and terminal exit status. If the handle is lost, the lane is infrastructure-inconclusive/manual-review-required and no replacement or fallback may be started or credited; process-tree and wrapper artifacts are diagnostic only. This is a procedural host obligation because the inner gate cannot observe the outer handle. Never copy a credential-like raw handle into shared evidence. `findings` is not pass; inconclusive, malformed, or free-form output stays interim. Do not add a separate behavior probe.
96
96
  - Challenge pass: for a non-wording lane, invoke `scripts/extraction_review_gate.sh --mode challenge` separately, with a focus, on the candidate after the review's fixes are applied (a local checkpoint commit is allowed, see SKILL.md). It is not bound to the review's candidate. Preserve a separate result row with the same binding, egress, attribution and conclusive checks. Review never satisfies challenge; missing or inconclusive required challenge keeps extraction interim. A wording-only lane has no challenge pass.
97
- - Treat review/challenge as batch-level gates over the landing candidate, not as a per-bullet or per-line edit loop. Apply all findings from a pass. Every commit after the last pass that changes more than non-executable record files in the round's evidence directory (register rows and scripts included) owes an Agent-run delta pass on that delta only (at most five; a P0/P1 still open after that is reverted or blocks the pull request). The pull-request description lists each pass and the commit it reviewed (`references/dual-track-review-gate.md`, extraction review lane).
97
+ - Treat review/challenge as batch-level gates over the landing candidate, not as a per-bullet or per-line edit loop. Apply all findings from a pass. Every commit after the last pass that changes more than non-executable record files in the round's evidence directory (register rows and scripts included) owes an Agent-run delta pass on that delta only. Necessary passes inherit task authority; repeated rounds use the review continuation checkpoint linked from `references/dual-track-review-gate.md`. Unresolved P0/P1 or an unreviewed delta blocks readiness. The pull-request description lists each pass and the commit it reviewed.
98
98
  - Test lane: run the full lane once before the review; after it, a fix confined to one suite reruns that suite plus `check-ccl-skills.sh`, and a controller, contract or shared-gate fix reruns the full lane.
99
99
  - Skipping a required challenge = work can only land as interim, not complete.
100
100
 
@@ -20,6 +20,7 @@ For the specific owner-dispatch case, the firing point is now **mechanically enf
20
20
 
21
21
  - `owner-dispatch` PreToolUse/Stop hooks (`hooks/owner-dispatch-guard.sh`, `hooks/owner-dispatch-stop.sh`) gate the first product-code edit and session close.
22
22
  - **Subagent extension** (delegated workers are a separate firing surface — SessionStart routing is NOT inherited by subagents, so a cold worker never sees the gate): `SubagentStart` (`hooks/subagent-start.sh`) injects a slim self-gating routing pointer, and `SubagentStop` reuses `owner-dispatch-stop.sh` (PreToolUse already fires inside subagents) with `agent_id`-scoped, actor-precise markers/cap so the invoke-owner backstop applies one level down for a worker's **path-attributable** gated Edit/Write (Bash-only and missing-baseline cases stay advisory under the session Stop / CI backstop, not SubagentStop). This closes the recurrence where `multi-agent-delegation` had to *manually* inject owners into every worker prompt (`skills/multi-agent-delegation/SKILL.md` Core Rules) and the user reminded when it was forgotten.
23
+ - **Extraction owner at the first shared-skill edit:** the default source-edit checkpoint (`hooks/skill-loading.py`) recognises a target on a ccl-skills checkout's `skills/`, `hooks/` or root `scripts/` surface (markdown included there only; the marker is probed at every such component, so a checkout under an ancestor carrying one of those names still counts; plugin-cache and `.codex` install copies excluded — the stop backstop's scope, which this round aligned) and, while `skill-extraction-workflow` is not loaded in the current context, names it and the charter-before-editing step in its one replan. Observed miss it answers: a shared-gate bug routed to its debugging owner satisfied the generic checkpoint, and the extraction backstop in `hooks/skill-extraction-gate-stop.sh` fired only at Stop, after commit, push and pull request. Limits: one replan per actor/context, Edit/Write/patch tools only (Bash-written edits still reach only the Stop backstop), and it cannot prove the charter was written.
23
24
  - `scripts/owner-dispatch/owner-dispatch.sh ci` is the host-agnostic merge backstop (engine: `scripts/owner-dispatch/owner-dispatch.sh`; rationale + safety posture: `scripts/owner-dispatch/README.md`).
24
25
 
25
26
  This is the worked example of moving an already-precise recognition-dependent gate ONTO the transition. **Honesty preserved — enforcement is partial, not total:** opt-in per product repo (`.owner-dispatch.json`), default `ask`, fail-open, Claude-hook-hard / Codex-advisory, in-session Bash-write detection is best-effort (the un-bypassable layer is CI, and only when the CI job itself is enforced), and the whole thing is a no-op on the generic Agent-Skills channel. So it raises enforcement where the plugin/hook + CI layers load, but the closeout gate + user-signal escalation remain the backstop everywhere they do not. The opt-in gap itself is now narrowed by a **closeout-acquire check** in `product-rd-workflow` (a gated multi-owner delivery whose repo lacks `.owner-dispatch.json` installs the backstop or records exempt at closeout) — so "silently absent" is caught, though still recognition-dependent at closeout, not a hard gate.
@@ -709,3 +709,13 @@ The pending classification above is superseded by the executed source comparison
709
709
  | Ordinary candidates cannot select extraction review or challenge | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | updated | Owner key `skill-extraction-workflow/SKILL.md`. A real-controller negative fixture fails before the ownership repair and passes afterward for both extraction modes, while ordinary staged review still reaches reviewer selection. The declared extraction owner in the fixture plan cannot substitute for candidate-derived evidence. |
710
710
  | Blocker and waiting-status handoffs do not request continuation | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/pre-final-continuation-gate.md#Do not request continuation for | updated | Owner key `product-rd-workflow/SKILL.md`. Native-shaped fixtures expose the extra reminder for a concrete blocked handoff or a none marker with a waiting explanation. These statuses now remain non-actionable; a separate action marker and action words that merely start with none still receive the bounded recheck. |
711
711
  | Long-history Stop checks suppress repeated incomplete notices and recover positive current-context evidence | `skill-extraction-workflow` / `product-rd-workflow` / `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:hooks/skill-extraction-gate-stop.sh | updated | `specs/146-bounded-hook-history/plan.md` binds repeated overflow, post-compaction recovery and early size-rejection cases. The strict whole-session audit remains unchanged; complete current-context evidence may prove invocation or handoff eligibility, while absence remains unknown. Atomic notice claims are isolated by session, actor and transcript identity and never suppress checks. Existing owner entrypoints remain unchanged; the executable prevention is in the shared hook helper and its subprocess regressions. |
712
+ | Necessary renewed review retains task authority across a progress checkpoint | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/code-review/references/development-completion.md#Run the necessary review of the changed candidate | updated | Owner key `code-review/SKILL.md` remains unchanged and already routes completion to this reference. Consolidation: merged task authority and repeated-review handling into the existing completion checkpoint. The former task-wide five-run stop is replaced by a progress check; explicit user limits, host permissions, individual chain ceilings, invocation timeouts and current-candidate review remain required. The extraction-policy regression fails on the prior policy and checks the canonical checkpoint and its boundaries. |
713
+ | Extraction delta review uses the same continuation decision as ordinary review | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | updated | `skill-extraction-workflow/SKILL.md`, `references/dual-track-review-gate.md` and `references/extraction-quickstart.md` route repeated passes to the code-review checkpoint. Independent review plus challenge, post-review delta coverage, finding dispositions and readiness requirements survive. This row supersedes the historical five-pass ceiling; its immutable locator is retired through the exact-row digest exception in `scripts/register-firing-path-resolution.rb`. `specs/147-review-continuation/plan.md` defines the migration and decision table. Tests reject the retired stop in operative carriers. Product workflow, testing and release owners retain their existing task-authority and verification rules. |
714
+ | A pull-request coverage reminder stays quiet only when the receipt covering HEAD recorded a passed review | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | updated | Owner key `code-review/SKILL.md` is unchanged. The hook regression is `hooks/test_remind_review_covers_head.sh`. `scripts/review_gate.py` records the receipt on a successful whole-worktree completion checkpoint, so a chain whose findings were disposed reads as passed; `hooks/remind-review-covers-head.sh` reminds on a covered findings or missing status and names the receipt path. The hook fixtures fail on the prior hook and under a copy with the status predicate removed; the controller receipt case fails under a copy with the completion write reverted. `specs/148-review-coverage-findings/plan.md` holds the decision table. |
715
+ | A source-refuted review chain closed through a completion checkpoint with finding dispositions records a passed receipt for the HEAD it bound | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/code-review/references/development-completion.md#a covered receipt with findings or no status must still be dispositioned | updated | Owner key `code-review/SKILL.md` is unchanged; `code-review/references/development-completion.md` now states that a completion checkpoint records the receipt and that a covered findings or status-less receipt still needs disposition. The regression is `scripts/test_review_client_compat.py`. The dispositions entry of `--mode complete` now has a receipt case on a committed `--base` candidate: the review and challenge receipts record findings, the completion replaces them with complete and passed for the same HEAD. Against the controller before `specs/148-review-coverage-findings/plan.md` the case fails with the receipt still at challenge and findings. The hook reminder for a covered receipt without a status no longer describes it as outstanding findings; `hooks/test_remind_review_covers_head.sh` fails on the prior wording. |
716
+ | A shared-skill or plugin-behavior edit on a ccl-skills checkout names the extraction owner at the first edit, not only at Stop | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/firing-point-placement.md#Extraction owner at the first shared-skill edit | updated | Owner key `skill-extraction-workflow/SKILL.md` is unchanged; `references/firing-point-placement.md` records the landed firing point and its limits. `hooks/skill-loading.py` adds the targeted reason, counts markdown only on the shared surface, and probes the marker at every `skills/`, `hooks/` or `scripts/` component; `hooks/skill-extraction-gate-stop.sh` gets the same every-component scope. `hooks/test_skill_loading.py` fails on the prior hook and kills six applied mutations (first-component probe, markdown clause, owner reason, plugin-cache disjunct, `.codex` disjunct, loaded-owner check), and `hooks/test_host_input.py` fails with the stop backstop restored to its prior scope, each only on its owning cases. The declared-but-unenforced gate class from the same session stays pending for `testing-strategy`: its rule exists and no mechanical firing path was found. Decision table: `specs/149-extraction-owner-first-edit/plan.md`. |
717
+ | A MEASUREMENT stated as a finding about the subject is a landed conclusion like an impossibility claim or a diagnosis, so it owes the falsifying operation before it reaches a shared surface, and its control leg varies the INSTRUMENT rather than the subject | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#A landed CONCLUSION is a hypothesis until an operation that could have falsified it has been run | updated | Owner key `skill-extraction-workflow/SKILL.md` carries the third enumerated form on both the authoring and the reviewing half; `references/validation-and-landing.md` carries the measurement control leg and its failure shape. RED-baseline (executed and recorded, not narrated): under the base text an uncalibrated instrument's number reached a user-facing report, a durable evidence document and a memory note without any operation that could have returned a different number; the two falsifying operations that later ran — a second, independently built measurement of the same quantity, and a sweep of the instrument's one undisclosed parameter — each contradicted it, and the withdrawal cost corrections on three surfaces. The adversarial challenge closed a cannot-fail control leg: the varied parameter must be one the reading could plausibly depend on, else the sweep could never have moved the number and certifies nothing. Entrypoint size and word count do not grow: the diagnosis form's search-hit illustration moved to the reference alongside the new detail |
718
+ | A review lane must not turn its own generated config into a release-vocabulary pin: an admission precondition the runtime may not recognise takes the whole lane down while proving nothing, so a generated setting a per-invocation environment variable already carries is belt and its rejection costs the setting, not the lane | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | updated | Owner key `code-review/SKILL.md` is unchanged; `code-review/scripts/AGENTS.md` widens the existing anti-pin rule from the parser judging runtime output to the config a wrapper writes, which is where the class re-entered a third time. `scripts/kimi_review.sh` regenerates a strict subset without the watcher table and revalidates when the runtime's own validator rejects it; only a second rejection is terminal, and the retry is unconditional rather than matched against the runtime's error wording. Read from the installed runtime, the environment override is consulted ahead of the config value, which is what makes the degraded path equal to the posture the default branch ships. Applied mutations on copies, each differentially attributed with the unmutated suite green at 237 checks: removing the retry reds only the fallback case; making the subset mode still write the table reds the fallback case and moves the terminal case's reason to the generation failure; dropping the per-invocation override at the capability probe alone reds all four override assertions; removing the digit-boundary guard reds only the numeric-offset case. Decision table: `specs/151-kimi-watch-and-quota-classification/plan.md`. |
719
+ | Free-form provider prose is not a status vocabulary: a probe failure is classified by what the message says happened, never by an integer it contains, because an integer there is as often an offset or a decode position as an HTTP status | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | updated | Owner key `code-review/SKILL.md` is unchanged. Three independent review rounds each broke a numeric predicate on a new message: a digit boundary excluded 4290 but not an offset of exactly 429, and requiring a status word before the number then matched `code` inside `decode`. Same-class recurrence, so the capability was removed rather than guarded a fourth time: `scripts/kimi_review.sh` now matches exhaustion wording and auth wording only, with exhaustion checked first so an auth envelope whose allowance actually ran out reports quota while one that names a limit as unavailable metadata does not. Both branches stay `die_inconclusive` and cascade-eligible, so the predicate decides the operator's reason string and never whether the lane passes. Applied mutations on copies, differentially attributed with the unmutated suite green at 244 checks: adding a bare 429 back reds the three offset cases; moving the auth branch first reds the three exhaustion cases. The round's review and challenge results are in `specs/151-kimi-watch-and-quota-classification/evidence/`. |
720
+ | Classifying a third-party CLI's free-form stderr is a predicate over a vocabulary the control does not own, so it is removed rather than guarded again: four independent review rounds each broke it on a message the previous fix had not considered, and the split only ever decided an operator hint | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | updated | Owner key `code-review/SKILL.md` is unchanged. The broken forms, one per round: a digit boundary excluded 4290 but not an offset of exactly 429; rate-limit exhaustion inside an auth envelope fell to the auth class; a required status word matched `code` inside `decode`; matching what the message said matched `hit` inside `whitelisted` and still missed `credits are exhausted`. `scripts/kimi_review.sh` now keeps only the EMFILE branch, so every other probe failure carries the one capability reason the default branch already ships — every reason involved was `die_inconclusive` and cascade-eligible, so no gate behaviour changes. The thirteen message fixtures are kept, asserting that single class, so the predicate cannot return without the diff saying so; the replacement, if wanted, is a classifier over the structured error the probe already streams, against a real sample. Unmutated suite green at 244 checks. Dispositions and the round-by-round record are in `specs/151-kimi-watch-and-quota-classification/evidence/`. |
721
+ | Stops that hand routine decisions back to the user receive one bounded decision recheck | `product-rd-workflow` / `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/pre-final-continuation-gate.md#gets one bounded decision recheck instead | updated | Owner key `product-rd-workflow/SKILL.md` is unchanged; `product-rd-workflow/references/pre-final-continuation-gate.md` and `agent-context/session-start.md` carry the rule. Baseline: a `blocked:` handoff, a `none` marker naming an approval or confirmation wait, and a closing permission question after edits all ended the turn unchecked, so agents stopped on security, ownership and next-step questions they could settle themselves; 16 new Stop-hook cases failed before the change. The hook now returns one recheck naming the real blockers (missing credentials or authority, facts unavailable locally, actions the safety rules gate, overturning an established user direction, an unsettled product tradeoff); finished `none` states stay quiet; a real blocker survives by restating `blocked:` after independent work. This narrows the earlier blocker-exemption row: blocked and waiting markers still never count as continuation requests. Host `stop_hook_active` bounds the recheck to one attempt, status-only markers stay quiet, and the reminder supplies no authorization. |
@@ -70,8 +70,8 @@ Relocated verbatim from `SKILL.md`'s `Validation & the dual-track gate` Core Rul
70
70
  - For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology — **and the firing point is BEFORE drafting the body, not only before finalizing**. Eval-first authoring for a NEW skill (or a new hard-rule section): (1) write the evaluation scenarios first — **at least three** for a new skill (both the vendor's published authoring guide and the high-star practice pack converge on three-plus scenarios before body text; a single-rule edit may scope down to that rule's own scenario); (2) run them WITHOUT the skill and record the observed failures verbatim — and for a discipline-slip failure (the agent knows the rule and skips it under pressure), capture the agent's rationalizations word-for-word: each verbatim excuse is the raw material for one rationalization-vs-reality row and one red-flag line in the skill text (the discipline-slip form in `rule-consolidation.md`'s form-by-failure table); an invented hypothetical excuse does not qualify — counter only what a run actually said, and don't add rows for excuses no run produced; **a no-skill control that does not exhibit the failure is a stop signal — do not author guidance for a failure you cannot observe** (record the null finding instead; this is the pre-draft face of "Evidence must come before new rules"); (3) draft the **minimal** content that addresses the observed failures, then re-run the same scenarios WITH the skill; (4) when a later run, review round, or live miss surfaces a NEW rationalization for an existing discipline gate, add its explicit counter row to that gate's table and re-run the tempting scenario — counter tables accrete from observed excuses across rounds, never from imagination. The code-level RED-GREEN-REFACTOR method (write the failing case first, watch a fresh agent violate the rule WITHOUT the skill, then add the skill and watch it comply) is owned by `superpowers:writing-skills` + `superpowers:test-driven-development` — **if installed, route there; otherwise apply the RED-baseline rule inline** (manually record the without-change failure and the with-change compliance). This is the skill-authoring face of **eval-driven development** (for a behavior/routing change, run the scenario before you finalize; never special-case the scenario just to make it pass) — borrow the *principle*, not a claim of production-grade eval rigor.
71
71
  - **What makes a `RED-baseline` valid is executed-and-recorded vs narrated — not recorded vs live.** Any evidence form (before-after diff, golden trace, or pressure scenario) is valid when it actually records the without-change failure AND the with-change compliance with a locator + expected-vs-actual (see `dual-track-review-gate.md`). Prose that merely *describes* an expected failure without running it is not a baseline; a pressure scenario you actually executed and recorded is.
72
72
  - **A claimed impossibility is falsified in-env before it lands (authoring and reviewing alike; relocated from `SKILL.md`).** Deferring work behind a conservative-sounding stub ("not implemented", "needs a future tested wrapper") is the avoidance form of a blocked-verification claim: it *feels* safe but ships an unverified impossibility as durable behavior — distinct from a real `unavailable`/`pending`-with-remediation+residual-risk record, which is what remains after a safe attempt was genuinely impossible (the attempt itself stays within the safety boundaries the `SKILL.md` rule names). Failure shape: a fallback reviewer lane shipped as a permanent fail-closed stub on an unverified "permission probe not implemented" premise that a ~60-second live tool run refuted, re-enabling the lane.
73
- - **The falsifying operation has exactly two admissible forms, and a search hit is neither (detail for the `SKILL.md` landed-conclusion rule).** (1) **Executed path** — run the suspected mechanism on the path that actually failed and collect an observation that only THIS cause predicts. Naming the call site is the entry bar, not the operation: **reachability alone rules out non-execution and nothing else** — a wrapper can demonstrably run while a downstream stall is what caused the failure, so "I watched the suspected code execute" clears no cause. The probe must produce a discriminating consequence (the mechanism's own signature in the output, a value only it would set, a timing or state it alone explains); if the run cannot separate the suspect from the alternatives still on the table, it has not falsified anything and form (2) is required. (2) **Single-variable control** — a paired probe whose arms differ in exactly ONE variable. Enumerate EVERY precondition of the predicate under test (each threshold, each uniqueness or vocabulary requirement, each surrounding-state assumption) and confirm both arms equal on all but the varied one; an arm differing in two preconditions yields a verdict attributable to neither, and it fails silently because the verdict still reads as decisive. A shared surface here is the source-register row, the commit or merge-request body, a durable note or memory, and the final response's causal account. The control leg a deferred registration owes (`source-to-skill-extraction.md`) is this rule's narrow instance, and the differential attribution a `RED-baseline` row owes (`dual-track-review-gate.md`) is its encoded form — both assume the arms are otherwise equal, which is the part that has to be enumerated rather than assumed.
74
- Failure shape: a gate rejected a ledger anchor; a `force_encoding(BINARY)` call found by grep in the gate script was recorded as the cause and landed in a merged PR body, a durable memory note, and a round charter. The suspected line was never on the checked path — that check resolves its blobs through a different helper — and the real cause was an unbounded string replace in the round's own edit, which had modified a pre-existing ledger row. Three paired probes built to test the theory each differed in more than one precondition (anchor length against the threshold, uniqueness within the file, vocabulary membership), so no verdict was attributable; the first of them produced the wrong cause. Withdrawing it cost corrections on three surfaces, against about a minute for the executed-path question.
73
+ - **The falsifying operation has exactly two admissible forms, and a search hit is neither (detail for the `SKILL.md` landed-conclusion rule).** (1) **Executed path** — run the suspected mechanism on the path that actually failed and collect an observation that only THIS cause predicts. Naming the call site is the entry bar, not the operation: **reachability alone rules out non-execution and nothing else** — a wrapper can demonstrably run while a downstream stall is what caused the failure, so "I watched the suspected code execute" clears no cause. The probe must produce a discriminating consequence (the mechanism's own signature in the output, a value only it would set, a timing or state it alone explains); if the run cannot separate the suspect from the alternatives still on the table, it has not falsified anything and form (2) is required. (2) **Single-variable control** — a paired probe whose arms differ in exactly ONE variable. Enumerate EVERY precondition of the predicate under test (each threshold, each uniqueness or vocabulary requirement, each surrounding-state assumption) and confirm both arms equal on all but the varied one; an arm differing in two preconditions yields a verdict attributable to neither, and it fails silently because the verdict still reads as decisive. A shared surface here is the source-register row, the commit or merge-request body, a durable note or memory, and the final response's causal account. The control leg a deferred registration owes (`source-to-skill-extraction.md`) is this rule's narrow instance, and the differential attribution a `RED-baseline` row owes (`dual-track-review-gate.md`) is its encoded form — both assume the arms are otherwise equal, which is the part that has to be enumerated rather than assumed. **For a MEASUREMENT the varied arm is the INSTRUMENT, not the subject, and it qualifies only on three counts together.** (a) **Power** — it could have returned a different reading, whether a number or a qualitative or derived verdict the tool states about the subject: an arm sharing the first's method, assumptions or error mode, or sweeping a parameter the reading does not depend on, could never have disagreed and certifies nothing, however independently built or undisclosed. (b) **Demonstration** — the record names a mover, an actual setting or input on which the arm returned a different value; "it could have disagreed" asserted without one is the same unfalsified claim a level up. (c) **Binding** — that mover moves the SAME property that was reported, at or spanning the configuration that produced the reported reading; a mover for a neighbouring property, or one only reachable far from that configuration, leaves the reported reading untested. Agreement is evidence only where disagreement was possible, demonstrated, and demonstrated here. A value that moves when the tool or its settings move is a reading of the tool, not a property of the subject; an instrument that ran to completion has demonstrated execution, not validity; and a maturity field the output declares about itself ("draft", "uncalibrated") is a hint, never the check.
74
+ Failure shape: a gate rejected a ledger anchor; a `force_encoding(BINARY)` call found by grep in the gate script was recorded as the cause and landed in a merged PR body, a durable memory note, and a round charter. The suspected line was never on the checked path — that check resolves its blobs through a different helper — and the real cause was an unbounded string replace in the round's own edit, which had modified a pre-existing ledger row. Three paired probes built to test the theory each differed in more than one precondition (anchor length against the threshold, uniqueness within the file, vocabulary membership), so no verdict was attributable; the first of them produced the wrong cause. Withdrawing it cost corrections on three surfaces, against about a minute for the executed-path question. Measurement shape: an uncalibrated detector borrowed from another corpus printed a coverage fraction; it was reported as a finding about the artifact, defended in a summary, and written into a durable evidence document and a memory note. An independently built measurement of the same quantity contradicted it, and sweeping the detector's single noise parameter moved the original figure across nearly the whole range it could take — the number had been a reading of the tool all along. Withdrawing it cost corrections on three surfaces, against one command for the second measurement.
75
75
  - **A headless code-writing RED/GREEN needs a *fair* violation-tempting scenario + an *independently-valid* objective measure — a capable agent complies on a clean task.** When the rule governs how an agent WRITES code, a clear well-specified prompt usually makes even the no-rule baseline produce compliant code (zero delta → no RED), so a clean-task baseline proves nothing. Surface the RED with **realistic inherited pressure** — a real legacy/house convention the rule must override, or a genuinely ambiguous spec — **NOT an explicit instruction to emit the anti-pattern**: leading the baseline directly into the violation launders a constructed failure into "RED" and is fabrication (the same defect as the rule above), so record why the prompt is fair and mark any direct-leading scenario synthetic/advisory, not RED. Score both runs with an **objective measure that detects the ANTI-PATTERN** — the rule's own checker counts ONLY if it was independently validated first (held-out positive/negative fixtures + a documented residual boundary, so RED genuinely fails); a same-change checker that merely whitelists the expected GREEN form makes GREEN trivially true. Run the agent **cross-model / fresh-context** so the baseline isn't primed by your session. If even the fair tempting baseline complies, that is an honest finding — the rule's marginal value is in edge/legacy cases, not the common one — record it, don't manufacture a RED.
76
76
  - **Optional real-agent RED-baseline (F4 Tier-3).** For a routing-surface or hub-skill change you can execute the baseline through the in-repo F4 Tier-3 harness (`scripts/eval-golden-trace.rb`; contract in `eval-routing.md`) instead of a hand-recorded scenario: run it manually **twice** — first on the checkout WITHOUT your change, then WITH it (the harness does **not** auto-checkout or auto-diff; you run both and preserve both reports as the without→with evidence). Caveats that keep it honest: (a) it is **advisory + non-deterministic + small-N** (a few hub traces, structured *routing* assertions only, statuses PASS/FAIL/INCONCLUSIVE) — a without-change run that returns PASS or INCONCLUSIVE does **not** establish a RED; only an actually-observed miss does; (b) prefer a **pre-existing frozen trace** — a trace you author alongside the change is self-authored evidence (the runner checks `frozen_at_sha` ancestry but does **not** detect a trace added/tuned in the same change), so the reviewer must confirm it is not trivially fail-before/pass-after; (c) it stays **optional and never mandated** — a properly executed-and-recorded pressure scenario remains a valid `RED-baseline` for most changes.
77
77
  - Create at least one pressure scenario: a realistic prompt where the agent should use the skill and avoid a known failure mode.
@@ -88,6 +88,8 @@ end
88
88
  # those exact rows (one digest, or an array when several rows cited the retired
89
89
  # locator); a new row cannot inherit it by reusing the locator.
90
90
  EXEMPT = {
91
+ "file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#After five delta passes a still-open P0/P1" =>
92
+ "147 replaces the task-wide review stop with a progress checkpoint; the superseding register row preserves required delta review",
91
93
  "file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#Disabled semantics are real, not painted" =>
92
94
  "065 replaced the combined platform walkthrough with an authority-classed claim ledger and executable delivery contract",
93
95
  "file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#predictive-back geometry routes to" =>
@@ -481,6 +483,8 @@ end
481
483
  exempt_uses = {}
482
484
  EXEMPT_USE_ALLOWANCE = 1
483
485
  EXEMPT_ROW_DIGESTS = {
486
+ "file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#After five delta passes a still-open P0/P1" =>
487
+ "44d4bf30abb863fd031f4bc47b12d1a3685062514ca3e344481f7faa1598bc5e",
484
488
  "file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#Disabled semantics are real, not painted" =>
485
489
  "729839dab2f900b92a3f7e866eb6602eb5261c112d39973d9eb470a797f8ca41",
486
490
  "file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#predictive-back geometry routes to" =>
@@ -244,8 +244,30 @@ assert re.search(r"[Ww]ording-only.{0,500}(?:single|one)[- ](?:round|review|pass
244
244
  docs["quickstart"], re.DOTALL), "quickstart lost the wording-only single-review exception"
245
245
  assert "--challenge-budget" not in docs["quickstart"], "quickstart must not hand callers the budget flag"
246
246
  dual = docs["dual-track"]
247
- for pinned in ("post-review delta", "Every post-review delta gets a delta pass", "After five delta passes", "never left to a human reader"):
247
+ for pinned in ("post-review delta", "Every post-review delta gets a delta pass", "never left to a human reader"):
248
248
  assert pinned in dual, f"dual-track lost '{pinned}'"
249
+ # The sixth necessary review inherits task authority. A convergence checkpoint
250
+ # must not become a fresh permission request, or override an explicit user limit.
251
+ completion = (root / "skills/code-review/references/development-completion.md").read_text(encoding="utf-8")
252
+ checkpoint = "#review-continuation-checkpoint"
253
+ assert checkpoint in dual, "extraction must use the canonical review continuation checkpoint"
254
+ assert "Unresolved P0/P1 or an unreviewed delta still blocks readiness" in dual, "extraction must keep unresolved findings and unreviewed changes pending"
255
+ assert "review continuation checkpoint" in docs["SKILL"], "entrypoint must route the continuation decision"
256
+ assert "review continuation checkpoint" in docs["quickstart"], "quickstart must route the continuation decision"
257
+ for label, text in {"completion": completion, **docs}.items():
258
+ for retired in ("Renewed runs stop at five", "After five delta passes a still-open", "delta pass (at most five)", "delta only (at most five"):
259
+ assert retired not in text, f"{label} restores a task-wide review permission ceiling"
260
+ for obligation in (
261
+ "Five renewed runs trigger a progress checkpoint, not an authorization request",
262
+ "Necessary in-scope review inherits the existing task authority",
263
+ "explicit user stop, count, cost or time limit",
264
+ "host permission denial",
265
+ "change the method or gather different evidence",
266
+ "tracked chain's own round ceiling",
267
+ "never reset an unchanged candidate's chain merely to obtain zero findings",
268
+ "A missing conclusive pass or unresolved P0/P1 remains pending",
269
+ ):
270
+ assert obligation in completion, f"review continuation lost '{obligation}'"
249
271
  wording_only = (root / "skills/code-review/references/wording-only-review.md").read_text(encoding="utf-8")
250
272
  assert "--wording-only-proof-file" in wording_only
251
273
  assert "--challenge-budget 0" in wording_only
@@ -1,8 +1,8 @@
1
1
  {
2
2
  "schema": 1,
3
3
  "npmPackage": "@ccoalm/ccl-skills",
4
- "version": "0.18.4",
5
- "sourceCommit": "c2455b6c9bde93b197c80139c6fa478e949a249c",
4
+ "version": "0.18.6",
5
+ "sourceCommit": "404c29b2d42bda7a427676da5ba96b782c7d2f85",
6
6
  "sourceState": "clean",
7
7
  "files": [
8
8
  {
@@ -42,12 +42,12 @@
42
42
  },
43
43
  {
44
44
  "path": "marketplace/plugins/ccl-skills/agent-context/session-policy.md",
45
- "sha256": "a54a3a1bde6a3e0691995865250d2b021eeb22ccebfa5f99a31b9447325eda55",
45
+ "sha256": "b1372ffc5458516c51a32b09194a8480c521dc58dea4f5e1a4eb4547b0262278",
46
46
  "mode": 420
47
47
  },
48
48
  {
49
49
  "path": "marketplace/plugins/ccl-skills/agent-context/session-start.md",
50
- "sha256": "ea0109f642fb331250e64bc28691439250575751165334f2b0175a9e3c286175",
50
+ "sha256": "8ce6cc34721028822a4ca0a215f747b0eb2d38b9763554a187a2827af12bf2b2",
51
51
  "mode": 420
52
52
  },
53
53
  {
@@ -82,7 +82,7 @@
82
82
  },
83
83
  {
84
84
  "path": "marketplace/plugins/ccl-skills/hooks/host-input.py",
85
- "sha256": "a124c55a9ad0f24f46cf3745fed1a0bfb2fdbfb4e0a12b0acbbf8d1dfc413ba5",
85
+ "sha256": "99de6454efb550efe70c514d81fae8ed2bdca98b11b6b529618dae669b8b7611",
86
86
  "mode": 420
87
87
  },
88
88
  {
@@ -112,7 +112,7 @@
112
112
  },
113
113
  {
114
114
  "path": "marketplace/plugins/ccl-skills/hooks/remind-review-covers-head.sh",
115
- "sha256": "4d31c4e7477af083f865611f266fbf0d2a26fd975e8e8bc09238286fe6cecdd9",
115
+ "sha256": "499bf748fcb5c995234c3074ef2b715c75d323ca809625c059dd9210863a81b4",
116
116
  "mode": 493
117
117
  },
118
118
  {
@@ -142,12 +142,12 @@
142
142
  },
143
143
  {
144
144
  "path": "marketplace/plugins/ccl-skills/hooks/skill-extraction-gate-stop.sh",
145
- "sha256": "082eef024bde4dcd842542e398e16b44e6dc5871d4fef168c49366dae020f4fe",
145
+ "sha256": "dccf213ef194ac85695916f85e32e113942929a619de533c4d7c579c38797729",
146
146
  "mode": 493
147
147
  },
148
148
  {
149
149
  "path": "marketplace/plugins/ccl-skills/hooks/skill-loading.py",
150
- "sha256": "f59162a353d905ef419a0075a2cf20063e7cd757799d4da7b9f7210e3ecd0272",
150
+ "sha256": "ba8bf014195d5d2213dde923af8101baa7f64ca1cef65142be45d8ad562cdc67",
151
151
  "mode": 493
152
152
  },
153
153
  {
@@ -177,7 +177,7 @@
177
177
  },
178
178
  {
179
179
  "path": "marketplace/plugins/ccl-skills/hooks/test_host_input.py",
180
- "sha256": "d29cd505d0551828b1c91db8490297a48ce44801e6ec6ba998c34a5b7270fa7b",
180
+ "sha256": "fe85a5a49b67d605db390bbc8366946cae2080f3405cdf3a0d06eca536fe37e1",
181
181
  "mode": 420
182
182
  },
183
183
  {
@@ -187,7 +187,7 @@
187
187
  },
188
188
  {
189
189
  "path": "marketplace/plugins/ccl-skills/hooks/test_proposed_next.py",
190
- "sha256": "936ffb3c57162cb712b51ee7dc33a775b96ef5d7d5807424f326c62a644b8638",
190
+ "sha256": "771bbc6fa63b0803567e13a8af4d4d78b8c9a73fcad05f8aa6612c56101e7416",
191
191
  "mode": 493
192
192
  },
193
193
  {
@@ -197,7 +197,7 @@
197
197
  },
198
198
  {
199
199
  "path": "marketplace/plugins/ccl-skills/hooks/test_remind_review_covers_head.sh",
200
- "sha256": "840543c0ea6efbfa322b87a87afb2018c0becdb5d513a5586fcb5d2f21f26cf7",
200
+ "sha256": "b379e5674f9f135859b192e5d75811baed852b3bd2c70332ef6b6c622b9f3cc8",
201
201
  "mode": 493
202
202
  },
203
203
  {
@@ -217,7 +217,7 @@
217
217
  },
218
218
  {
219
219
  "path": "marketplace/plugins/ccl-skills/hooks/test_skill_loading.py",
220
- "sha256": "78ca5d46dc116328a8eb0552011313dafdf20c472b0bd981937ec4f9f897f462",
220
+ "sha256": "c7d000028172049641b79935bcdf5e8837424880c2d821264e51fdd6b309c7ae",
221
221
  "mode": 493
222
222
  },
223
223
  {
@@ -347,7 +347,7 @@
347
347
  },
348
348
  {
349
349
  "path": "marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md",
350
- "sha256": "e1c0ed061a62f3c7d1bb54bb6b4a9ee48ced9da069a7c32e35b3bc706c1576c7",
350
+ "sha256": "b2af745ca8dbd64ed3e87c5e93d24ce39298ae9530ec98ad663b724b8e4d4fc6",
351
351
  "mode": 420
352
352
  },
353
353
  {
@@ -372,7 +372,7 @@
372
372
  },
373
373
  {
374
374
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md",
375
- "sha256": "1b7831a9bf483285682d7d751cdbb4bbc0eb43868fb3ea94904388f50fe50f31",
375
+ "sha256": "a9eecb32e6437afa7cf8efadb7b7abea4d3776484619f376dcd7560416d8a944",
376
376
  "mode": 420
377
377
  },
378
378
  {
@@ -417,7 +417,7 @@
417
417
  },
418
418
  {
419
419
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh",
420
- "sha256": "34881b897e8c28c883ce10ad387a5c18622bc6481f19e8905addc1ab65bdb7e2",
420
+ "sha256": "2f5ee6876b34c9ce9904624a6063acd050400081843e2b58525e3a4d2d5a4275",
421
421
  "mode": 493
422
422
  },
423
423
  {
@@ -452,7 +452,7 @@
452
452
  },
453
453
  {
454
454
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py",
455
- "sha256": "3227b42ca37a81e557bf92d726cd84fa209d7ea6ee2947e60915b5484c45272b",
455
+ "sha256": "7ad7115976a82c0c2c2bfa47c91ec1513a1ee09c29e5d1919e5aebd9027e3178",
456
456
  "mode": 493
457
457
  },
458
458
  {
@@ -487,7 +487,7 @@
487
487
  },
488
488
  {
489
489
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh",
490
- "sha256": "9c9d588cc991a6634d6e1e80e4c466537c1fe9ac611900d82d171e8fa59e9187",
490
+ "sha256": "ea3d76eb1c157368239bd3116d66e05c62aa9d57c8b38078ca9d7b04aee7a186",
491
491
  "mode": 493
492
492
  },
493
493
  {
@@ -542,7 +542,7 @@
542
542
  },
543
543
  {
544
544
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py",
545
- "sha256": "7e434a840acbeea63dd4906a8a4c4d7b8b94ecf7b8f2cc660ed059098c89d859",
545
+ "sha256": "283f4032ca54716614b6bb662635baa3c6806be3c8c9e9be0ec5071ca12fd4a6",
546
546
  "mode": 420
547
547
  },
548
548
  {
@@ -557,7 +557,7 @@
557
557
  },
558
558
  {
559
559
  "path": "marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh",
560
- "sha256": "75c3a939e659db104ef486b5e52bd926342b8388c4616d1ebc77552abf1a0c74",
560
+ "sha256": "bb054072877443d4c44c02843dc9cebdd786fb9e33248150887fb2991a96ce01",
561
561
  "mode": 493
562
562
  },
563
563
  {
@@ -1477,7 +1477,7 @@
1477
1477
  },
1478
1478
  {
1479
1479
  "path": "marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md",
1480
- "sha256": "498c56d47f9ced718173e2b537c03a767d01499b2556a8aa59893c2f39428630",
1480
+ "sha256": "c0403e25446b74a732dcd6f927d805a265fac13b9b781e995833ac153290b042",
1481
1481
  "mode": 420
1482
1482
  },
1483
1483
  {
@@ -2102,7 +2102,7 @@
2102
2102
  },
2103
2103
  {
2104
2104
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md",
2105
- "sha256": "129106a4e47a59fcce423fd58c75490e3bcc0faab599d15394afce8b70f80222",
2105
+ "sha256": "fbdc6975fcb4ac9aca32e14e0e77f6666b39d6e5491f234760218c1bc83614b6",
2106
2106
  "mode": 420
2107
2107
  },
2108
2108
  {
@@ -2132,12 +2132,12 @@
2132
2132
  },
2133
2133
  {
2134
2134
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md",
2135
- "sha256": "80241f2b188aa129bb1d008b22702ec8b1835f4c0f7c5b65c235ad41a9f04629",
2135
+ "sha256": "6e43aa5cb09f975f2258b1f62abd3c23d1e10a7ad287ff6cd5dbfb0e8e1c5466",
2136
2136
  "mode": 420
2137
2137
  },
2138
2138
  {
2139
2139
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md",
2140
- "sha256": "f9b610ab357e1fdf06bc014420c684ae5baa30bb70429fcc0f9759036dd9f658",
2140
+ "sha256": "d2945879c759f8a34175e14d08b22093e883094a4934c13c5e5b0bb25fe7ce4f",
2141
2141
  "mode": 420
2142
2142
  },
2143
2143
  {
@@ -2207,7 +2207,7 @@
2207
2207
  },
2208
2208
  {
2209
2209
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md",
2210
- "sha256": "948afdd0f0a84ce0cdc8a451e6833810a33f970658dda0ce67301c25ba614181",
2210
+ "sha256": "4694aa9cd45b4df1801067e537223badb6f127564db2eff89cedd9e8497347d6",
2211
2211
  "mode": 420
2212
2212
  },
2213
2213
  {
@@ -2232,7 +2232,7 @@
2232
2232
  },
2233
2233
  {
2234
2234
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md",
2235
- "sha256": "f5efcc2fbc7617ac703959ff4a4b36cd4929a43f174a9c613794351bfb3311e1",
2235
+ "sha256": "a3fda1e7482fd71de9db4f889d8a6911643da94f243ccba889c93c2ea46c7694",
2236
2236
  "mode": 420
2237
2237
  },
2238
2238
  {
@@ -2347,7 +2347,7 @@
2347
2347
  },
2348
2348
  {
2349
2349
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb",
2350
- "sha256": "09683a290e63e6d52782eb02368458c8a39b078367eb446d29ce6cd7f05c4e7c",
2350
+ "sha256": "93045a535caad9957284f543fada132f56187debfe306fc2868ad05725b05379",
2351
2351
  "mode": 493
2352
2352
  },
2353
2353
  {
@@ -2497,7 +2497,7 @@
2497
2497
  },
2498
2498
  {
2499
2499
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh",
2500
- "sha256": "0f474ac1861a1fdd8bf36c2c955c69f127b77ed1e5bd460b35d3d73290c1d15f",
2500
+ "sha256": "3d736839af868e0d3f168a93e87cdcf04383cd8daa18f5591cf434111409e543",
2501
2501
  "mode": 493
2502
2502
  },
2503
2503
  {
@@ -2647,7 +2647,7 @@
2647
2647
  },
2648
2648
  {
2649
2649
  "path": "marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md",
2650
- "sha256": "68bc84034b6eb0f42a9f570dfc914f05c600cc287fc524eebdf3cd68d90568ec",
2650
+ "sha256": "fd77cfa5cde2c994dbd2a532986ac55a73a662e86dd36901decdac6fa57ebf32",
2651
2651
  "mode": 420
2652
2652
  },
2653
2653
  {
@@ -3518,5 +3518,5 @@
3518
3518
  "mode": 420
3519
3519
  }
3520
3520
  ],
3521
- "snapshotHash": "6fc5965309e0fb84b2392f6c57e9eeafa87d640f213dd32a0383d1af7ccaebe9"
3521
+ "snapshotHash": "1c834267787c28801e4e6a233529d3974542ebdbf016f84c7745bd38fdbf794b"
3522
3522
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ccoalm/ccl-skills",
3
- "version": "0.18.4",
3
+ "version": "0.18.6",
4
4
  "description": "Reusable workflows that help coding agents plan, build, test, review, and release software — for Claude Code, Codex, and OpenCode",
5
5
  "keywords": ["skills", "agent-skills", "claude", "claude-code", "codex", "opencode", "agent", "ai", "ai-agents", "cli", "anthropic", "developer-tools"],
6
6
  "type": "module",