@ccoalm/ccl-skills 0.15.5 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +21 -19
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +113 -123
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/wording-only-review.md +136 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +99 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +297 -100
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +162 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +17 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_order.sh +30 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +445 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +14 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +56 -18
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +41 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +68 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/closeout-reread.md +40 -0
- package/dist/assets/release.json +31 -21
- package/package.json +1 -1
|
@@ -74,7 +74,7 @@ Positive challenge capacity opens it at index 1; budget zero is untracked.
|
|
|
74
74
|
The sole release/high-risk budget-zero exception is a controller-proved
|
|
75
75
|
`markdown-punctuation-only` review: it requires `wording_only_boundary`, permits
|
|
76
76
|
no `complete`, and rejects an author assertion alone (recipe:
|
|
77
|
-
`references/
|
|
77
|
+
`references/wording-only-review.md`). After a clean/source-refuted tracked
|
|
78
78
|
challenge, `complete` may close early and preserve unused rounds. Every result
|
|
79
79
|
exposes controller-owned `self_review_gate`; an outstanding checkpoint blocks
|
|
80
80
|
only external review or completion, not implementation or tests. Even a passed
|
|
@@ -134,17 +134,16 @@ echo "code_review_skill_dir=$CODE_REVIEW_SKILL_DIR" >&2
|
|
|
134
134
|
: "${REVIEW_CHAIN_ID:?set REVIEW_CHAIN_ID to a task-scoped chain id: letters, digits, dot, underscore, hyphen only}"
|
|
135
135
|
: "${REVIEW_STAGE:?set REVIEW_STAGE to the stage this candidate is actually at: explore, build, or release}"
|
|
136
136
|
: "${REVIEW_EVIDENCE_DIR:?set REVIEW_EVIDENCE_DIR to a durable directory you control for the per-round result rows}"
|
|
137
|
-
#
|
|
138
|
-
#
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
elif [ -n "${REVIEW_BASE:-}" ]; then
|
|
144
|
-
PACKET_ARGS=(--base "$REVIEW_BASE")
|
|
145
|
-
else
|
|
146
|
-
echo "set exactly one of REVIEW_DIFF_FILE or REVIEW_BASE" >&2; exit 1
|
|
137
|
+
# REVIEW_BASE names the candidate, REVIEW_DIFF_FILE widens what the reviewer reads,
|
|
138
|
+
# and BOTH is a widened packet that must BEGIN with the candidate (composition
|
|
139
|
+
# rules: references/staged-review-contract.md). if-blocks, not `[ -n ... ] && ...`: a
|
|
140
|
+
# trailing false test returns non-zero and `set -e` would kill the caller.
|
|
141
|
+
if [ -z "${REVIEW_DIFF_FILE:-}" ] && [ -z "${REVIEW_BASE:-}" ]; then
|
|
142
|
+
echo "set REVIEW_BASE, REVIEW_DIFF_FILE, or both" >&2; exit 1
|
|
147
143
|
fi
|
|
144
|
+
PACKET_ARGS=()
|
|
145
|
+
if [ -n "${REVIEW_BASE:-}" ]; then PACKET_ARGS+=(--base "$REVIEW_BASE"); fi
|
|
146
|
+
if [ -n "${REVIEW_DIFF_FILE:-}" ]; then PACKET_ARGS+=(--diff-file "$REVIEW_DIFF_FILE"); fi
|
|
148
147
|
# REVIEW_RUN_DIR holds the raw round-1 result only for the chain handoff; the durable
|
|
149
148
|
# per-round evidence is persisted to REVIEW_EVIDENCE_DIR, whose confidentiality you own.
|
|
150
149
|
REVIEW_RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/review-run.XXXXXX")" || exit 1
|
|
@@ -199,10 +198,11 @@ bash "$CODE_REVIEW_SKILL_DIR/scripts/review_gate.sh" \
|
|
|
199
198
|
>"$REVIEW_RUN_DIR/round2.json"
|
|
200
199
|
require_tracked_result "$REVIEW_RUN_DIR/round2.json" challenge 2
|
|
201
200
|
# Both rounds must bind the SAME candidate: the chain accepts older candidate hashes,
|
|
202
|
-
# so a
|
|
201
|
+
# so a candidate edited between rounds would otherwise be persisted as one coherent
|
|
202
|
+
# pair. Their packets may differ; each receipt records the packet it actually read.
|
|
203
203
|
python3 -c 'import json,sys; a=json.load(open(sys.argv[1])); b=json.load(open(sys.argv[2])); h=a.get("candidate_sha256"); sys.exit(0 if h and h==b.get("candidate_sha256") else 1)' \
|
|
204
204
|
"$ROUND1_RESULT_FILE" "$REVIEW_RUN_DIR/round2.json" \
|
|
205
|
-
|| { echo "round 2
|
|
205
|
+
|| { echo "round 2 bound a different candidate than round 1; rerun the pair on one frozen candidate" >&2; exit 1; }
|
|
206
206
|
cp "$REVIEW_RUN_DIR/round2.json" "$EVIDENCE_RUN_DIR/round2-challenge.json" || exit 1
|
|
207
207
|
cat "$EVIDENCE_RUN_DIR/round2-challenge.json"
|
|
208
208
|
# A secret-free diff egresses to non-Claude reviewers automatically; add
|
|
@@ -216,10 +216,10 @@ Run the script by path while keeping `--cwd` pointed at the product repository u
|
|
|
216
216
|
**The packet is the reviewer's whole world — compose it deliberately.** Review and challenge are built packet-bounded — Claude runs `--tools ""` with no `--add-dir`, and the other wrappers run in an isolated run workspace or a packet-only read surface. Treat the packet as the reviewer's whole world when deciding coverage: it is the only content bound by the packet hash and scanned before egress, so anything outside it is neither reliably visible to the reviewer nor covered by the verdict; a diff-only packet surfaces defects visible inside the changed lines and little else, and `--paths` only narrows it further. Whatever is absent from the packet is unreachable, not merely missed: a contradiction with an unchanged sibling clause, drift against a carrier outside the diff, or a silent weakening of upstream wording cannot be found by a reviewer who never saw the other side — that is the packet's shape, not the reviewer's weakness.
|
|
217
217
|
|
|
218
218
|
- Codex permits frozen-packet read/search; see [tool boundaries](references/development-completion.md#review-tools).
|
|
219
|
-
- To widen the packet, assemble it yourself and pass `--diff-file
|
|
220
|
-
- A verdict covers exactly the packet it was taken on
|
|
219
|
+
- To widen the packet, assemble it yourself and pass `--diff-file`, plus `--base` whenever the round must bind a landing candidate ([composition rules](references/staged-review-contract.md#the-packet-and-the-candidate)). It must name a regular file (no symlink or hardlink) holding text without NUL bytes. The gate hard-caps a packet at 200,000 bytes; split a larger candidate as described in the next bullet.
|
|
220
|
+
- A verdict covers exactly the packet it was taken on; the receipt records `packet_sha256` for those bytes and `candidate_sha256` for the base-derived candidate that will land, equal unless the packet was widened. A candidate too large for one packet is split by file group or risk class into a partition that still covers the whole candidate — every part in some packet, none dropped — each partition's verdict recorded against its own packet hash, and the candidate-wide claim withheld until every partition is conclusive; one partition's `no blocking findings` is never a verdict on the landing candidate. Cross-partition contradictions are unreachable by construction, so repeat the shared canonical context in every partition's packet and review anything that spans partitions as its own packet.
|
|
221
221
|
- Added context egresses to the selected reviewer exactly like the diff does, through the same credential tripwire — which catches machine-detectable secrets only. Paste rule text, carriers, and tool output; never paste credentials or material you would not send to that provider.
|
|
222
|
-
- A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane rather than editing the candidate to satisfy it.
|
|
222
|
+
- A finding that the input is insufficient to judge the change is an input defect, not a candidate defect: widen the packet and rerun that lane, keeping `--base` so it still binds the same candidate, rather than editing the candidate to satisfy it.
|
|
223
223
|
|
|
224
224
|
When intentionally reviewing `code-review` itself, override the resolver from the ccl-skills repo under review before invoking the gate:
|
|
225
225
|
|
|
@@ -280,7 +280,7 @@ For diffs over roughly 2,000 changed lines or 50 files, split review/challenge b
|
|
|
280
280
|
|
|
281
281
|
Never treat a timeout, silence, or empty output as approval or "no findings": any timeout is inconclusive, and the final status must say `inconclusive` with the timeout reason so downstream review or merge state cannot treat the missing lane as approval. A live host execution handle such as `session_id` or `cell_id` means the same command is still running; poll that exact handle to terminal exit, and never start a replacement/fallback while its process is live. Do not use a 30 second silence as a review failure — narrow diff reviews legitimately take 2-3 minutes and broad reviews about 5. Challenge makes one formal Claude invocation; review and consult may make at most two only for their existing bounded result-recovery paths. After that, mark the Claude lane inconclusive and apply fallback only if the owning gate allows it. The wrapper traps TERM/INT/HUP and emits terminal `operator_interrupt`; the gate never starts another client after an operator interruption. SIGKILL and host crashes cannot be trapped, so non-zero exit without valid JSON remains inconclusive/manual-review-required, never as success. The timeout bound is per formal invocation, not per wrapper run; outer timeouts must cover the mode's worst case and must never kill the wrapper and then treat the killed output as success. For a yielded run, the caller's lane evidence records handle type, an opaque host transcript/tool-call reference rather than a raw credential-like handle, and terminal exit status. If that handle is lost, the lane is infrastructure-inconclusive/manual-review-required and no replacement or fallback may be started or credited; a `ps`/process-tree capture and wrapper artifacts are diagnostic only and cannot reconstruct the missing terminal result. The outer host assigns this handle after launch, so this is a host-workflow obligation rather than a controller-owned field; exact enforcement requires a trusted host adapter. Recovery detail and timing formulas live in `references/timeout-auth-and-capabilities.md`.
|
|
282
282
|
|
|
283
|
-
`review_gate.sh` also enforces a cumulative reviewer-lane budget
|
|
283
|
+
`review_gate.sh` also enforces a cumulative reviewer-lane budget (`--total-timeout`), shared by git preflight but not by direct filesystem reads, which still need the host's outer timeout. Its default, range, reserved controller seconds, per-invocation division, and per-mode fail-closed minimums are in `references/timeout-auth-and-capabilities.md`. A timed-out client process group gets bounded cleanup and may cascade only while enough total budget remains. Total exhaustion returns terminal inconclusive `gate_timeout`; killed output is never a verdict.
|
|
284
284
|
|
|
285
285
|
## Auth And CLI Pitfalls
|
|
286
286
|
|
|
@@ -311,7 +311,8 @@ activation was observed; only a public event/export may populate
|
|
|
311
311
|
|
|
312
312
|
## Reference Loading
|
|
313
313
|
|
|
314
|
-
- `references/staged-review-contract.md` —
|
|
314
|
+
- `references/staged-review-contract.md` — review-plan schema, stage concerns, high-risk depth, prompt layers, challenge budget. Load before review/challenge.
|
|
315
|
+
- `references/wording-only-review.md` — the proof-bound wording-only single review. Load when claiming it.
|
|
315
316
|
- `references/manual-invocation-and-prompts.md` — manual command shape, filesystem-boundary text, and the review/challenge prompt templates. Load only when debugging or patching the wrapper or its prompt construction.
|
|
316
317
|
- `references/timeout-auth-and-capabilities.md` — wait-policy timing tables, the numbered auth-recovery procedure, per-mode tool-flag matrix, and CLI capability adoption notes. Load on timeout/auth failures or when maintaining wrapper flag adoption.
|
|
317
318
|
- `references/client-routing.md` — `review_gate.sh` client order, family exclusion, egress, Kimi/Codex boundaries, OpenCode user-model binding, and concurrency rollback. Load when running or diagnosing review/challenge routing.
|
|
@@ -341,8 +342,9 @@ In the final work summary, include:
|
|
|
341
342
|
- mode: review, challenge, complete, or consult
|
|
342
343
|
- command scope, not the full prompt unless useful
|
|
343
344
|
- result: blocking findings, no blocking findings, or inconclusive
|
|
344
|
-
- the reviewed identity — a hash of the exact diff packet reviewed, **required** whenever the reviewed content includes staged, unstaged, untracked, or generated files (later worktree edits keep the same base/head SHA, so SHA alone cannot detect the change); the base/head commit SHA alone suffices only for a clean, fully-committed candidate tree. A
|
|
345
|
+
- the reviewed identity — a hash of the exact diff packet reviewed, **required** whenever the reviewed content includes staged, unstaged, untracked, or generated files (later worktree edits keep the same base/head SHA, so SHA alone cannot detect the change); the base/head commit SHA alone suffices only for a clean, fully-committed candidate tree. A `no blocking findings` result is valid **only** for that exact content: any later edit, rebase, amend, or new commit voids it and requires a fresh run (mirrors the agentic candidate-SHA binding), and no caller — least of all one invoking this skill standalone, outside a controller tracking the head SHA — may reuse a prior pass as approval for changed content.
|
|
345
346
|
- any follow-up fixes made because of the review
|
|
347
|
+
- if `recurring_findings_design_check` fired, the `keep`/`delete`/`narrow`/`replace` decision, what it recurred across, and who ratified it
|
|
346
348
|
- if skipped or inconclusive, the exact reason
|
|
347
349
|
- for a host-yielded execution, the handle type, opaque host transcript/tool-call reference, and terminal exit status; if the handle was lost, record that infrastructure-inconclusive state, diagnostic artifacts, and that fallback was unavailable; never persist a credential-like raw handle in shared evidence
|
|
348
350
|
|
|
@@ -58,7 +58,12 @@ hand-attested plan (`review_plan_source=implementer-supplied` otherwise).
|
|
|
58
58
|
Self-review accumulates stage concerns: explore covers correctness and
|
|
59
59
|
safety; build adds failure paths, tests, and compatibility; release adds rollout
|
|
60
60
|
and operations. High-risk input raises depth to release and adds
|
|
61
|
-
`high_risk_boundary`.
|
|
61
|
+
`high_risk_boundary`. That set has one owner, and
|
|
62
|
+
`review_gate.sh --print-required-concerns --stage <stage> [--risk-tag <tag>]`
|
|
63
|
+
prints it, so a caller building a plan derives the list instead of keeping a copy
|
|
64
|
+
that silently stops satisfying the gate when the set changes. It prints what the
|
|
65
|
+
PLAN owes: the synthetic challenge slot and the wording-only boundary, which the
|
|
66
|
+
controller adds for the reviewer and never for the plan, are absent.
|
|
62
67
|
|
|
63
68
|
The serialized plan is at most 32,000 bytes and `intent` is 8..4,000
|
|
64
69
|
characters. Those are validation limits, not permission for a caller to slice a
|
|
@@ -168,6 +173,28 @@ top-level agents, commands, hooks, or MCP servers is not loaded.
|
|
|
168
173
|
Wrappers keep an explicit selected-owner count instead of testing empty Bash
|
|
169
174
|
arrays under `set -u`, preserving the no-owner lane on Bash 3.2.
|
|
170
175
|
|
|
176
|
+
### The claim-strength walk, and why the late correction is not cheaper
|
|
177
|
+
|
|
178
|
+
`claim_strength` is a required self-review concern at build and release depth. The
|
|
179
|
+
plan walks the candidate's load-bearing claims — absolutes, universals, causal
|
|
180
|
+
statements, exhaustiveness — and for each one either names evidence that would
|
|
181
|
+
survive a challenge or weakens the claim on the spot. It is owed before round 1
|
|
182
|
+
because that is the only point in a round where correcting a claim is free: once a
|
|
183
|
+
round binds the candidate, an edit inside a selected owner package voids every
|
|
184
|
+
receipt bound to it, so a sentence that claims too much costs exactly what a changed
|
|
185
|
+
predicate costs. The concern also reaches the reviewer, so a claim that survives the
|
|
186
|
+
walk comes back as a round-1 finding — inside the fix batch the round was going to
|
|
187
|
+
pay for anyway — rather than at closeout, where the remaining moves are a fresh chain
|
|
188
|
+
or leaving it standing.
|
|
189
|
+
|
|
190
|
+
There is deliberately no cheap late path. The proof-bound single review
|
|
191
|
+
(`wording-only-review.md`) refuses any changed non-punctuation character, which is
|
|
192
|
+
exactly what weakening a claim is, and an exception keyed on the author's own "this
|
|
193
|
+
edit only weakens a claim" is an assertion the controller cannot re-derive — a waiver
|
|
194
|
+
of that shape was carried here once and removed, because a predicate that approximates
|
|
195
|
+
meaning keeps admitting shapes it did not anticipate. The price stays uniform in both
|
|
196
|
+
directions; the walk is what moves the correction to where the price is zero.
|
|
197
|
+
|
|
171
198
|
## Base-derived packet input boundary
|
|
172
199
|
|
|
173
200
|
`--base` freezes the tracked diff plus every non-ignored untracked path in
|
|
@@ -184,128 +211,51 @@ commit an in-scope path when Git should represent it; or compose complete
|
|
|
184
211
|
`--diff-file` partitions when the candidate must be split. Never omit a path
|
|
185
212
|
and report the remaining packet as the whole candidate.
|
|
186
213
|
|
|
214
|
+
## The packet and the candidate
|
|
215
|
+
|
|
216
|
+
They are two objects. The **packet** is what the reviewer reads; the **candidate**
|
|
217
|
+
is what will land and what `review_ledger_binding.py` recomputes at merge time.
|
|
218
|
+
A receipt records both hashes.
|
|
219
|
+
|
|
220
|
+
They hold the same value when the packet came from `--base` alone. Pass
|
|
221
|
+
`--diff-file` **with** `--base`/`--paths` to widen what the reviewer reads while
|
|
222
|
+
the round still binds the landing candidate — the shape an evidence-gap finding
|
|
223
|
+
needs, since editing the candidate would answer an input defect with a candidate
|
|
224
|
+
change. `--diff-file` alone binds no landing; only the combined form rejects a
|
|
225
|
+
`--wording-only-proof-file`.
|
|
226
|
+
|
|
227
|
+
What makes the widened form safe is a **prefix requirement**: the packet begins
|
|
228
|
+
with the base-derived candidate, byte for byte, so nothing lands unread.
|
|
229
|
+
|
|
230
|
+
- **Append context after the candidate diff.** Putting anything before the
|
|
231
|
+
candidate fails, and that is not cosmetic: a packet preceding it with a decoy
|
|
232
|
+
diff would read as the change while the real candidate read as context.
|
|
233
|
+
Interleaving context inside the candidate fails, as does dropping any part of
|
|
234
|
+
it. The reviewer is told where the candidate ends — the profile carries
|
|
235
|
+
`candidate_bytes` and states that exactly the first N packet bytes land — so
|
|
236
|
+
appended hunks that continue or seem to revert the diff cannot pass as it.
|
|
237
|
+
- **Keep the packet file outside the repository, and put nothing else in the
|
|
238
|
+
tree while the rounds run.** The controller counts every untracked path into
|
|
239
|
+
the candidate; the binder counts only committed content minus the receipt JSON
|
|
240
|
+
a round adds. A packet file, a superseded round's receipt, or any scratch
|
|
241
|
+
artifact in the worktree therefore moves the candidate the rounds bind and the
|
|
242
|
+
binder never computes it — the mirror of committing a plain-text attestation
|
|
243
|
+
after the rounds. Bound evidence lands before the rounds, receipts after,
|
|
244
|
+
nothing else present.
|
|
245
|
+
- **Read the candidate identity, do not reconstruct it.** `--print-candidate`
|
|
246
|
+
is the authority: its base is a fork point, its paths carry the round's
|
|
247
|
+
exclusions, and it refuses an uncommitted tree.
|
|
248
|
+
- Worth adding beyond the diff — the canonical rule the changed lines must not
|
|
249
|
+
contradict, sibling clauses, the carriers restating the change, gate output.
|
|
250
|
+
|
|
251
|
+
Rounds in one chain agree on the **candidate**, not the packet, which lets a
|
|
252
|
+
later round read more than an earlier one.
|
|
253
|
+
|
|
187
254
|
## Proof-bound wording-only single review
|
|
188
255
|
|
|
189
|
-
The wording-only exception
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
Without that proof, an explore/build budget-zero review remains an ordinary
|
|
193
|
-
single review and cannot be recorded as the wording-only exception;
|
|
194
|
-
release/high-risk budget zero fails before inference.
|
|
195
|
-
|
|
196
|
-
At release depth, including depth raised by a high-risk tag, only a
|
|
197
|
-
controller-proved `markdown-punctuation-only` check may use this exception.
|
|
198
|
-
`markdown-token-replacement` remains available for explore/build budget-zero
|
|
199
|
-
review, but it cannot waive the release/high-risk challenge: byte-exact token
|
|
200
|
-
replacement does not prove that the old and new tokens have the same meaning.
|
|
201
|
-
|
|
202
|
-
The proof is a single-link regular UTF-8 JSON file of at most 16,000 bytes:
|
|
203
|
-
|
|
204
|
-
```json
|
|
205
|
-
{"schema_version":1,"candidate_sha256":"<packet-sha256>","check":{"kind":"markdown-punctuation-only"}}
|
|
206
|
-
```
|
|
207
|
-
|
|
208
|
-
The other fixed check is
|
|
209
|
-
`markdown-token-replacement`, whose `check` also contains `old_token`,
|
|
210
|
-
`new_token`, and integer `expected_count` (1..100). The controller never trusts
|
|
211
|
-
a caller-supplied pass result. It reparses the frozen packet and derives the
|
|
212
|
-
status, files, changed-line count, replacement count, and scope SHA-256.
|
|
213
|
-
|
|
214
|
-
The accepted packet is deliberately narrow: a canonical full-context unified
|
|
215
|
-
Git diff, LF-terminated, at most 200,000 bytes, changing existing regular
|
|
216
|
-
Markdown files inside exactly one existing non-linked skill package. Every
|
|
217
|
-
file's first hunk starts at line 1 so frontmatter is inspectable. Adds,
|
|
218
|
-
deletes, renames, multi-skill changes, frontmatter or `description` edits,
|
|
219
|
-
non-regular Git modes, custom/compact packets, extra context outside the diff,
|
|
220
|
-
and files without a final newline fail closed. `markdown-punctuation-only`
|
|
221
|
-
accepts only one-for-one plain-prose line replacements whose non-punctuation
|
|
222
|
-
characters remain identical; numeric tokens must additionally survive
|
|
223
|
-
byte-for-byte (deleting the dot in `5.5` is a threshold change, not
|
|
224
|
-
punctuation), and a question mark may not be added or removed (a statement
|
|
225
|
-
turned into a question is a meaning change). Lines must start at column zero
|
|
226
|
-
and contain prose; line adds/deletes, Markdown headings, lists, block quotes,
|
|
227
|
-
links, tables, inline code, fenced or indented code, and raw HTML `pre`/`code`
|
|
228
|
-
containers fail closed. `markdown-token-replacement` requires every changed
|
|
229
|
-
line pair to differ only by the named whole-token replacement, with the exact
|
|
230
|
-
total count, and rejects packets whose changed lines touch a Markdown or HTML
|
|
231
|
-
code container.
|
|
232
|
-
|
|
233
|
-
This recipe produces the exact packet and proof without a second parser or a
|
|
234
|
-
pretend verifier command. Set `WORDING_KIND=markdown-punctuation-only`, or set
|
|
235
|
-
`WORDING_KIND=markdown-token-replacement` plus `WORDING_OLD`, `WORDING_NEW`, and
|
|
236
|
-
`WORDING_COUNT`:
|
|
237
|
-
|
|
238
|
-
```bash
|
|
239
|
-
: "${CODE_REVIEW_SKILL_DIR:?set the installed code-review skill directory}"
|
|
240
|
-
: "${REPO_ROOT:?set the absolute repository root}"
|
|
241
|
-
: "${REVIEW_BASE:?set the exact base ref}"
|
|
242
|
-
: "${SKILL_NAME:?set the one existing skill package name}"
|
|
243
|
-
: "${REVIEW_STAGE:?set explore, build, or release}"
|
|
244
|
-
: "${IMPLEMENTER_FAMILY:?set the implementer model family}"
|
|
245
|
-
: "${REVIEW_PLAN_FILE:?set the absolute review-plan JSON path}"
|
|
246
|
-
: "${REVIEW_EVIDENCE_DIR:?set an existing durable private evidence directory}"
|
|
247
|
-
: "${WORDING_KIND:?set one supported wording-only check kind}"
|
|
248
|
-
|
|
249
|
-
umask 077
|
|
250
|
-
WORDING_RUN_DIR="$(mktemp -d "$REVIEW_EVIDENCE_DIR/wording-review.XXXXXX")" || exit 1
|
|
251
|
-
WORDING_DIFF="$WORDING_RUN_DIR/candidate.diff"
|
|
252
|
-
WORDING_PROOF="$WORDING_RUN_DIR/proof.json"
|
|
253
|
-
WORDING_RESULT="$WORDING_RUN_DIR/review.json"
|
|
254
|
-
|
|
255
|
-
git -C "$REPO_ROOT" diff --no-color --no-ext-diff --no-textconv --full-index \
|
|
256
|
-
--src-prefix=a/ --dst-prefix=b/ --unified=1000000 \
|
|
257
|
-
"$REVIEW_BASE" -- "skills/$SKILL_NAME" >"$WORDING_DIFF" || exit 1
|
|
258
|
-
|
|
259
|
-
python3 - "$WORDING_DIFF" "$WORDING_PROOF" "$WORDING_KIND" \
|
|
260
|
-
"${WORDING_OLD:-}" "${WORDING_NEW:-}" "${WORDING_COUNT:-0}" <<'PY'
|
|
261
|
-
import hashlib
|
|
262
|
-
import json
|
|
263
|
-
import sys
|
|
264
|
-
from pathlib import Path
|
|
265
|
-
|
|
266
|
-
diff_path, proof_path = map(Path, sys.argv[1:3])
|
|
267
|
-
kind, old, new, count = sys.argv[3:]
|
|
268
|
-
check = {"kind": kind}
|
|
269
|
-
if kind == "markdown-token-replacement":
|
|
270
|
-
check.update(old_token=old, new_token=new, expected_count=int(count))
|
|
271
|
-
elif kind != "markdown-punctuation-only":
|
|
272
|
-
raise SystemExit("unsupported WORDING_KIND")
|
|
273
|
-
payload = {
|
|
274
|
-
"schema_version": 1,
|
|
275
|
-
"candidate_sha256": hashlib.sha256(diff_path.read_bytes()).hexdigest(),
|
|
276
|
-
"check": check,
|
|
277
|
-
}
|
|
278
|
-
proof_path.write_text(
|
|
279
|
-
json.dumps(payload, ensure_ascii=False, separators=(",", ":")) + "\n",
|
|
280
|
-
encoding="utf-8",
|
|
281
|
-
)
|
|
282
|
-
PY
|
|
283
|
-
|
|
284
|
-
WORDING_RISK_ARGS=()
|
|
285
|
-
for tag in ${REVIEW_RISK_TAGS:-}; do WORDING_RISK_ARGS+=(--risk-tag "$tag"); done
|
|
286
|
-
if ! bash "$CODE_REVIEW_SKILL_DIR/scripts/review_gate.sh" \
|
|
287
|
-
--mode review --stage "$REVIEW_STAGE" --challenge-budget 0 \
|
|
288
|
-
--cwd "$REPO_ROOT" --diff-file "$WORDING_DIFF" \
|
|
289
|
-
--review-plan-file "$REVIEW_PLAN_FILE" \
|
|
290
|
-
--wording-only-proof-file "$WORDING_PROOF" \
|
|
291
|
-
${WORDING_RISK_ARGS[@]+"${WORDING_RISK_ARGS[@]}"} \
|
|
292
|
-
--implementer-family "$IMPLEMENTER_FAMILY" >"$WORDING_RESULT"; then
|
|
293
|
-
cat "$WORDING_RESULT" >&2
|
|
294
|
-
exit 1
|
|
295
|
-
fi
|
|
296
|
-
cat "$WORDING_RESULT"
|
|
297
|
-
```
|
|
298
|
-
|
|
299
|
-
A valid result carries `wording_only_proof_sha256`, controller-derived
|
|
300
|
-
`wording_only_scope.status=passed`, and a reviewed
|
|
301
|
-
`wording_only_boundary` concern. That concern independently confirms the edit
|
|
302
|
-
changes no trigger, scope, routing, validation, acceptance, rule, threshold,
|
|
303
|
-
boundary, frontmatter, description, or other meaning. If it is missing,
|
|
304
|
-
inconclusive, or reports a possible semantic change, the wording-only exception
|
|
305
|
-
does not apply: use the normal challenge and behavioral-evidence path. Any
|
|
306
|
-
candidate edit regenerates the packet and proof and requires a new review.
|
|
307
|
-
Keep the diff, proof, and result together; a digest whose source artifact was
|
|
308
|
-
deleted is not independently auditable evidence.
|
|
256
|
+
The wording-only exception — its depth limits, the proof schema, the accepted
|
|
257
|
+
packet, the recipe that produces both, and how a valid result is read — is
|
|
258
|
+
specified in `wording-only-review.md`.
|
|
309
259
|
|
|
310
260
|
## Agent review chain
|
|
311
261
|
|
|
@@ -322,8 +272,10 @@ index 1; an untracked initial review is single-round and therefore uses budget 0
|
|
|
322
272
|
candidate can never be challenged inside it. One succeeding chain may open at
|
|
323
273
|
index 1 in `challenge` mode by supplying `--predecessor-chain-result-file` — the
|
|
324
274
|
ended chain's terminal receipt — instead of an in-chain prior result. The
|
|
325
|
-
controller accepts it only when that receipt is
|
|
326
|
-
|
|
275
|
+
controller accepts it only when that receipt is the tracked round its chain ended
|
|
276
|
+
on — a terminal challenge, or a round-1 review whose own
|
|
277
|
+
arithmetic still reports its challenge unspent — carrying this chain's
|
|
278
|
+
`review_scope_sha256` and matching
|
|
327
279
|
stage/depth/risk-tags/budget, preserves the controller digest, owner-selection
|
|
328
280
|
source, and selected owner names, and binds a candidate that DIFFERS from this
|
|
329
281
|
packet: the owner-package digest is the one binding allowed to move, because its
|
|
@@ -335,6 +287,24 @@ differ from every focus the ended chain spent. The result records
|
|
|
335
287
|
satisfied self-review trigger. Succession carries history rather than resetting
|
|
336
288
|
it: consumers still sum rounds across both chains.
|
|
337
289
|
|
|
290
|
+
A chain ends where the candidate moves, and a fix applied straight after the review
|
|
291
|
+
moves the owner digest exactly as one applied after the challenge does. Requiring a
|
|
292
|
+
challenge receipt here never protected the landing candidate — the succession
|
|
293
|
+
challenge binds that either way — it only forced the challenge to be spent on a
|
|
294
|
+
candidate the author had already decided to replace. The single class that stops
|
|
295
|
+
being owed is a challenge on a candidate that will never land, which carries no
|
|
296
|
+
evidence about the one that does; every other binding is unchanged, the candidate
|
|
297
|
+
must still have moved, succession still does not compose, and this path spends
|
|
298
|
+
fewer rounds than the old one, never more. What bounds it is the receipt's own arithmetic, and that is a
|
|
299
|
+
forgery guard rather than a history check: a genuine round-1 review reads the same
|
|
300
|
+
whether its chain later ran a challenge or not, so a caller who spent the challenge
|
|
301
|
+
and presents only the review is accepted, and the successor inherits no challenge
|
|
302
|
+
focuses — a focus that chain did spend can be spent again. This is the same
|
|
303
|
+
omitted-history boundary the rest of this contract states rather than a new one, and
|
|
304
|
+
the closeout validator's ordered receipt set is where a retained challenge receipt
|
|
305
|
+
would show it; no check at the succession call site can close it, and none is
|
|
306
|
+
claimed.
|
|
307
|
+
|
|
338
308
|
The chain binds task scope, candidate identity per round, result hashes, mode,
|
|
339
309
|
status, challenge focus, controller, and selected owners. The opaque
|
|
340
310
|
`review_scope_sha256` always hashes normalized intent, acceptance, stage/depth,
|
|
@@ -390,6 +360,26 @@ checkpoint, and before a completion claim. Findings never produce a blind
|
|
|
390
360
|
review-fix-review loop: they block another reviewer call, return to implementer
|
|
391
361
|
triage, and still allow implementation, tests, and independent runnable work.
|
|
392
362
|
|
|
363
|
+
**Findings that come back are a design question.** When a round returns findings and
|
|
364
|
+
the history it carries already holds one — an earlier round of this chain, or the
|
|
365
|
+
predecessor chain a succession names — the gate adds
|
|
366
|
+
`recurring_findings_design_check` to the required triggers and
|
|
367
|
+
`decide_keep_delete_narrow_replace` to the allowed actions. It blocks nothing that
|
|
368
|
+
`findings_returned` does not already block; what it adds is the question the next
|
|
369
|
+
patch would walk past: whether the reviewed surface should exist in this shape at
|
|
370
|
+
all, answered as `keep`, `delete`, `narrow`, or `replace`, resting on the rounds and
|
|
371
|
+
findings it recurred across, and ratified by a risk owner other than the one
|
|
372
|
+
proposing it. `../../skill-extraction-workflow/SKILL.md` owns that rule; this is where
|
|
373
|
+
it fires, because the situation arises inside a chain and that skill is usually not
|
|
374
|
+
loaded there. Two findings rounds need not share a class, so the trigger over-fires
|
|
375
|
+
by design — answering an inapplicable question costs a line, and the round it saves
|
|
376
|
+
does not.
|
|
377
|
+
|
|
378
|
+
The count is what the controller can prove, and no more: the rounds of this chain plus
|
|
379
|
+
the predecessor a succession names, which is why the trigger reaches across a chain
|
|
380
|
+
break at all (Chain succession, below). Succession does not compose, so a third chain
|
|
381
|
+
opened fresh carries no history and the recurrence becomes the round's own record.
|
|
382
|
+
|
|
393
383
|
A passed final external round returns
|
|
394
384
|
`next_action=deep_self_review_before_completion` and remains
|
|
395
385
|
`completion_gated=true`. `--mode complete` accepts one exact-candidate passed
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# Proof-bound wording-only single review
|
|
2
|
+
|
|
3
|
+
The wording-only exception to `staged-review-contract.md`: when one review may
|
|
4
|
+
stand in for the review-plus-challenge pair, what the controller re-derives
|
|
5
|
+
before it will say so, and how to produce the packet and proof it accepts.
|
|
6
|
+
|
|
7
|
+
## The exception and its depth limits
|
|
8
|
+
|
|
9
|
+
The wording-only exception is one untracked `review` with
|
|
10
|
+
`challenge_budget=0`; it is not a chain, challenge, or `complete` checkpoint.
|
|
11
|
+
Supply `--wording-only-proof-file` to bind the exception to the exact packet.
|
|
12
|
+
Without that proof, an explore/build budget-zero review remains an ordinary
|
|
13
|
+
single review and cannot be recorded as the wording-only exception;
|
|
14
|
+
release/high-risk budget zero fails before inference.
|
|
15
|
+
|
|
16
|
+
At release depth, including depth raised by a high-risk tag, only a
|
|
17
|
+
controller-proved `markdown-punctuation-only` check may use this exception.
|
|
18
|
+
`markdown-token-replacement` remains available for explore/build budget-zero
|
|
19
|
+
review, but it cannot waive the release/high-risk challenge: byte-exact token
|
|
20
|
+
replacement does not prove that the old and new tokens have the same meaning.
|
|
21
|
+
|
|
22
|
+
## The proof
|
|
23
|
+
|
|
24
|
+
The proof is a single-link regular UTF-8 JSON file of at most 16,000 bytes:
|
|
25
|
+
|
|
26
|
+
```json
|
|
27
|
+
{"schema_version":1,"candidate_sha256":"<packet-sha256>","check":{"kind":"markdown-punctuation-only"}}
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
The other fixed check is
|
|
31
|
+
`markdown-token-replacement`, whose `check` also contains `old_token`,
|
|
32
|
+
`new_token`, and integer `expected_count` (1..100). The controller never trusts
|
|
33
|
+
a caller-supplied pass result. It reparses the frozen packet and derives the
|
|
34
|
+
status, files, changed-line count, replacement count, and scope SHA-256.
|
|
35
|
+
|
|
36
|
+
## The accepted packet
|
|
37
|
+
|
|
38
|
+
The accepted packet is deliberately narrow: a canonical full-context unified
|
|
39
|
+
Git diff, LF-terminated, at most 200,000 bytes, changing existing regular
|
|
40
|
+
Markdown files inside exactly one existing non-linked skill package. Every
|
|
41
|
+
file's first hunk starts at line 1 so frontmatter is inspectable. Adds,
|
|
42
|
+
deletes, renames, multi-skill changes, frontmatter or `description` edits,
|
|
43
|
+
non-regular Git modes, custom/compact packets, extra context outside the diff,
|
|
44
|
+
and files without a final newline fail closed. `markdown-punctuation-only`
|
|
45
|
+
accepts only one-for-one plain-prose line replacements whose non-punctuation
|
|
46
|
+
characters remain identical; numeric tokens must additionally survive
|
|
47
|
+
byte-for-byte (deleting the dot in `5.5` is a threshold change, not
|
|
48
|
+
punctuation), and a question mark may not be added or removed (a statement
|
|
49
|
+
turned into a question is a meaning change). Lines must start at column zero
|
|
50
|
+
and contain prose; line adds/deletes, Markdown headings, lists, block quotes,
|
|
51
|
+
links, tables, inline code, fenced or indented code, and raw HTML `pre`/`code`
|
|
52
|
+
containers fail closed. `markdown-token-replacement` requires every changed
|
|
53
|
+
line pair to differ only by the named whole-token replacement, with the exact
|
|
54
|
+
total count, and rejects packets whose changed lines touch a Markdown or HTML
|
|
55
|
+
code container.
|
|
56
|
+
|
|
57
|
+
## Producing the packet and proof
|
|
58
|
+
|
|
59
|
+
This recipe produces the exact packet and proof without a second parser or a
|
|
60
|
+
pretend verifier command. Set `WORDING_KIND=markdown-punctuation-only`, or set
|
|
61
|
+
`WORDING_KIND=markdown-token-replacement` plus `WORDING_OLD`, `WORDING_NEW`, and
|
|
62
|
+
`WORDING_COUNT`:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
: "${CODE_REVIEW_SKILL_DIR:?set the installed code-review skill directory}"
|
|
66
|
+
: "${REPO_ROOT:?set the absolute repository root}"
|
|
67
|
+
: "${REVIEW_BASE:?set the exact base ref}"
|
|
68
|
+
: "${SKILL_NAME:?set the one existing skill package name}"
|
|
69
|
+
: "${REVIEW_STAGE:?set explore, build, or release}"
|
|
70
|
+
: "${IMPLEMENTER_FAMILY:?set the implementer model family}"
|
|
71
|
+
: "${REVIEW_PLAN_FILE:?set the absolute review-plan JSON path}"
|
|
72
|
+
: "${REVIEW_EVIDENCE_DIR:?set an existing durable private evidence directory}"
|
|
73
|
+
: "${WORDING_KIND:?set one supported wording-only check kind}"
|
|
74
|
+
|
|
75
|
+
umask 077
|
|
76
|
+
WORDING_RUN_DIR="$(mktemp -d "$REVIEW_EVIDENCE_DIR/wording-review.XXXXXX")" || exit 1
|
|
77
|
+
WORDING_DIFF="$WORDING_RUN_DIR/candidate.diff"
|
|
78
|
+
WORDING_PROOF="$WORDING_RUN_DIR/proof.json"
|
|
79
|
+
WORDING_RESULT="$WORDING_RUN_DIR/review.json"
|
|
80
|
+
|
|
81
|
+
git -C "$REPO_ROOT" diff --no-color --no-ext-diff --no-textconv --full-index \
|
|
82
|
+
--src-prefix=a/ --dst-prefix=b/ --unified=1000000 \
|
|
83
|
+
"$REVIEW_BASE" -- "skills/$SKILL_NAME" >"$WORDING_DIFF" || exit 1
|
|
84
|
+
|
|
85
|
+
python3 - "$WORDING_DIFF" "$WORDING_PROOF" "$WORDING_KIND" \
|
|
86
|
+
"${WORDING_OLD:-}" "${WORDING_NEW:-}" "${WORDING_COUNT:-0}" <<'PY'
|
|
87
|
+
import hashlib
|
|
88
|
+
import json
|
|
89
|
+
import sys
|
|
90
|
+
from pathlib import Path
|
|
91
|
+
|
|
92
|
+
diff_path, proof_path = map(Path, sys.argv[1:3])
|
|
93
|
+
kind, old, new, count = sys.argv[3:]
|
|
94
|
+
check = {"kind": kind}
|
|
95
|
+
if kind == "markdown-token-replacement":
|
|
96
|
+
check.update(old_token=old, new_token=new, expected_count=int(count))
|
|
97
|
+
elif kind != "markdown-punctuation-only":
|
|
98
|
+
raise SystemExit("unsupported WORDING_KIND")
|
|
99
|
+
payload = {
|
|
100
|
+
"schema_version": 1,
|
|
101
|
+
"candidate_sha256": hashlib.sha256(diff_path.read_bytes()).hexdigest(),
|
|
102
|
+
"check": check,
|
|
103
|
+
}
|
|
104
|
+
proof_path.write_text(
|
|
105
|
+
json.dumps(payload, ensure_ascii=False, separators=(",", ":")) + "\n",
|
|
106
|
+
encoding="utf-8",
|
|
107
|
+
)
|
|
108
|
+
PY
|
|
109
|
+
|
|
110
|
+
WORDING_RISK_ARGS=()
|
|
111
|
+
for tag in ${REVIEW_RISK_TAGS:-}; do WORDING_RISK_ARGS+=(--risk-tag "$tag"); done
|
|
112
|
+
if ! bash "$CODE_REVIEW_SKILL_DIR/scripts/review_gate.sh" \
|
|
113
|
+
--mode review --stage "$REVIEW_STAGE" --challenge-budget 0 \
|
|
114
|
+
--cwd "$REPO_ROOT" --diff-file "$WORDING_DIFF" \
|
|
115
|
+
--review-plan-file "$REVIEW_PLAN_FILE" \
|
|
116
|
+
--wording-only-proof-file "$WORDING_PROOF" \
|
|
117
|
+
${WORDING_RISK_ARGS[@]+"${WORDING_RISK_ARGS[@]}"} \
|
|
118
|
+
--implementer-family "$IMPLEMENTER_FAMILY" >"$WORDING_RESULT"; then
|
|
119
|
+
cat "$WORDING_RESULT" >&2
|
|
120
|
+
exit 1
|
|
121
|
+
fi
|
|
122
|
+
cat "$WORDING_RESULT"
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
## Validating the result
|
|
126
|
+
|
|
127
|
+
A valid result carries `wording_only_proof_sha256`, controller-derived
|
|
128
|
+
`wording_only_scope.status=passed`, and a reviewed
|
|
129
|
+
`wording_only_boundary` concern. That concern independently confirms the edit
|
|
130
|
+
changes no trigger, scope, routing, validation, acceptance, rule, threshold,
|
|
131
|
+
boundary, frontmatter, description, or other meaning. If it is missing,
|
|
132
|
+
inconclusive, or reports a possible semantic change, the wording-only exception
|
|
133
|
+
does not apply: use the normal challenge and behavioral-evidence path. Any
|
|
134
|
+
candidate edit regenerates the packet and proof and requires a new review.
|
|
135
|
+
Keep the diff, proof, and result together; a digest whose source artifact was
|
|
136
|
+
deleted is not independently auditable evidence.
|