@ccoalm/ccl-skills 0.16.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +8 -7
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-review-covers-head.sh +128 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-untracked-background.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_review_covers_head.sh +104 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_untracked_background.sh +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +13 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +78 -126
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/wording-only-review.md +136 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +521 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +25 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_order.sh +30 -15
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +439 -19
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +14 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +6 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +48 -119
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +9 -9
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check_review_evidence_present.py +122 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +4 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +37 -8
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +102 -20
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +16 -27
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +3 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_review_evidence_present.sh +81 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +137 -279
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +41 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +49 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/closeout-reread.md +40 -0
- package/dist/assets/release.json +72 -52
- package/package.json +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +0 -1242
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +0 -978
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +0 -1477
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +0 -1183
|
@@ -20,13 +20,13 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
|
|
|
20
20
|
├─ d. Sanitization pass with checklist (cheap, seconds)
|
|
21
21
|
├─ e. Owner review gate per mandatory table (deep, minutes)
|
|
22
22
|
│ ├─ Strict wording-only → one independent code-review pass
|
|
23
|
-
│ └─ Non-wording → extraction_review_gate review + challenge (
|
|
23
|
+
│ └─ Non-wording → extraction_review_gate review + challenge (single-shot each)
|
|
24
24
|
├─ f. Apply fixes, re-sanitize
|
|
25
25
|
├─ g. Commit per batch on a feature branch → MR pending review (never push to main)
|
|
26
26
|
└─ h. Update charter completion log
|
|
27
27
|
│
|
|
28
28
|
4. Closeout → ~/.<host>/skills/.extraction-work/<project>-completion.md
|
|
29
|
-
|
|
29
|
+
Recorded passes + post-review delta + final state + deferred backlog
|
|
30
30
|
│
|
|
31
31
|
5. Provenance migration → ~/.<host>/.private-aliases/<project>.yaml
|
|
32
32
|
Move file keys / paths / counts / dates out of working files
|
|
@@ -91,10 +91,11 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
|
|
|
91
91
|
|
|
92
92
|
- When required: see `references/dual-track-review-gate.md` table.
|
|
93
93
|
- Choose the review tier from that table, not from intuition. Do not restate the rows locally; record the exact `dual-track-review-gate.md` table row used. Record `challenge: not-required` only when that row classifies the actual diff as challenge-not-required (for shared skills, this means strict wording-only with deterministic scope proof + independent review confirmation). Non-wording shared-skill changes cannot skip challenge.
|
|
94
|
-
- Run deterministic checks and implementer self-review first, and record what each proves before invoking review/challenge (this self-review-before-review ordering applies to every non-wording shared-skill change the dual-track table requires review for, not only the rows that look high-risk): `git diff --check` proves whitespace/conflict-marker hygiene only; validators prove schema/link/routing invariants; leakage/sanitization scans prove only their configured patterns; scope checks must name the changed files or expected file set; the self-review row is conclusive only when each required field is non-empty (acceptance criteria, changed-file scope, edge/failure paths, known residual risks) and the changed-file scope equals the candidate diff's changed-file set, or explicitly explains any excluded generated/irrelevant file. Persist it before the review/challenge run in a fresh, non-overwritten task-evidence path outside the candidate diff, pass that exact file as the gate's review plan, and retain the gate result that binds its profile hash; do not edit the candidate merely to record self-review or review outcome, because that creates self-referential candidate churn. A candidate-local row is appropriate only when the row itself is a substantive deliverable under review. A plain in-place-editable MR description or scratch log is not ordering proof unless its edit history is retrievable and checked; a backfilled row is invalid and forces a rerun. If the candidate diff changes after the row is saved — a file added/removed OR the content of any listed file materially changed — refresh the row;
|
|
95
|
-
- Review pass: persist the complete self-review row and encode it in the review plan. For a **non-wording** lane, resolve the repository-owned `scripts/extraction_review_gate.sh`
|
|
96
|
-
- Challenge pass: for a non-wording lane, invoke `scripts/extraction_review_gate.sh` separately with
|
|
97
|
-
- Treat review/challenge as batch-level gates over the landing candidate, not as a per-bullet or per-line edit loop. Apply all findings from a
|
|
94
|
+
- Run deterministic checks and implementer self-review first, and record what each proves before invoking review/challenge (this self-review-before-review ordering applies to every non-wording shared-skill change the dual-track table requires review for, not only the rows that look high-risk): `git diff --check` proves whitespace/conflict-marker hygiene only; validators prove schema/link/routing invariants; leakage/sanitization scans prove only their configured patterns; scope checks must name the changed files or expected file set; the self-review row is conclusive only when each required field is non-empty (acceptance criteria, changed-file scope, edge/failure paths, known residual risks) and the changed-file scope equals the candidate diff's changed-file set, or explicitly explains any excluded generated/irrelevant file. Persist it before the review/challenge run in a fresh, non-overwritten task-evidence path outside the candidate diff, pass that exact file as the gate's review plan, and retain the gate result that binds its profile hash; do not edit the candidate merely to record self-review or review outcome, because that creates self-referential candidate churn. A candidate-local row is appropriate only when the row itself is a substantive deliverable under review. A plain in-place-editable MR description or scratch log is not ordering proof unless its edit history is retrievable and checked; a backfilled row is invalid and forces a rerun. If the candidate diff changes after the row is saved — a file added/removed OR the content of any listed file materially changed — refresh the row; a changed candidate does not by itself owe another external pass (the extraction review lane in `references/dual-track-review-gate.md` decides which passes are owed). Changing only the external self-review record refreshes the profile binding; it does not by itself invalidate implementation tests or the candidate packet. A missing field, "ok" placeholder, mismatched scope, or unprovable ordering makes the row inconclusive. Do not spend LLM review rounds on issues a script or implementer-side checklist can decide. If the independent pass is the first place basic scope, contract, privacy, or test issues surface, apply those findings to the diff, close the self-review gap, and rerun the deterministic gates before rerunning review/challenge; the process-defect repair is in addition to resolving the findings, not a way to discard or downgrade them.
|
|
95
|
+
- Review pass: persist the complete self-review row and encode it in the review plan. For a **non-wording** lane, resolve the repository-owned `scripts/extraction_review_gate.sh` once per pass; never substitute the generic controller, scan writable plugin roots, or pass chain or budget options (the wrapper refuses them). For a strictly proven **wording-only** lane, use the generic `code-review` proof-bound single-review recipe in `code-review/references/staged-review-contract.md` and record `challenge: not-required`; require its controller-derived wording scope plus the independent `wording_only_boundary` confirmation. The gate, not this page, decides whether the wording-only single review is legal, and it may still demand the review-plus-challenge pair. Take all controller options from that runnable recipe, supplying the actual stage and exact candidate rather than an example default. Read the packet-composition rules in `references/dual-track-review-gate.md` first. Require conclusive JSON, selected-client attribution, packet/profile binding, family exclusion, and wrapper runtime evidence. When the host returns a live execution handle (`session_id`, `cell_id`, or equivalent), keep polling that exact handle until terminal exit; empty current output is progress, not a verdict, and no replacement/fallback reviewer may start while the original process is live. The result row records handle type, an opaque host transcript/tool-call reference and terminal exit status. If the handle is lost, the lane is infrastructure-inconclusive/manual-review-required and no replacement or fallback may be started or credited; process-tree and wrapper artifacts are diagnostic only. This is a procedural host obligation because the inner gate cannot observe the outer handle. Never copy a credential-like raw handle into shared evidence. `findings` is not pass; inconclusive, malformed, or free-form output stays interim. Do not add a separate behavior probe.
|
|
96
|
+
- Challenge pass: for a non-wording lane, invoke `scripts/extraction_review_gate.sh --mode challenge` separately, with a focus, on the candidate after the review's fixes are applied (a local checkpoint commit is allowed, see SKILL.md). It is not bound to the review's candidate. Preserve a separate result row with the same binding, egress, attribution and conclusive checks. Review never satisfies challenge; missing or inconclusive required challenge keeps extraction interim. A wording-only lane has no challenge pass.
|
|
97
|
+
- Treat review/challenge as batch-level gates over the landing candidate, not as a per-bullet or per-line edit loop. Apply all findings from a pass. Every commit after the last pass that changes more than non-executable record files in the round's evidence directory (register rows and scripts included) owes an Agent-run delta pass on that delta only (at most five; a P0/P1 still open after that is reverted or blocks the pull request). The pull-request description lists each pass and the commit it reviewed (`references/dual-track-review-gate.md`, extraction review lane).
|
|
98
|
+
- Test lane: run the full lane once before the review; after it, a fix confined to one suite reruns that suite plus `check-ccl-skills.sh`, and a controller, contract or shared-gate fix reruns the full lane.
|
|
98
99
|
- Skipping a required challenge = work can only land as interim, not complete.
|
|
99
100
|
|
|
100
101
|
#### 3f. Apply fixes, re-sanitize
|
|
@@ -120,7 +121,7 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
|
|
|
120
121
|
- Final state: which batches done, which deferred, which sources unavailable.
|
|
121
122
|
- Lessons: what surprised; what would change in next extraction; what to add to skill-extraction-workflow.
|
|
122
123
|
- Cost row (process toil is measured, not felt): review/challenge rounds run, findings fixed / accepted / deferred, wall-clock from charter to PR, and net body-word delta per touched entrypoint (`scripts/check-size-budget.sh` prints it). A round that grew a `severe_debt` entrypoint's references without retiring anything records that as the outcome; the next round must read this row before deciding its batch shape.
|
|
123
|
-
-
|
|
124
|
+
- Record the passes and the post-review delta in the round's `evidence/` directory and the pull-request description per `references/dual-track-review-gate.md` (Recording the passes). There is no closeout ledger and no merge-side candidate binding.
|
|
124
125
|
|
|
125
126
|
### 5. Provenance migration
|
|
126
127
|
|
|
@@ -150,8 +151,7 @@ Skip this step when nothing transferable surfaced.
|
|
|
150
151
|
| Anti-pattern grep panel | `references/recurring-anti-patterns-checklist.md` | Every commit; ~30s |
|
|
151
152
|
| `check-ccl-skills.sh` | `scripts/check-ccl-skills.sh` | Every commit; ~10s |
|
|
152
153
|
| Generic `code-review` gate | repository-owned skill | Strict wording-only independent review; ~5-10 min |
|
|
153
|
-
| `scripts/extraction_review_gate.sh` | this skill package | Non-wording review
|
|
154
|
-
| `scripts/validate_extraction_review_state.py <closeout.json>` | this skill package | Every non-wording terminal checkpoint |
|
|
154
|
+
| `scripts/extraction_review_gate.sh` | this skill package | Non-wording review and challenge, one single-shot call each; ~5-15 min each |
|
|
155
155
|
| Source-read fallback ladder | `SKILL.md` Source-read remediation | When a source read fails or times out |
|
|
156
156
|
| Sibling mini-map | `SKILL.md` Step 4 stack-specific updates | Every stack-specific change |
|
|
157
157
|
| Private alias map `audit_cmd` | `~/.<host>/.private-aliases/<project>.yaml` or process-retro profile | Every commit's R0 audit |
|
|
@@ -89,3 +89,25 @@ Firing point: **producing or first-publishing a reader-facing deliverable is an
|
|
|
89
89
|
但按该规则自己的定义,**针对外部源的缺口清单就是它所说的 findings 回合**:一旦产出,charter 就只能事后补写。
|
|
90
90
|
|
|
91
91
|
观测实例:一轮里先产出四条「外部有我们没有」的缺口,之后才 invoke 提炼工作流;改前的触发词表逐字检索该轮实际措辞得零命中。
|
|
92
|
+
|
|
93
|
+
## The review-chain case — the owning skill is not loaded where the situation arises
|
|
94
|
+
|
|
95
|
+
The same-class-recurrence rule is owned by this workflow, but the situation it governs —
|
|
96
|
+
findings coming back round after round — arises inside a `code-review` chain, where this
|
|
97
|
+
skill is typically never loaded. Naming the owner in prose therefore never made it fire.
|
|
98
|
+
The trigger sits at the transition instead:
|
|
99
|
+
|
|
100
|
+
- **The controller must raise it, not the reader:** when a round returns findings and the
|
|
101
|
+
history it carries already holds one, the gate adds `recurring_findings_design_check` to
|
|
102
|
+
that round's required self-review triggers and `decide_keep_delete_narrow_replace` to its
|
|
103
|
+
allowed actions, in the round's own envelope
|
|
104
|
+
(`../../code-review/references/staged-review-contract.md`). What discharges it is the
|
|
105
|
+
`keep` / `delete` / `narrow` / `replace` decision the owning rule defines, ratified by a
|
|
106
|
+
risk owner other than the one proposing it.
|
|
107
|
+
- **What it counts is bounded by what a receipt carries:** the chain the controller is
|
|
108
|
+
handed, plus the predecessor a succession names. Succession does not compose, so a third
|
|
109
|
+
chain opened fresh carries no history and the controller claims none; from there the
|
|
110
|
+
recurrence is the round's own record to keep.
|
|
111
|
+
- **It over-fires by design:** two findings rounds need not share a risk class, so the
|
|
112
|
+
question is sometimes inapplicable. Answering an inapplicable question costs a line; the
|
|
113
|
+
round a missed design question costs does not.
|
|
@@ -679,3 +679,21 @@ The pending classification above is superseded by the executed source comparison
|
|
|
679
679
|
| A value used as a redaction NEEDLE is rejected when it is only structure: a home directory of `/` is a legitimate environment and a catastrophic needle, because replacing it rewrites every separator in the text and disables every rule that runs after it | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. Raised by the landing review against the normalization the previous round added: gathering every spelling of the home directory is right, but a spelling that carries no content is not a path to elide. With `HOME=/` -- root, or an arbitrary-uid container -- the needle set contained `/`, the replacement ran before the URL rules, and the excerpt came out mangled with its credentials intact. The general shape is that a needle derived from the environment needs a content test, not only a presence test. RED-baseline (applied): a row invoking the wrapper with `HOME=/` reds without the content test and greens with it, and removing only that test reds that row alone. |
|
|
680
680
|
| A filter over free text in a persisted artifact is replaced by having no free text: "nothing secret-shaped survives" is not decidable over arbitrary text, so an adversarial reviewer can always spell one more escape, and the terminal state is a constant the input cannot influence rather than a filter that keeps growing | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/codex_review.sh | `updated` | Owner key `code-review/SKILL.md`. The measurement is the row: eight review chains after the input was already narrowed to CLI-authored error messages, each found a different escape -- an unlisted key name, an assignment form, URL userinfo, a password containing the separator, a fixture whose own shape tripped a neighbouring gate, an escaped quote closing a quoted value early, an uppercase scheme, a separator-only home used as a needle. Every one was real and none was derivable from the previous one. Two intermediate diagnoses were wrong on the way and are recorded above: swapping a key-name list for an assignment-shape rule was called an invariant change and was another enumeration, and narrowing the input was called sufficient when it only slowed the rate. What ends the class is that the receipt now carries a constant and the transport's output stays in the preserved run directory. The property is stated as equality with that constant, which a test can hold, instead of the absence of a list of shapes, which no test can. RED-baseline (applied): echoing the extracted message into the receipt reds the invariant row, and the unmutated control is green. Cost, recorded because the next round should be able to weigh it: each chain was roughly twelve minutes of wall clock, and the merge gate accepts no open challenge finding, so there was no landing state that carried the residue. |
|
|
681
681
|
| A fixture string is read by every scanner in the repository, not only by the suite it belongs to, so it is chosen to be inert under all of them: a host:port that exists only inside a quoted test payload still reads as a listening port to a lane-isolation scanner | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_cli_review_wrappers.sh | `updated` | Owner key `code-review/SKILL.md`. Fourth guard this round to reject the round's own test data, after the credential scanner, the egress tripwire and the public-sanitization gate. The userinfo fixture carried a port it never needed, and the parallel-lane isolation scanner reads any host:port in a lane member as evidence that concurrent suites could race on it. Dropping the port exercises the same wrapper behaviour. Recorded as one rule with the three before it: the cost of learning this one guard at a time was a full verification cycle each, and the cheaper order is to sweep every local gate after touching a fixture, before spending a review chain on the candidate. RED-baseline (applied): `test_lane_isolation.py` reds on the ported form and greens on the bare host, with the wrapper suite green either way -- which is why the suite alone was not evidence. |
|
|
682
|
+
| A rule that lives in a skill the situation never loads does not fire, however well it is written: the controller that runs the review chain raises the design question itself, inside that round's own envelope, and counts only the history a receipt carries | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md`. Both halves were registered deferrals, reproduced against the current controller before any code was written, each probe paired with a control leg. `recurring_findings_design_check` joins the required self-review triggers, with `decide_keep_delete_narrow_replace` among the allowed actions, whenever a findings round's own history already holds one — an earlier round of this chain, or the predecessor a succession names. `claim_strength` becomes a required self-review concern at build and release depth, owed before round 1 because that is the only point in a round where correcting an overstated claim is free. Applied-mutation RED baseline, differential: control 274 ok / 0 FAIL; removing the succession carry fails exactly `findings after a predecessor chain that also returned findings raise the design check` (273 ok / 1 FAIL); exempting `claim_strength` from the plan-coverage check fails exactly `a plan that skips the claim-strength walk fails before any provider runs` (273 ok / 1 FAIL); no non-owning assertion moves in either mutant. Two control legs ship inside the suite so a FIRST findings round is proved not to raise the trigger. The additions crossed the 500-line reference cap, so the wording-only exception moved verbatim into `code-review/references/wording-only-review.md` (the ratchet's own split-by-subtopic remedy) and the entrypoint's cumulative-budget paragraph became a pointer whose every number already lives in `code-review/references/timeout-auth-and-capabilities.md`; entrypoint body words end 8 below base. Round-1 review (kimi) returned one P2 on exactly that relocation — the numbers were delegated to a file outside the packet with nothing pinning them there — so the three literals are now contract anchors; the applied deletion mutation on one of them turns the anchor gate red (1 of 13) and the unmutated control runs green. Adding a required concern is a repository-wide compatibility event: five suites carried review-plan fixtures that omitted it and only the full lane found them, so the lane is run to green before a review chain opens rather than after. Supporting evidence: `code-review/scripts/review_gate.py`, `code-review/scripts/test_review_gate.sh`, `code-review/references/staged-review-contract.md`. |
|
|
683
|
+
| The same-class-recurrence rule states where it fires and what the firing controller is allowed to count, because the owner skill is not loaded at the transition where the situation arises | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/firing-point-placement.md#The controller must raise it, not the reader | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The rule is owned here, but the situation it governs arises inside a review chain where this skill is never loaded, so naming the owner in prose never made it fire; the firing point moves onto the transition and the statement of it lands in `references/firing-point-placement.md`, the reference the entrypoint already points at for firing-point mechanics, because the attention-budget ratchet holds this entrypoint at its base measure. The reference records what discharges the trigger (the keep/delete/narrow/replace decision, ratified by a risk owner other than the one proposing it), the bound on what the controller may count (this chain plus the predecessor a succession names; succession does not compose, so a third chain opened fresh carries none), and that the trigger over-fires by design. RED baseline is the controller's own suite: with the succession carry removed, the chain this text describes stops raising the check and exactly that assertion fails (273 ok / 1 FAIL against a 274 ok / 0 FAIL control), with no other assertion moving. |
|
|
684
|
+
| A gate that forces evidence to be bought on an artifact that will never land is charging for the wrong thing: a chain ends where the candidate moves, so the round it ended on is whichever round came last -- and requiring that round to be a challenge only moved the spend, never the proof | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md`. A succession may now carry a chain whose terminal receipt is its REVIEW, not only its challenge. Cost of the old shape, recorded as the author's measurement of a prior round rather than as anything this packet can reproduce: a fix applied straight after a review ended the chain by moving the owner digest, and reaching the post-fix candidate then required spending the chain's challenge on the pre-fix candidate first, so challenges were bought on candidates that never landed. Exempted class, stated behaviourally: a challenge on a candidate that will never land, which carries no evidence about the one that does. Every other binding holds -- the candidate must still have moved, succession still does not compose, the per-chain budget is untouched, and this path spends fewer rounds than the old one. What bounds it is the receipt's own arithmetic, and review named that for what it is: a FORGERY guard, not a history check. A genuine round-1 review reads the same whether its chain later ran a challenge or not, so a caller who spent the challenge and presents only the review is accepted, and the successor inherits no challenge focuses -- one that chain did spend can be spent again. The round had claimed that refusal in its acceptance and 'proved' it with a receipt shape the controller never emits; the claim is withdrawn rather than mechanised, because no check at the succession call site can close an omitted-history gap that the rest of this contract already declares. Controller comment, contract text, probe name and acceptance criterion all now say only what is enforced. Applied-mutation RED baseline, differential and attributed to the guard rather than to a diagnostic string: reverting the whole change reds the new probes only because the base rejects EVERY review predecessor, which proves nothing about the arithmetic guard, so the recorded mutation disables that guard ALONE -- review predecessors still admitted, their chain-ended arithmetic no longer checked. Under it exactly one assertion fails, `a succession rejects a forged review receipt whose own arithmetic says its chain is spent`, and it fails because the succession was ACCEPTED; the unmutated control runs the suite green. The same round adds `--print-required-concerns`. Parity is proved over ANSWERS and over REFUSALS, the second only after challenge found the export answering for inputs the enforcer rejects -- a risk tag containing whitespace, an empty tag, an over-long tag -- which is the divergence the export exists to remove, reproduced against the shipped binary and now refused identically on both sides -- asserted on BOTH, after challenge observed the first parity probes invoking only the printer, so removing the enforcer's own rejection would have left them green. Removing it in an isolated clone now reds exactly those three assertions and nothing else. The clone matters: three earlier attempts mutated the worktree and restored it at the end of the same command, and one such restore -- from a backup another still-running task had taken while the file was already mutated -- put a controller with its tag validation stripped back into the tree, caught only because the task's output was shorter than expected. Destructive probes run on a copy. Challenge then found the parity checks themselves discarding exit status: this suite runs without errexit, so a command substitution swallows it and a printer that emitted the right concerns before failing would still have satisfied a non-empty check. The valid calls now assert their status, and a mutant that prints correctly then returns non-zero reds exactly those two assertions. The same failure-propagation blind spot appeared twice in one round -- the first parity probe had instead enabled errexit, aborting the suite at its first non-zero command with zero FAIL lines -- so both directions are now covered by assertions rather than by shell defaults. Asked to sweep the class rather than patch the instance, the next challenge found the remaining one -- a printer call NESTED inside a comparison, whose status no assignment could capture. All four printer invocations in the suite now capture status; a mutant failing only for explicit release depth with no risk tag reds exactly the assertion that covers it, and nothing else. The answer direction is proved in both senses: a plan built from it is accepted, and dropping each printed concern in turn turns the gate red -- at BOTH depths, after the challenge observed that proving it only at build left the release/high-risk branch free to diverge with every test green. |
|
|
685
|
+
| A consumer that keeps its own copy of a set the controller owns drifts the moment that set changes, and the suite runner aborts at its first failing target so the drift surfaces rounds later -- while relocating a working pin mechanism to buy faster feedback costs more than the latency it buys | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. This owner's wrapper suite derives the required concern set from the controller that enforces it instead of holding a copy. RED-baseline, paired control differing in exactly one variable: with one concern added to the controller's build stage, the base fixture fails `self_review_incomplete` on the no-independent-reviewer assertion (rc 1) while the derived fixture on this candidate passes (rc 0); with the mutation removed the derived fixture passes again. Second thread, withdrawn rather than landed: the round first relocated twenty prose pins from a slow suite into the fast registry, and five consecutive challenge findings landed inside the matching normalisation that relocation required -- this repo's own cue to question the capability instead of patching it again -- while the change additionally applied looser whitespace-insensitive matching to fourteen pre-existing anchors that never asked for it. The four mechanism files are byte-identical to base. What lands from that thread is the write-side norm that sends an author to BOTH pin surfaces, with a command for each that was run before it was written down -- the first draft shipped a registry lookup that scanned the wrong scripts directory and returned nothing, which challenge caught; the replacement lists a file's anchor ids and was verified against a file that has them -- before a budget-funded trim; the latency that motivated the relocation is left to its own change, where wiring the owner suite into the fast gate buys the same feedback with no new matching semantics. |
|
|
686
|
+
| 改写既有文档时新写的句子不得把用词水位抬到原文之上——病根是编辑落笔用的是自己的词库而不是宿主文档的;查法是把本轮新句单独拎出、逐个术语查它在原文里出现过没有、没出现的换成原文说法或当场白话解释 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/SKILL.md#改写时新句用词不得高出原文水位、抬高读者门槛 | `updated` | Owner key `tighten-doc/SKILL.md`。Observed failure:一篇面向业务读者的大白话协作文档被逐句就地修订约 25 次,每一次替换单独看都正确;收尾回读的既有触发词是「分享 / 发布之前」,而就地编辑的每一刀落地即发布,永远到不了那个时刻,整体回读因此一次也没跑;新写的句子同时带进了编辑自己的行话,文档 owner 的反应是「改成看不懂的了」,返工重写了全部新句。两个缺陷都只在跨刀整体读时显形。证据边界照实说明:prose 收尾规则在本仓没有可执行的行为 oracle,锚点钉住的是规则在场与措辞,不是模型行为差分,不按行为实测记。owner-generalization map 在 `specs/125-doc-closeout-and-register-drift/evidence/owner-map.md`(十个 owner 逐条 updated/unchanged/routed/not-applicable)。RED-baseline(applied,differential):把该锚定句改成非规范措辞,`register-firing-path-resolution.rb` rc=1 并点名 `source-register.md` 的这一行与该 locator;控制组与恢复后均 rc=0(落地时重跑为 527 locators resolved,与同轮提交的 mutation-walk.txt 一致;518 是本行初稿时的捕获值,账本此后被追加过),同一 diff 的其余检查两侧不变。 |
|
|
687
|
+
| 就地编辑一篇已发布文档时不存在「分享前」这一刻——每一刀落地即发布——所以挂在分享前的收尾回读永远不触发;触发点顺延到本轮最后一次写操作之后,收工前必须整体回读一遍 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/closeout-reread.md#没有「发布前」这一刻**:每一刀落地即发布,读者随时可能正在读。触发点顺延到**最后一次写操作之后**:收工前必须把整篇(或受影响那一面)整体回读一遍 | `updated` | Owner key `tighten-doc/SKILL.md`。Observed failure:一篇面向业务读者的大白话协作文档被逐句就地修订约 25 次,每一次替换单独看都正确;收尾回读的既有触发词是「分享 / 发布之前」,而就地编辑的每一刀落地即发布,永远到不了那个时刻,整体回读因此一次也没跑;新写的句子同时带进了编辑自己的行话,文档 owner 的反应是「改成看不懂的了」,返工重写了全部新句。两个缺陷都只在跨刀整体读时显形。证据边界照实说明:prose 收尾规则在本仓没有可执行的行为 oracle,锚点钉住的是规则在场与措辞,不是模型行为差分,不按行为实测记。owner-generalization map 在 `specs/125-doc-closeout-and-register-drift/evidence/owner-map.md`(十个 owner 逐条 updated/unchanged/routed/not-applicable)。细节落在同包的 `tighten-doc/references/closeout-reread.md`(触发点、这一遍要拿出的证据、三类不能顶替它的东西、语域漂移查法),entrypoint 只留触发与硬规则并因此净缩小(bytes 49905→49454,body words 9822→9806)。RED-baseline(applied,differential):把该锚定的规范列表行改写掉,`register-firing-path-resolution.rb` rc=1 并点名该 locator;控制组与恢复后 rc=0。这一行与上一行分别钉住本轮两条规则,任何一半**被锚定的那段字面**被删或被改写都会红,不靠同一个锚代管两件事;但改动规则的**适用条件**(例如给它加一个前置条件)不动锚内任何字,闸检不出——这一条与本表下方那行同口径,不作更强声称。锚点特意跨过承重从句——「没有发布前这一刻 / 每一刀落地即发布 / 顺延到最后一次写操作之后 / 收工前必须」连成一条字面量,删掉其中任一从句而只留末句都会让 locator 失配;这是评审指出的绕过(只锚末句时,删掉前面的适用条件仍能过闸)。覆盖边界照实说明:entrypoint 里那句一行复述不单独钉锚,本行不声称闸能护住它。 |
|
|
688
|
+
| 语域漂移的查法本身是规则的一部分:新句里原文没出现过的术语是候选,每个候选必须换成原文已有的说法或当场用一句白话解释,原样留着不算处理 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/closeout-reread.md#每个候选必须二选一:换成原文已经在用的说法,或当场用一句白话把它解释掉 | `updated` | Owner key `tighten-doc/SKILL.md`。查法与触发点同属本轮那条收尾规则,分行钉锚是**归属选择**不是解析器限制——守卫会把逗号分隔的多个 locator 拆开各自解析(`register-firing-path-resolution.rb` 的 locator 拆分与 multi_bad/multi_ok 两条用例);分行是为了让红的时候直接指到是哪一步被动了。促成拆分的实测是:只钉规则句时,把查法三步删掉仍能过闸。RED-baseline(applied,differential):删掉该规范列表行,`register-firing-path-resolution.rb` rc=1 并点名该 locator;控制组与恢复后 rc=0。覆盖边界:本行只钉「候选术语必须处理」这一步;隔离新句与对照改前文本两步由下两行分别钉住。 |
|
|
689
|
+
| 语域漂移的查法要先隔离本轮新句:必须把新写的句子单独拎出来单独看,混在整篇里读就看不出用词是谁的 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/closeout-reread.md#必须把本轮**新写的句子**单独拎出来单独看 | `updated` | Owner key `tighten-doc/SKILL.md`。与上三行同属本轮那条收尾规则,分行是归属选择——守卫支持一条 firing-path 里放多个 locator 并各自解析,分行只是为了让红时能指到具体那一步。RED-baseline(applied,differential):删掉或改写该规范列表行,`register-firing-path-resolution.rb` rc=1 并点名该 locator;控制组与恢复后 rc=0。这一行来自人工授权轮的评审:三个 locator 都不覆盖「隔离新句」,删掉它仍能过闸。 |
|
|
690
|
+
| 语域漂移的候选判定必须以改动之前的文本为对照:不得拿改完的文档做对照,否则新词已经在里面,永远查不出候选 | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/closeout-reread.md#逐个术语查它**在改动之前的文本里**出现过没有;不得拿改完的文档做对照,否则新词自己就在里面,永远查不出候选。没出现过的就是候选 | `updated` | Owner key `tighten-doc/SKILL.md`。与上三行同属本轮那条收尾规则,分行是归属选择——守卫支持一条 firing-path 里放多个 locator 并各自解析,分行只是为了让红时能指到具体那一步。RED-baseline(applied,differential):删掉或改写该规范列表行,`register-firing-path-resolution.rb` rc=1 并点名该 locator;控制组与恢复后 rc=0。这一行同样来自人工授权轮。锚点从「逐个术语」起,覆盖遍历范围、对照对象、禁令与候选定义四段连写:上一轮 challenge 实测出,只钉禁令时把「在改动之前的文本里」换成「在术语表里」,五条 locator 全部照旧匹配而查法已废;扩锚后该替换直接失配。同类 finding 已连出三轮(只护一半规则 / 锚点截断 / 换掉正面对照对象),据此在账本里把结论写死:substring 锚钉的是字面不是语义,它保证的只是**被锚定的那段字面**被删或被改写时会红;它不保证规则被改写时会红——实测:把查法的引导词改成「以下查法仅在用户明确要求时执行」,五条 locator 全部照常匹配而整条查法已成可选。适用条件与语义完整性由评审与人读负责,本行不作此声称。 |
|
|
691
|
+
| 钉住散文规则的锚点闸必须自带能失败的测试:删掉被锚定的规则行必须让闸变红并点名该 locator,而锚点之外的文字被掏空时闸不得报红——后一条把「锚钉字面不钉语义」这条边界写成被执行的事实,而不是账本里的一句声明 | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`(本轮未改,改动落在同包的该测试脚本)。Observed failure:本轮连续三轮 challenge 都指出同一件事——变异记录依赖的守卫不在按 diff 装的评审包里,包内无法核验;先后用「路径+blob 哈希」和「候选内的风险责任人接受书」作答都被驳回,后者尤其是错的:被评审的东西不能自己给自己发授权。RED-baseline(applied,differential,隔离路径归因):完整套件 fail-fast,任何守卫变异都先撞红最早受影响的既有用例,新增两条根本跑不到——本轮 challenge 正是据此推翻了先前那条「改诊断串」的证据,那次变异撞的是既有的 reworded-anchor 用例,什么也没归因到。改用**每条用例各一份单例副本**:把 `next if body.include?(anchor)` 换成整行相等时,边界用例红、删除用例绿;换成从不报缺失锚点时,删除用例红、边界用例绿;控制组与恢复四格全绿。两次不翻转的格子都是实跑观测;探针已提交为 `specs/125-doc-closeout-and-register-drift/evidence/attribution-probe.sh`,一条命令重跑整张矩阵;它在被测提交的一次性 detached worktree 里执行,调用者的检出只读不写(首版写进活动检出,被本轮评审判为 P1 并已重写)。三次被推翻的归因尝试(改诊断串撞到既有用例/两条用例同副本致后一条不执行/副本临时不可复核)连同原因一并留在走查里。两次变异与原始输出见 `specs/125-doc-closeout-and-register-drift/evidence/mutation-walk.txt` 的第二段。 |
|
|
692
|
+
| 非 wording 的提炼改动只欠一次独立 review 加一次对抗 challenge,各为单次调用、不挂评审链;最后一次评审之后凡改了证据与台账以外内容的提交,都由 agent 自动跑一次只看该增量的复查,最多两次,仍有 P0/P1 就撤回该改动或把 PR 标为阻塞,不交人逐行把关;不再要求落地树等于某次评审的候选哈希,CI 只检查改了 skills/ 或 hooks/ 的 PR 至少带一份结论性的 review 结果,challenge 仍由流程规则约束 | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#Every post-review delta gets a delta pass | `updated` | Owner key `skill-extraction-workflow/SKILL.md`;配套改动 `references/dual-track-review-gate.md`、`references/extraction-quickstart.md`、`references/validation-and-landing.md`、`scripts/extraction_review_gate.sh`;删除 `scripts/review_ledger_binding.py`、`scripts/validate_extraction_review_state.py` 及两套测试,CI 的合并侧绑定步骤换成 `scripts/check_review_evidence_present.py` 存在性检查(只查有没有,不比对候选)。Observed failure:候选哈希绑定让技能仓里几乎每次修复都作废已有评审结果并重开序列,此前连续几轮的证据目录各自只留下最后两轮收据,其间又为它补了分区清单、首父链逐轮重跑、候选与评审包拆分、成本收窄等轮次;引入绑定的提交补的是自身节奏里的一致性缺口,不是有未审改动出过事。RED-baseline(applied,differential):新测试放进 base 的一次性 detached worktree 对旧 wrapper 跑,第一条断言即红(review 调用带着 `--challenge-budget 1`),当前 wrapper 全绿。 |
|
|
693
|
+
| 被正式废止的台账锚点可以由一条豁免绑定一组固定的历史行:每行按 SHA-256 摘要逐一核对,删掉、改写或多出一行引用都必须报红;豁免的 command 锚点所指脚本不存在时视为已记录的退休 | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`;改动在 `scripts/register-firing-path-resolution.rb` 与其接线测试。本行也是退休锚点的 supersede 行:`test_review_ledger_binding.sh`(19 行引用)、`test_validate_extraction_review_state.sh`(5 行)、`review_ledger_binding.py`(1 行),以及 `dual-track-review-gate.md` 与 `extraction-quickstart.md` 各两句被本轮删除或改写的锚点,逐条原因见该脚本的 EXEMPT 表。Observed failure:删除脚本后整本台账校验 rc=1、30 个 locator 不可解析,而原豁免每个锚点只容许 1 行,表达不了一个脚本被 19 行引用的退休。RED-baseline(applied):加豁免前 rc=1 并逐条点名,加后 rc=0;接线测试新增删一行、改一行、多一行三条回归,各自以准确诊断报红。 |
|
|
694
|
+
| code-review 的分阶段契约不得再许诺一个在合并时重算候选的脚本:该脚本已退休,契约只说明收据同时记录评审包与候选两个哈希,需要比对落地树的调用方自行比对 | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md`(本轮未改);改动在本包分阶段评审契约的一句、控制器两处注释、控制器测试新增一条断言。通用控制器的评审链模式保留给其他调用方,未改。RED-baseline(applied,differential):新断言对 base 的契约文字为红(含 1 处已退休脚本名),对当前文字为绿。 |
|
|
695
|
+
| 调研类请求(含深度调研、deep research、调研某个产品)必须先进 multi-perspective-research:宿主另装的通用 deep-research 技能只能作为本技能循环里的检索工具,不替代它;常驻路由层同步加这一条 | `multi-perspective-research` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/multi-perspective-research/SKILL.md#description; bank-evidence: file:eval/routing-tasks.jsonl#route-product-deep-research | `updated` | Owner key `multi-perspective-research/SKILL.md`;包内只改 frontmatter description(触发词前置,Skip 条款原样后移),另改 `agent-context/session-start.md` 常驻路由加一行(17051→17042 字节,删三处冗余措辞抵消)与 `docs/SKILLS.md` 目录标记由 leaf 改为 entry。Observed failure:codex 宿主上一条「深度调研这个产品」的请求选中了宿主自带的通用 deep-research 技能(其描述为 Use only when the user asks for deep research),CCL 调研 owner 直到用户追问才加载;当时常驻路由层没有调研条目,本技能描述开头是近 200 字的 Skip 条款。RED-baseline:失败侧是记录在维护者私有宿主会话里的实际事件;改后侧是题库新增用例与两条相邻用例各跑 3 次、9/9 判定正确(筛查精度,未达 10 次观测下限);codex 端同句复测要等宿主插件更新后才能跑,记为待办。 |
|
|
696
|
+
| 评审之后候选的任何改动(任意严重度的修复、补的测试、changelog)都要在开 / 就绪 / 合并 PR 或报完成之前由 agent 按所属闸自己重审:通用流程完整重跑、最多 5 次,仍有 P0/P1 就撤回或报阻塞,不交人工 review,最后一次结论性评审覆盖 HEAD 时才能说 HEAD 已评审;控制器在 worktree 的 git 目录里记下每次结论性评审时的 HEAD,插件的 PR hook 在开 / 就绪 / 合并命令前比对并把漏审提交注入给 agent | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/code-review/references/development-completion.md#Pushing it for a human to review does not discharge it | `updated` | Owner key `code-review/SKILL.md`(入口未改);改动在 `skills/code-review/references/development-completion.md`(新增「Before a pull request or a ready report」五步必走检查)、`skills/code-review/references/staged-review-contract.md`、`skills/code-review/scripts/review_gate.py`(本机回执:只在候选由 `--base` 从整个工作区派生、冻结前后 HEAD 与干净度一致、结果结论性时写,锚点记下解析后的 git 目录路径与身份,写入时从根逐级不跟随链接打开它、核对身份,再经 no-follow 目录描述符写入)、`hooks/remind-review-covers-head.sh`(新 PreToolUse 提醒,只提醒不拦截,OpenCode 插件同步镜像)、常驻层 `agent-context/session-start.md` 一句(净 -4 字节)。Observed failure:另一仓库的生产会话在最后一次评审之后提交了为一条 P2 补的测试和 changelog,随即推分支开 MR,报告写「没有再单独 review、等人工 review」,合并后又把未评审的 HEAD 说成评审过的。当时规则已写「候选变了要重审」(本 reference 与 product-rd 验证门都有),失效在触发点:从修复到开 MR 的动作序列里没有一步比对评审覆盖的提交与 HEAD。RED-baseline:红的一半只有这次生产失效;两种隔离探针(直接问下一步、流程中只给命令,均 n=3/臂,判分标准先于运行冻结)在改前文本上都没复现(问答 3/3 先重审再开 MR;流程中 0/3 推送或开 MR),所以文本改动本身的效果未被证明,归为假设,摘要见 `specs/128-renewed-review-and-repo-contract/evidence/probe-summary.md`。hook 的机械部分有 applied 差分:在副本上分别去掉内容树相等判断、祖先判断、`--ready` 匹配、引号屏蔽,各自只让对应的那一条用例变红;评审后补的「同一命令里先 commit 再开 PR」「`--help ;` 后接真正的开 PR」「带引号带空格的 `cd`」「先开草稿、再提交、再就绪」四类用例在修复前的 hook 上为红(现共 25 条);控制器的去重、冻结前后锚点一致两条用例,以及「原 git 目录被挪走、旧路径换成指向它的链接」场景,在修复前的控制器上为红;回执目录被换成链接的竞态按构造封住、没有确定性测试。提醒是否改变 agent 行为未测量。 |
|
|
697
|
+
| 评审模式把被评审仓库自己入库的约定文件引在候选之后交给评审员:从仓库根到每个改动路径的每层目录取 `AGENTS.override.md`(否则 `AGENTS.md`)、`CLAUDE.md`、`.claude/CLAUDE.md`,根在前,32 KiB 以内;未入库和 `CLAUDE.local.md` 一律不读(不得外发给别家评审模型),链接、非 UTF-8、超预算的整份省略并记原因,永不因约定文件让评审失败;有引用时加 `repository_contract` concern,要求评审员报改动违反的规则和规则本身的缺陷(`Contract defect:`),规则只作数据、不能为缺陷开脱 | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md`(入口未改);改动在 `skills/code-review/scripts/review_gate.py`(`repository_contract_section`、profile 与结果的 `repository_contract` 字段、条件 concern、trust boundary 一句)、`skills/code-review/references/staged-review-contract.md`、`skills/code-review/references/development-completion.md`(偏离约定要在 plan intent 里声明、`Contract defect:` 的处置、不许悄悄放松约定文件),`skills/code-review/scripts/test_review_client_compat.py` 的 provider mock 改为放行控制器自己的 git 读取;共享读取函数 `read_bounded_regular_file` 逐级打开目录中途失败时不再泄漏已持有的目录描述符(修复前 100 次失败读取后打开的描述符由 4 个涨到 254 个,约定文件被逐个省略时会反复触发)。发现规则按两家宿主一手文档核对(Codex:每层目录 override 优先、根到深拼接、默认 32 KiB;Claude Code:`CLAUDE.md` 或 `.claude/CLAUDE.md`,`CLAUDE.local.md` 是个人文件)。`candidate_sha256` 在追加前算定,引用内容不能增加候选路径;challenge、complete、wording-only 不附加。不读 `@path` 导入、`.claude/rules`、宿主配置的备用文件名(结果里 `sources_not_read` 列明)。RED-baseline(applied,differential):在副本上分别去掉「只收已入库文件」、去掉 override 优先、不加 concern,三条约定用例里恰好两条变红、其余用例全绿;base 版控制器配新套件时三条约定用例全红;中间目录被换成链接、带第二个硬链接的约定文件都被省略且内容不进评审包。 |
|
|
698
|
+
| 产品研发验证门里「评审后有实质改动才全量重跑」的「实质」一词删除:评审后的任何改动(含测试 / 文档)都触发全量重跑,作者不能自行判定改动无关紧要 | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#any post-review change, tests/docs included | `updated` | Owner key `product-rd-workflow/SKILL.md`(改一句,字数与字节都不增)。Observed failure 同上一条生产会话:它把评审后的提交归为「只有测试和文档」而没有重审,原句的「实质」正好给了这个归类余地。RED-baseline 同上:红的一半是生产失效,隔离探针在改前文本上未复现,改动效果未被证明。 |
|
|
699
|
+
| 提炼流程的增量复查上限由两次改为五次(用户裁决),同步 `skills/skill-extraction-workflow/SKILL.md`、`dual-track-review-gate.md`、`extraction-quickstart.md` 与被测试钉住的原文;本行按指针取代 127 轮评审线那一行里的「最多两次」 | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#After five delta passes a still-open P0/P1 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`(同一行内省 6 字节,入口不增长)。delta 复查改为 `--base <已评审提交>` 取增量,使它绑定工作区并留下 PR hook 读取的本机回执。RED-baseline(applied):`test_extraction_review_gate.sh` 钉住的短语由两次改为五次后,对 base 文本变红、对 head 文本变绿。 |
|
|
@@ -141,7 +141,7 @@ Before commit, for any rule that appears in more than one authored file (`SKILL.
|
|
|
141
141
|
|
|
142
142
|
## Bounded Independent Review Packet
|
|
143
143
|
|
|
144
|
-
Use this when an independent review is required but a broad reviewer prompt hangs, returns no output, or starts expanding beyond the intended review scope. For non-wording extraction work, the gate-valid path is `scripts/extraction_review_gate.sh`, which
|
|
144
|
+
Use this when an independent review is required but a broad reviewer prompt hangs, returns no output, or starts expanding beyond the intended review scope. For non-wording extraction work, the gate-valid path is `scripts/extraction_review_gate.sh`, which makes each pass single-shot while delegating transport to the provider-neutral `code-review` controller. A strictly proven wording-only change uses the proof-bound generic single-review recipe in `code-review/references/staged-review-contract.md`; its controller-derived wording scope and independent `wording_only_boundary` result replace neither one another nor a failed semantic check. Both paths preserve the frozen packet, family exclusion, structured validation and client-specific recovery; raw provider CLI packets are debugging/advisory only and must not be recorded as passing review evidence.
|
|
145
145
|
|
|
146
146
|
Required flow:
|
|
147
147
|
|
|
@@ -164,7 +164,7 @@ Review only stdin. Do not use tools. Do not request more context. Return finding
|
|
|
164
164
|
5. Apply or explicitly reject actionable findings.
|
|
165
165
|
6. Rerun the owning skill validators and `git diff --check`.
|
|
166
166
|
7. Record the result as `findings applied`, `no blocking findings`, or `review unavailable after remediation` only when the wrapper or approved alternate produced valid structured evidence. Raw packet output is recorded separately as debugging/advisory and cannot close the gate.
|
|
167
|
-
8.
|
|
167
|
+
8. Record each pass and the post-review delta per `references/dual-track-review-gate.md` (Recording the passes). A wording-only review records its single independent-review row.
|
|
168
168
|
|
|
169
169
|
Do not count as completed review:
|
|
170
170
|
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Refuse a pull request that changes shared skill behavior with no recorded review.
|
|
3
|
+
|
|
4
|
+
The extraction review lane owes one review and one challenge per non-wording
|
|
5
|
+
change (references/dual-track-review-gate.md). The merge-side candidate binding
|
|
6
|
+
that used to sit here was retired: it tied every pass to one tree hash, so every
|
|
7
|
+
fix voided the passes and restarted the sequence. What it also did — refuse a
|
|
8
|
+
landing with no review at all — is the half worth keeping, and it needs no hash.
|
|
9
|
+
|
|
10
|
+
This gate asks one question: does a pull request that changes `skills/` or
|
|
11
|
+
`hooks/` carry at least one conclusive review result added or modified under a
|
|
12
|
+
specs/<round>/evidence/ directory?
|
|
13
|
+
|
|
14
|
+
It deliberately does not require a challenge result. An earlier version did,
|
|
15
|
+
with a waiver for wording-only changes; independent review broke that waiver in
|
|
16
|
+
four successive forms (a global waiver, a second skill, a later edit to the same
|
|
17
|
+
file, a rewritten frontmatter delimiter), because deciding "still wording-only"
|
|
18
|
+
means re-parsing diffs this gate does not own. Same-class recurrence is the cue
|
|
19
|
+
to delete the capability, so the challenge obligation stays in the review-lane
|
|
20
|
+
rule and this gate only catches a round that recorded no external pass at all.
|
|
21
|
+
|
|
22
|
+
A result is conclusive when it is a schema-3 controller envelope whose status is
|
|
23
|
+
`passed` or `findings` and which names the client that ran it. It does not check
|
|
24
|
+
which candidate a result reviewed, and it cannot tell a genuine result from a
|
|
25
|
+
hand-written one: like the repository's other author-declared gates it catches a
|
|
26
|
+
pass that was never recorded, not a forged one.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import argparse
|
|
32
|
+
import json
|
|
33
|
+
import subprocess
|
|
34
|
+
import sys
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
|
|
37
|
+
SUBJECT_PREFIXES = ("skills/", "hooks/")
|
|
38
|
+
CONCLUSIVE = ("passed", "findings")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def changed_paths(root: Path, base: str) -> list[str]:
|
|
42
|
+
result = subprocess.run(
|
|
43
|
+
# No --diff-filter: every change class counts, including a type change
|
|
44
|
+
# (a regular file replaced by a symlink), which a filter list omits.
|
|
45
|
+
["git", "-C", str(root), "diff", "--name-only", "--no-renames", base, "HEAD"],
|
|
46
|
+
capture_output=True, text=True,
|
|
47
|
+
)
|
|
48
|
+
if result.returncode != 0:
|
|
49
|
+
raise RuntimeError(result.stderr.strip() or "git diff failed")
|
|
50
|
+
return [line for line in result.stdout.splitlines() if line]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def is_evidence_json(path: str) -> bool:
|
|
54
|
+
parts = path.split("/")
|
|
55
|
+
return (
|
|
56
|
+
len(parts) >= 4
|
|
57
|
+
and parts[0] == "specs"
|
|
58
|
+
and "evidence" in parts[2:-1]
|
|
59
|
+
and parts[-1].endswith(".json")
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def load_result(root: Path, path: str) -> dict | None:
|
|
64
|
+
target = root / path
|
|
65
|
+
if not target.is_file():
|
|
66
|
+
return None
|
|
67
|
+
try:
|
|
68
|
+
value = json.loads(target.read_text(encoding="utf-8"))
|
|
69
|
+
except (OSError, UnicodeDecodeError, json.JSONDecodeError):
|
|
70
|
+
return None
|
|
71
|
+
if not isinstance(value, dict) or value.get("schema_version") != 3:
|
|
72
|
+
return None
|
|
73
|
+
if value.get("mode") not in ("review", "challenge"):
|
|
74
|
+
return None
|
|
75
|
+
if value.get("status") not in CONCLUSIVE or not value.get("selected_client"):
|
|
76
|
+
return None
|
|
77
|
+
return value
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def main(argv: list[str] | None = None) -> int:
|
|
81
|
+
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
82
|
+
parser.add_argument("--repo-root", default=".")
|
|
83
|
+
parser.add_argument("--base", required=True)
|
|
84
|
+
args = parser.parse_args(argv)
|
|
85
|
+
root = Path(args.repo_root).resolve()
|
|
86
|
+
|
|
87
|
+
try:
|
|
88
|
+
paths = changed_paths(root, args.base)
|
|
89
|
+
except RuntimeError as exc:
|
|
90
|
+
print(f"review_evidence_unevaluated: {exc}")
|
|
91
|
+
return 2
|
|
92
|
+
|
|
93
|
+
subjects = [p for p in paths if p.startswith(SUBJECT_PREFIXES)]
|
|
94
|
+
if not subjects:
|
|
95
|
+
print("review_evidence_not_required: no change under skills/ or hooks/")
|
|
96
|
+
return 0
|
|
97
|
+
|
|
98
|
+
reviews: list[str] = []
|
|
99
|
+
challenges: list[str] = []
|
|
100
|
+
for path in paths:
|
|
101
|
+
if not is_evidence_json(path):
|
|
102
|
+
continue
|
|
103
|
+
result = load_result(root, path)
|
|
104
|
+
if result is None:
|
|
105
|
+
continue
|
|
106
|
+
(reviews if result["mode"] == "review" else challenges).append(path)
|
|
107
|
+
|
|
108
|
+
if not reviews:
|
|
109
|
+
print(
|
|
110
|
+
"review_evidence_missing: this pull request changes "
|
|
111
|
+
f"{len(subjects)} path(s) under skills/ or hooks/ but carries no conclusive "
|
|
112
|
+
"review result under specs/<round>/evidence/"
|
|
113
|
+
)
|
|
114
|
+
print(" fix: commit the controller result JSON of each owed pass "
|
|
115
|
+
"(references/dual-track-review-gate.md, Recording the passes)")
|
|
116
|
+
return 1
|
|
117
|
+
print(f"review_evidence_present_ok: {len(reviews)} review, {len(challenges)} challenge")
|
|
118
|
+
return 0
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
if __name__ == "__main__":
|
|
122
|
+
sys.exit(main())
|
|
@@ -13,4 +13,7 @@ burn-rate-page-row-reference skills/platform-observability/references/sli-slo-de
|
|
|
13
13
|
coverage-tier-provenance skills/testing-strategy/references/test-code-authoring-patterns.md 60% acceptable / 75% commendable / 90% exemplary 分档 071-chainC-r1f4: externally verified coverage tiers (specs/071 source-verification)
|
|
14
14
|
pairwise-trigger-range skills/test-artifact-management/references/classical-test-design-techniques.md 2-way 累计触发 53–97% 071-chainC-r1f4: NIST SP 800-142 empirical range, externally verified (specs/071 source-verification)
|
|
15
15
|
bva-two-vs-three-value skills/test-artifact-management/references/classical-test-design-techniques.md 2-value(边界 + 下一格)和 3-value(边界 + 两侧) 071-chainC-r1f4: ISTQB v4 BVA variant definitions, externally verified (specs/071 source-verification)
|
|
16
|
-
|
|
16
|
+
lane-budget-default-range skills/code-review/references/timeout-auth-and-capabilities.md defaults to 2400 seconds and accepts 5 to 3600 125-r1f1: the entrypoint now points here for the cumulative lane budget instead of restating it; dropping or drifting the default/range would silently strip the bound from both surfaces
|
|
17
|
+
lane-budget-mode-minimums skills/code-review/references/timeout-auth-and-capabilities.md 21 total seconds for review and 16 125-r1f1: the per-mode fail-closed minimums the entrypoint delegates here; without the pin the fail-closed limit can vanish with no suite failing
|
|
18
|
+
lane-budget-reserved-seconds skills/code-review/references/timeout-auth-and-capabilities.md while reserving ten controller 125-r1f1: the reserved controller seconds in the per-invocation division the entrypoint delegates here
|
|
19
|
+
recurrence-rule-owner-pointer skills/code-review/references/staged-review-contract.md `../../skill-extraction-workflow/SKILL.md` owns that rule 125-binding-r2f1: the chain-bound reader cannot load the owning skill, so this pointer is the only route to the rule that discharges the trigger; a wrong depth resolves inside code-review and no link checker sees a backticked path
|
|
@@ -1,22 +1,51 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
|
-
# Extraction-owned
|
|
2
|
+
# Extraction-owned review wrapper: every call is one single-shot pass.
|
|
3
|
+
#
|
|
4
|
+
# A non-wording extraction owes one review and one challenge, plus a delta pass
|
|
5
|
+
# per fixed P0/P1 (references/dual-track-review-gate.md). None of them is bound
|
|
6
|
+
# to another pass or to one candidate hash, so this wrapper fixes the controller
|
|
7
|
+
# options that make a pass single-shot and refuses the options that would open a
|
|
8
|
+
# tracked review chain. The generic controller keeps its chain mode for other
|
|
9
|
+
# callers; this lane does not use it.
|
|
3
10
|
set -euo pipefail
|
|
4
11
|
|
|
5
12
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
|
6
13
|
CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
|
|
7
14
|
|
|
15
|
+
fail() {
|
|
16
|
+
echo "extraction_review_gate_error: $*" >&2
|
|
17
|
+
exit 2
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
mode=""
|
|
21
|
+
expect_mode=0
|
|
8
22
|
for arg in "$@"; do
|
|
23
|
+
if [[ "$expect_mode" == 1 ]]; then
|
|
24
|
+
mode="$arg"
|
|
25
|
+
expect_mode=0
|
|
26
|
+
continue
|
|
27
|
+
fi
|
|
9
28
|
case "$arg" in
|
|
10
|
-
--
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
29
|
+
--mode) expect_mode=1 ;;
|
|
30
|
+
--mode=*) mode="${arg#--mode=}" ;;
|
|
31
|
+
# Prefixes cover the controller's unambiguous abbreviations as well as the
|
|
32
|
+
# full spellings, so a shortened flag cannot reopen a chain.
|
|
33
|
+
--challenge-b*|--challenge-i*)
|
|
34
|
+
fail "the extraction lane fixes the challenge budget and index; do not pass $arg" ;;
|
|
35
|
+
--review-c*|--au*|--prio*|--pre*|--com*)
|
|
36
|
+
fail "the extraction lane is single-shot; review-chain option $arg is not accepted" ;;
|
|
14
37
|
esac
|
|
15
38
|
done
|
|
16
39
|
|
|
40
|
+
case "$mode" in
|
|
41
|
+
review) fixed=(--challenge-budget 0) ;;
|
|
42
|
+
challenge) fixed=(--challenge-budget 1 --challenge-index 1) ;;
|
|
43
|
+
"") fail "pass --mode review or --mode challenge" ;;
|
|
44
|
+
*) fail "the extraction lane runs only --mode review or --mode challenge, not $mode" ;;
|
|
45
|
+
esac
|
|
46
|
+
|
|
17
47
|
if [[ ! -x "$CONTROLLER" ]]; then
|
|
18
|
-
|
|
19
|
-
exit 2
|
|
48
|
+
fail "code-review controller is unavailable"
|
|
20
49
|
fi
|
|
21
50
|
|
|
22
|
-
exec bash "$CONTROLLER"
|
|
51
|
+
exec bash "$CONTROLLER" "${fixed[@]}" "$@"
|
|
@@ -83,16 +83,31 @@ unless File.file?(register_path)
|
|
|
83
83
|
exit 1
|
|
84
84
|
end
|
|
85
85
|
|
|
86
|
-
# Reviewed waivers cover
|
|
86
|
+
# Reviewed waivers cover the immutable historical rows whose locator was retired
|
|
87
87
|
# by an explicit superseding round. The digest table below binds each waiver to
|
|
88
|
-
#
|
|
88
|
+
# those exact rows (one digest, or an array when several rows cited the retired
|
|
89
|
+
# locator); a new row cannot inherit it by reusing the locator.
|
|
89
90
|
EXEMPT = {
|
|
90
91
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#Disabled semantics are real, not painted" =>
|
|
91
92
|
"065 replaced the combined platform walkthrough with an authority-classed claim ledger and executable delivery contract",
|
|
92
93
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#predictive-back geometry routes to" =>
|
|
93
94
|
"065 moved platform mechanics to the canonical client-owner return while keeping platform guidance scoped",
|
|
94
95
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#Interaction-state matrix is complete" =>
|
|
95
|
-
"065 replaced walkthrough-level proof with criterion IDs, test-layer selection, runtime evidence, and a candidate-bound verdict"
|
|
96
|
+
"065 replaced walkthrough-level proof with criterion IDs, test-layer selection, runtime evidence, and a candidate-bound verdict",
|
|
97
|
+
"command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh" =>
|
|
98
|
+
"127 retired the closeout ledger and its validator; the extraction lane records single-shot passes instead",
|
|
99
|
+
"command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh" =>
|
|
100
|
+
"127 retired the merge-side candidate binding; the post-review delta and the human merge replace it",
|
|
101
|
+
"command:skills/skill-extraction-workflow/scripts/review_ledger_binding.py" =>
|
|
102
|
+
"127 retired the merge-side candidate binding; the post-review delta and the human merge replace it",
|
|
103
|
+
"file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#Sum spent rounds across all chains before opening one more" =>
|
|
104
|
+
"127 replaced the round budget with one review, one challenge and bounded delta passes",
|
|
105
|
+
"file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen" =>
|
|
106
|
+
"127 lets the challenge run after the review's fixes; nothing is bound to one candidate",
|
|
107
|
+
"file:skills/skill-extraction-workflow/references/extraction-quickstart.md#still owes the two-round chain before it can land" =>
|
|
108
|
+
"127 retired the merge-side binding that imposed the chain on wording-only changes",
|
|
109
|
+
"file:skills/skill-extraction-workflow/references/extraction-quickstart.md#must exclude every evidence JSON the round has already added" =>
|
|
110
|
+
"127 retired the merge-side binding whose evidence exclusion this rule mirrored"
|
|
96
111
|
}.freeze
|
|
97
112
|
|
|
98
113
|
# `\p{Word}` rather than `[a-z0-9]`: GitHub keeps non-ASCII characters in a slug,
|
|
@@ -471,7 +486,45 @@ EXEMPT_ROW_DIGESTS = {
|
|
|
471
486
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#predictive-back geometry routes to" =>
|
|
472
487
|
"6847e68f062c2fe65a32f34bc743a6162a4e245a3a5d858c020faa841868f929",
|
|
473
488
|
"file:skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md#Interaction-state matrix is complete" =>
|
|
474
|
-
"9671789a4bac36787dc26965ca03c98c7ab1c10de3312c6792587cb34e7c548e"
|
|
489
|
+
"9671789a4bac36787dc26965ca03c98c7ab1c10de3312c6792587cb34e7c548e",
|
|
490
|
+
"command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh" => [
|
|
491
|
+
"da20ca2ada11f70a55bbfbffde84a80d0a8a8a46c9e4010a96a9614ca854fd34",
|
|
492
|
+
"b8eede70a62910f0a7f2d11cfd854537c3450889bc294cfc933b37f2a13c55b8",
|
|
493
|
+
"1c575fdad17f3fca00569832d487dcf0ac0f28fe5f8fb4bfa21e784d9111a243",
|
|
494
|
+
"b40a21a554cbc8bb3484c345160403a69f1cfb0b41702559fc2442572f52afce",
|
|
495
|
+
"0579ef1411fc866c0e612f9475634f073cfc0e0be429482ee9e8b4fd7abafe10"
|
|
496
|
+
],
|
|
497
|
+
"command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh" => [
|
|
498
|
+
"0579ef1411fc866c0e612f9475634f073cfc0e0be429482ee9e8b4fd7abafe10",
|
|
499
|
+
"fdb08300002289dc5b3d0bcd279589ec8eca7dd5d3b14f0b7b079d34372b41f7",
|
|
500
|
+
"4a593029ba36b128519adeb8a3b9cbee32053caec7953c884ad587efaf88a3aa",
|
|
501
|
+
"f56d544525a45d0b26e8752eec4f2ff516239c3ca866df600506ea3ccb39f4b2",
|
|
502
|
+
"6f3b518a04ea8c20adc1ec3cf377a7eda2ef775903354cf54bed59268a8332b4",
|
|
503
|
+
"dfa47eac33add5ce0a7a73ce237b8bf4a1589bd325eb34b4531e25f9adbb574d",
|
|
504
|
+
"c6e6f74ca2407db2d98124e0517e6d70fae6f9df669f190bb7336ab76f13eb56",
|
|
505
|
+
"16f862e9df7fea0ea817e392e78e561690b0120f97125c2db5f7c7b987154385",
|
|
506
|
+
"c450336b8031f4ea9b3a614db3908b1f21c37559098b4e24672947172bb42de4",
|
|
507
|
+
"aaa569c4934575aae78c2a3901ef44696501ee56d0cb700bfa2559ac79677d6c",
|
|
508
|
+
"a9a2484c102a9392ffa00edc684b8feddea9ccc6a881d35433c72594e72959c5",
|
|
509
|
+
"f9cfc29b561d55da7838d371f1307bf015822bdea2961cfb6dd0b6da6232eede",
|
|
510
|
+
"1e769efadec0325d53de5d0c1023eea675499d9d67b086bcfc4c507ca0f25dac",
|
|
511
|
+
"749ef2d63c4dda83e3885b7c2dfaedc9f2970dc09ae310082ad5367bc322cebc",
|
|
512
|
+
"2402bf5b6b7a00aba5e6200c47277bdf6bb65e4d255f8126b8615eb9d1168545",
|
|
513
|
+
"7f68e4204587602d53d135152d9a0b4fa7e8da1cc0c43f02813debb8c3a99528",
|
|
514
|
+
"0d7e1a6de3ced156e54b9a50737835f49c424846e86adf28f1b19a1aa46f0f5a",
|
|
515
|
+
"2c894914dd47b1764c63bd525891246c8adcb14b0f00c446e84e28320451a213",
|
|
516
|
+
"0ab74921c2222979701499bc12d53c5a0942f51a7b1dd3a090ef58e83cea42f9"
|
|
517
|
+
],
|
|
518
|
+
"command:skills/skill-extraction-workflow/scripts/review_ledger_binding.py" =>
|
|
519
|
+
"6216104a650a7c30f78d0163d0a7cde37c9d809eb5967f291a30ebd516b1e337",
|
|
520
|
+
"file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#Sum spent rounds across all chains before opening one more" =>
|
|
521
|
+
"988cd8128ee6f93b4ea5a0cc8ada170401a1790831b2cd7b95a79642a76f1112",
|
|
522
|
+
"file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen" =>
|
|
523
|
+
"d999fadd8e3778bfdb9d54d34d3cabf965547ed39448a6ffba5e8310ca151612",
|
|
524
|
+
"file:skills/skill-extraction-workflow/references/extraction-quickstart.md#still owes the two-round chain before it can land" =>
|
|
525
|
+
"bf1fa382021b079b200b6d8d367e407bd0adbbac1d862e22d78062d83906c05e",
|
|
526
|
+
"file:skills/skill-extraction-workflow/references/extraction-quickstart.md#must exclude every evidence JSON the round has already added" =>
|
|
527
|
+
"d762cbab094279d26461843dfe12c5dd28ae41a1e9a40b7ae9f7f806fee51c36"
|
|
475
528
|
}.freeze
|
|
476
529
|
# TRUST BOUNDARY. The count answers "one row"; it cannot answer "WHICH row", so
|
|
477
530
|
# it is paired with the digest of the citing row in EXEMPT_ROW_DIGESTS above.
|
|
@@ -535,6 +588,11 @@ File.foreach(register_path).with_index(1) do |line, lineno|
|
|
|
535
588
|
target = File.join(root, rel)
|
|
536
589
|
if !syntactically_contained?(rel)
|
|
537
590
|
unresolved << [lineno, locator, "path escapes the repository"]
|
|
591
|
+
elsif !File.file?(target) && anchor_waived
|
|
592
|
+
# A waived command locator names an executable a superseding round
|
|
593
|
+
# deliberately retired; its absence is the recorded retirement, and the
|
|
594
|
+
# digest check below still pins which historical rows may cite it.
|
|
595
|
+
next
|
|
538
596
|
elsif !File.file?(target)
|
|
539
597
|
unresolved << [lineno, locator, "executable not found"]
|
|
540
598
|
elsif !resolves_inside?(root, rel)
|
|
@@ -701,20 +759,29 @@ end
|
|
|
701
759
|
|
|
702
760
|
exempt_uses.each do |locator, linenos|
|
|
703
761
|
rows = linenos.uniq.sort
|
|
704
|
-
|
|
705
|
-
|
|
762
|
+
# A waiver binds either one row (a digest string) or a fixed set of rows (an
|
|
763
|
+
# array of digests): a retired script or rule that several historical rows
|
|
764
|
+
# cited is still one retirement, and each of those rows is named by its digest.
|
|
765
|
+
expected = exempt_row_digests[locator]
|
|
766
|
+
expected_digests = expected.is_a?(Array) ? expected : [expected].compact
|
|
767
|
+
# How many rows a waiver covers is a property of the waiver, so it is read from
|
|
768
|
+
# the built-in table even when a test injects its own identity table.
|
|
769
|
+
builtin = EXEMPT_ROW_DIGESTS[locator]
|
|
770
|
+
builtin_rows = builtin.is_a?(Array) ? builtin.length : (builtin ? 1 : 0)
|
|
771
|
+
allowance = [expected_digests.length, builtin_rows, EXEMPT_USE_ALLOWANCE].max
|
|
772
|
+
if rows.length > allowance
|
|
773
|
+
rows.drop(allowance).each do |lineno|
|
|
706
774
|
unresolved << [lineno, locator,
|
|
707
|
-
"EXEMPT locator cited by #{rows.length} rows (allowance #{
|
|
708
|
-
"a waiver covers the
|
|
775
|
+
"EXEMPT locator cited by #{rows.length} rows (allowance #{allowance}); " \
|
|
776
|
+
"a waiver covers the recorded historical rows starting at line #{rows.first}, " \
|
|
709
777
|
"not a new row quoting the same retired locator"]
|
|
710
778
|
end
|
|
711
779
|
next
|
|
712
780
|
end
|
|
713
|
-
# The count says
|
|
714
|
-
# historical row and writing a different claim that cites the same
|
|
715
|
-
# the count
|
|
716
|
-
|
|
717
|
-
unless expected
|
|
781
|
+
# The count says how many rows; the digests say WHICH rows. Without them,
|
|
782
|
+
# deleting a historical row and writing a different claim that cites the same
|
|
783
|
+
# locator keeps the count and silently inherits the waiver.
|
|
784
|
+
if expected_digests.empty?
|
|
718
785
|
# A waiver with no recorded row identity keeps only the use-count layer,
|
|
719
786
|
# which cannot tell a rewritten or repurposed row from the one that was
|
|
720
787
|
# waived. On the BUILT-IN table that is a silent downgrade, so a new EXEMPT
|
|
@@ -729,13 +796,28 @@ exempt_uses.each do |locator, linenos|
|
|
|
729
796
|
end
|
|
730
797
|
next
|
|
731
798
|
end
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
799
|
+
remaining = expected_digests.tally
|
|
800
|
+
mismatched = 0
|
|
801
|
+
rows.each do |lineno|
|
|
802
|
+
actual = Digest::SHA256.hexdigest(register_lines[lineno - 1].to_s.rstrip)
|
|
803
|
+
if remaining[actual].to_i.positive?
|
|
804
|
+
remaining[actual] -= 1
|
|
805
|
+
next
|
|
806
|
+
end
|
|
807
|
+
mismatched += 1
|
|
808
|
+
unresolved << [lineno, locator,
|
|
809
|
+
"EXEMPT citing row does not match the waived row (digest #{actual[0, 12]} not recorded); " \
|
|
810
|
+
"a waiver covers specific unrepairable historical rows, so a rewritten or replaced row " \
|
|
811
|
+
"does not inherit it — restore the row, or land a new waiver entry with its own digest and reason"]
|
|
812
|
+
end
|
|
813
|
+
# A rewritten row is already reported above; only rows that are gone entirely
|
|
814
|
+
# are left to name here.
|
|
815
|
+
missing = remaining.values.sum - mismatched
|
|
816
|
+
next unless missing.positive?
|
|
817
|
+
unresolved << [rows.first, locator,
|
|
818
|
+
"EXEMPT entry has no citing row in the ledger for #{missing} of its #{expected_digests.length} " \
|
|
819
|
+
"recorded rows; a waived historical row was deleted or its locator was edited — restore the row, " \
|
|
820
|
+
"or retire its digest in the same change"]
|
|
739
821
|
end
|
|
740
822
|
|
|
741
823
|
# Both groups print before exiting. Bailing out on `malformed` alone would hide
|