@ccoalm/ccl-skills 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +8 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +29 -31
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +32 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
- package/dist/assets/release.json +79 -24
- package/package.json +1 -1
|
@@ -20,7 +20,7 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
|
|
|
20
20
|
├─ d. Sanitization pass with checklist (cheap, seconds)
|
|
21
21
|
├─ e. Owner review gate per mandatory table (deep, minutes)
|
|
22
22
|
│ ├─ Strict wording-only → one independent code-review pass
|
|
23
|
-
│ └─ Non-wording → extraction_review_gate review +
|
|
23
|
+
│ └─ Non-wording → extraction_review_gate review + challenge (wrapper-fixed budget)
|
|
24
24
|
├─ f. Apply fixes, re-sanitize
|
|
25
25
|
├─ g. Commit per batch on a feature branch → MR pending review (never push to main)
|
|
26
26
|
└─ h. Update charter completion log
|
|
@@ -91,10 +91,10 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
|
|
|
91
91
|
|
|
92
92
|
- When required: see `references/dual-track-review-gate.md` table.
|
|
93
93
|
- Choose the review tier from that table, not from intuition. Do not restate the rows locally; record the exact `dual-track-review-gate.md` table row used. Record `challenge: not-required` only when that row classifies the actual diff as challenge-not-required (for shared skills, this means strict wording-only with deterministic scope proof + independent review confirmation). Non-wording shared-skill changes cannot skip challenge.
|
|
94
|
-
- Run deterministic checks and implementer self-review first, and record what each proves before invoking review/challenge (this self-review-before-review ordering applies to every non-wording shared-skill change the dual-track table requires review for, not only the rows that look high-risk): `git diff --check` proves whitespace/conflict-marker hygiene only; validators prove schema/link/routing invariants; leakage/sanitization scans prove only their configured patterns; scope checks must name the changed files or expected file set; the self-review row is conclusive only when each required field is non-empty (acceptance criteria, changed-file scope, edge/failure paths, known residual risks) and the changed-file scope equals the candidate diff's changed-file set, or explicitly explains any excluded generated/irrelevant file. Persist it before the review/challenge run in a fresh, non-overwritten task-evidence path outside the candidate diff, pass that exact file as the gate's review plan, and retain the gate result that binds its profile hash; do not edit the candidate merely to record self-review or review outcome, because that creates self-referential candidate churn. A candidate-local row is appropriate only when the row itself is a substantive deliverable under review. A plain in-place-editable MR description or scratch log is not ordering proof unless its edit history is retrievable and checked; a backfilled row is invalid and forces a rerun. If the candidate diff changes after the row is saved — a file added/removed OR the content of any listed file materially changed — refresh the row
|
|
94
|
+
- Run deterministic checks and implementer self-review first, and record what each proves before invoking review/challenge (this self-review-before-review ordering applies to every non-wording shared-skill change the dual-track table requires review for, not only the rows that look high-risk): `git diff --check` proves whitespace/conflict-marker hygiene only; validators prove schema/link/routing invariants; leakage/sanitization scans prove only their configured patterns; scope checks must name the changed files or expected file set; the self-review row is conclusive only when each required field is non-empty (acceptance criteria, changed-file scope, edge/failure paths, known residual risks) and the changed-file scope equals the candidate diff's changed-file set, or explicitly explains any excluded generated/irrelevant file. Persist it before the review/challenge run in a fresh, non-overwritten task-evidence path outside the candidate diff, pass that exact file as the gate's review plan, and retain the gate result that binds its profile hash; do not edit the candidate merely to record self-review or review outcome, because that creates self-referential candidate churn. A candidate-local row is appropriate only when the row itself is a substantive deliverable under review. A plain in-place-editable MR description or scratch log is not ordering proof unless its edit history is retrievable and checked; a backfilled row is invalid and forces a rerun. If the candidate diff changes after the row is saved — a file added/removed OR the content of any listed file materially changed — refresh the row; any rerun of review/challenge against the new candidate draws on the remaining cross-chain Agent budget (the self-hosted-chain rule in `references/dual-track-review-gate.md`), and at the cap, or at the effective exhaustion that rule defines, the terminal-disposition path governs instead of a rerun. Changing only the external self-review record refreshes the profile binding; it does not by itself invalidate implementation tests or the candidate packet. A missing field, "ok" placeholder, mismatched scope, or unprovable ordering makes the row inconclusive. Do not spend LLM review rounds on issues a script or implementer-side checklist can decide. If the independent pass is the first place basic scope, contract, privacy, or test issues surface, apply those findings to the diff, close the self-review gap, and rerun the deterministic gates before rerunning review/challenge; the process-defect repair is in addition to resolving the findings, not a way to discard or downgrade them.
|
|
95
95
|
- Review pass: persist the complete self-review row and encode it in the review plan. For a **non-wording** lane, resolve the repository-owned `scripts/extraction_review_gate.sh` and use it from round 1; never substitute the generic controller, scan writable plugin roots, or supply a caller-selected budget. For a strictly proven **wording-only** lane, use the generic `code-review` proof-bound single-review recipe in `code-review/references/staged-review-contract.md` and record `challenge: not-required`; require its controller-derived wording scope plus the independent `wording_only_boundary` confirmation. This is the only extraction path that stays outside the multi-round wrapper and terminal ledger; the gate, not this page, decides whether a chainless review is legal, and it may still demand the tracked pair. Take all controller options from that runnable recipe, supplying the actual stage and exact candidate rather than an example default. The non-wording chain cannot be retrofitted, so a run started outside its owner wrapper is thrown away and restarted. Read the chain-opening and packet-composition rules in `references/dual-track-review-gate.md` first. Require conclusive JSON, selected-client attribution, packet/profile binding, family exclusion, and wrapper runtime evidence. When the host returns a live execution handle (`session_id`, `cell_id`, or equivalent), keep polling that exact handle until terminal exit; empty current output is progress, not a verdict, and no replacement/fallback reviewer may start while the original process is live. The result row records handle type, an opaque host transcript/tool-call reference and terminal exit status. If the handle is lost, the lane is infrastructure-inconclusive/manual-review-required and no replacement or fallback may be started or credited; process-tree and wrapper artifacts are diagnostic only. This is a procedural host obligation because the inner gate cannot observe the outer handle. Never copy a credential-like raw handle into shared evidence. `findings` is not pass; inconclusive, malformed, or free-form output stays interim. Do not add a separate behavior probe.
|
|
96
96
|
- Challenge pass: for a non-wording lane, invoke `scripts/extraction_review_gate.sh` separately with the same plan, stage, candidate, family and tracked chain. Pass the next one-based index; later rounds include a distinct focus and all prior focuses. Preserve a separate result row with the same binding, exclusion, egress, attribution and conclusive checks. Review never satisfies challenge; missing or inconclusive required challenge keeps extraction interim. A wording-only lane has no challenge pass.
|
|
97
|
-
- Treat review/challenge as batch-level gates over the landing candidate, not as a per-bullet or per-line edit loop. Apply all findings from a round; when both lenses are required, re-run both on the updated candidate before landing.
|
|
97
|
+
- Treat review/challenge as batch-level gates over the landing candidate, not as a per-bullet or per-line edit loop. Apply all findings from a round; when both lenses are required and cross-chain Agent budget remains, re-run both on the updated candidate before landing — every re-run sums into the same wrapper-fixed budget, and at the cap, or at the effective exhaustion that rule defines, the terminal-disposition path in `references/dual-track-review-gate.md` replaces further re-runs.
|
|
98
98
|
- Skipping a required challenge = work can only land as interim, not complete.
|
|
99
99
|
|
|
100
100
|
#### 3f. Apply fixes, re-sanitize
|
|
@@ -119,7 +119,7 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
|
|
|
119
119
|
- File: `~/.<host>/skills/.extraction-work/<project>-completion.md`
|
|
120
120
|
- Final state: which batches done, which deferred, which sources unavailable.
|
|
121
121
|
- Lessons: what surprised; what would change in next extraction; what to add to skill-extraction-workflow.
|
|
122
|
-
- For every non-wording review chain, build the receipt-bound closeout ledger and run `scripts/validate_extraction_review_state.py <closeout.json>` before reporting a terminal state. A clean Round 2 plus its exact-candidate completion receipt may validate as `ready_for_human_decision`; Round
|
|
122
|
+
- For every non-wording review chain, build the receipt-bound closeout ledger and run `scripts/validate_extraction_review_state.py <closeout.json>` before reporting a terminal state. A clean Round 2 challenge plus its exact-candidate completion receipt may validate as `ready_for_human_decision`; Round 2 findings at the exhausted budget validate as `continuation_authorization_required`; a second ordered base drift validates as `baseline_race`. Unknown, stale, omitted, or invalid evidence remains `interim`. The strict wording-only single-review path records its independent review row but does not fabricate a multi-round ledger.
|
|
123
123
|
|
|
124
124
|
### 5. Provenance migration
|
|
125
125
|
|
|
@@ -149,7 +149,7 @@ Skip this step when nothing transferable surfaced.
|
|
|
149
149
|
| Anti-pattern grep panel | `references/recurring-anti-patterns-checklist.md` | Every commit; ~30s |
|
|
150
150
|
| `check-ccl-skills.sh` | `scripts/check-ccl-skills.sh` | Every commit; ~10s |
|
|
151
151
|
| Generic `code-review` gate | repository-owned skill | Strict wording-only independent review; ~5-10 min |
|
|
152
|
-
| `scripts/extraction_review_gate.sh` | this skill package | Non-wording review plus
|
|
152
|
+
| `scripts/extraction_review_gate.sh` | this skill package | Non-wording review plus the wrapper-fixed challenge budget; ~5-15 min each |
|
|
153
153
|
| `scripts/validate_extraction_review_state.py <closeout.json>` | this skill package | Every non-wording terminal checkpoint |
|
|
154
154
|
| Source-read fallback ladder | `SKILL.md` Source-read remediation | When a source read fails or times out |
|
|
155
155
|
| Sibling mini-map | `SKILL.md` Step 4 stack-specific updates | Every stack-specific change |
|
|
@@ -24,7 +24,7 @@ The at-add-time check above decides *where* a rule lands (merge vs new bullet);
|
|
|
24
24
|
|
|
25
25
|
| Baseline failure | Right form | Wrong form |
|
|
26
26
|
|---|---|---|
|
|
27
|
-
| Agent knows the gate and walks past it under pressure (discipline slip — our merge-authorization / R0 / done-claim class) | Prohibition + rationalization-vs-reality pairs + red-flag self-check | Soft guidance ("prefer…", "consider…") |
|
|
27
|
+
| Agent knows the gate and walks past it under pressure (discipline slip — our merge-authorization / R0 / done-claim class) | Prohibition + rationalization-vs-reality pairs + red-flag self-check — pairs quote excuses actually captured verbatim from baseline/pressure runs (`validation-and-landing.md` eval-first step 2), never invented ones | Soft guidance ("prefer…", "consider…") |
|
|
28
28
|
| Agent complies but the artifact's SHAPE is wrong (bloated review packet, buried verdict, register row restating the source) | Positive recipe/contract: state what the artifact IS — its parts, in order | A "don't"-list about the shape — the measured backfire above |
|
|
29
29
|
| Agent omits a required element from an artifact it already produces (missing status/evidence cell, absent map row) | A REQUIRED slot in the template/validator it must fill (our closeout rows and register gate are this form) | Prose reminders near the template |
|
|
30
30
|
| Behavior should differ by situation | Conditional keyed to an observable predicate ("fan-out → name the tier") | Unconditional rule + exemption clauses |
|
|
@@ -459,5 +459,37 @@ and inverted the sense (production, not product), and the coordinator now shares
|
|
|
459
459
|
| A compression that drops a scope qualifier widens the rule it hosts: the external-pack routing clause routes only a MISSING method/tool-layer capability, and restoring the dropped word is the fix, not rewording around it | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#M needs the functional-equivalent check | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: a size-budget compression rewrote the reference-only routing clause from routing a missing method/tool-layer capability to routing that layer categorically, which would displace locally covered P-verdict capabilities and contradict the functional-equivalent check landed in the same round; caught by the supplementary post-delta challenge round enumerating compressed sentences (same class as the metered model/tool qualifier drop fixed in the sibling owner). Fix: the missing-capability qualifier restored; byte offsets from this branch's own pointer sentence, whose semantics live in the owning reference. RED baseline (replayed): word-level diff against the base revision showed the dropped qualifier before the fix and shows it restored after; the class sweep over all three owner entrypoints found no further load-bearing drops. |
|
|
460
460
|
| Stop reporting keeps its specificity qualifiers: the entrypoint demands the concrete stop reason and the exact evidence checked, and budget offsets come from relocating clauses whose semantics already live verbatim in the owning reference | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#state the concrete stop reason and the exact evidence checked | `updated` | Owner key `product-rd-workflow/SKILL.md`. Observed failure: a size-budget compression dropped the-concrete/the-exact from the stop-reporting sentence, licensing generic stop reports; the challenger graded it load-bearing, the implementer's word-sweep had graded it neutral, and the maintainer's standing delegation resolves such token disputes by the repository's fail-closed obligation standard, so the qualifiers are restored. Byte offset: the stale-source parenthetical is removed from the entrypoint because its full sentence lives verbatim in references/pre-final-continuation-gate.md (Status-source reconciliation) which the same sentence already cites — relocation, not compression. RED baseline (replayed): grep for the restored phrase zero-hit on the pre-fix entrypoint, exactly one hit after; the removed parenthetical greps once in the owning reference. |
|
|
461
461
|
| The dual-track reviewer's verification scope is a documented boundary: content semantics belong to the reviewer, deterministic-gate claims to CI, historical-process claims are testimony unless receipt-bound — ruled on once so packet-verifiability findings stop recurring per round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#a finding that only restates this boundary is dispositioned against this rule, never re-litigated per round | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; the change lands in references/dual-track-review-gate.md (new Reviewer verification scope section). Observed failure: the packet-verifiability finding class recurred across four supplementary review rounds and roughly a dozen occurrences in this round's chains — every reviewer independently rediscovered that the packet cannot carry the deterministic oracles, and every round paid the same finding again because the boundary was undocumented. The maintainer confirmed the operating reality (all consumers and reviewers are agents; the human role is authority, not readership), so the boundary is now standing text agents can disposition against, with the receipt-embedding backlog item named in place. RED baseline (replayed): grep for the boundary phrase zero-hit before this change, exactly one hit after, on an added normative list line. |
|
|
462
|
+
| The Agent-autonomous review budget is summed across review chains, never per chain: in a self-hosted skill repository a finding fix that touches a selected-owner tree breaks the tracked chain by design (nearly every fix in such a round; a fix confined to files outside every selected owner drifts only the candidate hash and continues in-chain), so below the cap the break is recovered by a ledger-counted restart (batched dispositions first, full-context first packet, plan frozen with the candidate, final chain still able to hold the review-plus-challenge ready floor), and at the cap the designed terminal is disposition plus the honest terminal record — `continuation_authorization_required` when the final round itself returned findings, otherwise an interim record naming the last reviewed candidate and all later deltas — while restarting chains until a clean pass, or counting a restart as fresh budget, is the named contract violation | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#Sum spent rounds across all chains before opening one more | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; changes land in references/dual-track-review-gate.md — new self-hosted-chain rule with a four-step walked enumeration, the continuation bullet's dead-end sentence scoped to the at-cap case, an anti-pattern merge — and in references/extraction-quickstart.md 3e, whose two rerun clauses gain the cross-chain budget qualifier). Observed failure (production receipts, per-host archive, mechanism at the same commit as the current base): one round ran 20+ reviewer rounds, the next 12 restarted chains / 21 reviewer invocations to land a three-line diff, none human-authorized; receipts show every chain restarted at index 1 with a new candidate hash after fixes, while a control chain on an unchanged candidate ran review plus two challenges without invalidation. Root cause is a corpus-level contradiction, confirmed by frozen-criteria elicitation runs (n=3 per arm, isolated cwd, arms byte-identical to versioned text): the pre-change gate section alone elicits the STRICT reading three of three — zero autonomous restarts, interim-then-human after any break — while the quickstart page simultaneously mandated "re-run both on the updated candidate before landing" and the wrapper mechanically accepts fresh chains; jointly unsatisfiable, resolved in production by improvised unbounded restarts. RED-baseline: the occurred production failure plus the three-of-three strict/mandated contradiction on the pre-change corpus; post-change runs elicit the landed semantics three of three (cross-chain summing to the three-round cap, ledger-counted restart below it, terminal disposition at it, no laundering) with no contradiction-rationalizing text. Semantics delta declared honestly: below-cap restarts move from ask-human (strict reading) to Agent-autonomous ledger-counted rounds — a deliberate loosening grounded in the review-efficiency adjudication, the wrapper's three-round design, and two informed merges of rounds that ended in the terminal-disposition shape; the maintainer can revert to the strict reading by decision. Supersedes by pointer the recovery clause of the earlier four-process-controls row ("recovered by an interim checkpoint … plus a human continuation authorization"): that recovery is now the at-cap path only, and the row stays unedited per the append-only contract. Oracle: delete this row on the candidate and run the repo check with the base ref set — it prints the impact-chain missing token. |
|
|
463
|
+
| "Candidate edits do not reset Agent authority" promises no continuation: the selected-owner digest hashes each selected owner package's current working tree and owners derive from candidate paths, so a candidate edit inside any selected owner package invalidates every prior receipt and the next tracked round fails `review_chain_invalid` — for a self-hosted skill-repo candidate that is every applied fix, and a restarted chain re-enters the same cumulative Agent budget | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: file:skills/code-review/references/staged-review-contract.md#invalidates every prior receipt | `updated` | Owner key `code-review/SKILL.md` (entrypoint unchanged; the change adds two consequence bullets to the Agent review chain section of references/staged-review-contract.md). Observed failure: the section's tolerance sentence ("older candidate hashes; they remain consumed") reads as continuation-after-fix, while the stable-binding predicate in review_gate.py (this owner's script) compares controller digest, owner-selection source, owner names, and selected-owner digest on every prior receipt, with the digest computed over the owner packages' live working tree — so the documented tolerance is unreachable exactly when the candidate lives inside its owner package; archived receipts from one extraction round show 12 chains each restarted at index 1 with a new candidate hash after fixes, and a control chain on an unchanged candidate continuing three rounds. Declaration-contradicted-by-implementation class: the correction documents the implemented refusal instead of changing it — binding semantics, trust model, wrapper, and validator behavior are untouched. RED-baseline: grep for the consequence phrase is zero-hit before this change and exactly one hit after, on an added normative list line; the mechanism is reproducible from the chain-validation predicates plus the archived receipts. |
|
|
464
|
+
| The extraction lane's autonomous budget is one review plus one challenge, pinned in exactly two executable places and derived everywhere else: the wrapper passes the single value, the closeout validator computes every numeric bound from two module constants, and prose surfaces name the wrapper-fixed budget instead of repeating numerals — under this budget a fix-restart is never fundable, so the hold-fixes branch is the standing path, the self-hosted chain-break conflict becomes unreachable without touching the binding trust model, and a two-receipt final-round-findings chain validates as `continuation_authorization_required`, closing the cross-chain machine-terminal gap | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (dual-track bullet numerals synced; frontmatter untouched; severe-entrypoint byte budget held at net zero by equivalent shortenings in the same bullet). The budget-size choice was maintainer-delegated in-session after two cost escalations (a review loop can burn two days on one issue) and restores the earlier one-review-plus-at-most-one-challenge adjudication for this repository; the challenge remains mandatory for every non-wording shared-skill change. Changes: extraction_review_gate.sh passes the fixed budget 1 and its guard message matches; validate_extraction_review_state.py derives round bounds, remaining counts, per-round state legitimacy, and the continuation predicate from WRAPPER_CHALLENGE_BUDGET/WRAPPER_AUTONOMOUS_ROUNDS (three hidden hardcodes found and converted during the green run: completion remaining, review_state round semantics, the continuation receipt count); both regression suites' fixtures converted from three-round to two-round shapes with occurrence semantics preserved (multi-finding rounds keep sweep-triggering occurrence counts). RED-baseline (applied): reverting the wrapper to the old budget in a throwaway edit turns test_extraction_review_gate.sh red at the budget-argument assertion and the restore is green; the validator suite was red at each hidden hardcode until converted, then fully green; catalog byte gate and implementation-gates suite green on the final candidate. Supersedes by pointer the wording of the two rows above where they cite a three-round cap or an unfundable-restart arithmetic tied to it: their production evidence and anchors are unaffected, and the budget-agnostic four-step enumeration they land is unchanged — only the numeral moved. Elicitation runs were re-taken against the final section with a corrected prompt (the scenario's own budget parenthetical had contradicted the attached rule text; the stale batch is archived unscored). |
|
|
465
|
+
| The closeout validator must reject a controller chain longer than the wrapper can mint: without an upper bound, a caller-supplied third receipt under budget one claims a negative remaining count and reaches completion validation as a ready-state budget bypass; and every prose surface advertising the retired budget is a laundering affordance, so the stale phrases are closed and pinned closed by documentation assertions | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (unchanged this batch; changes land in validate_extraction_review_state.py — the receipts loop now fails `controller chain exceeds the wrapper budget` past WRAPPER_AUTONOMOUS_ROUNDS — its regression suite, the gate suite's documentation assertions, dual-track-review-gate.md, extraction-quickstart.md, and the register rows above). Provenance: the post-budget batch of this candidate's own two-round review chain — the final challenge (receipt archived per-host) surfaced the over-budget bypass and the stale third-round/two-challenge phrases; the review round surfaced the stale phrases independently, the uncertified-post-batch gap (closed by the pending-branch certification sentence in the terminal-disposition step), and a residual absolute in the consequence row above — superseded by pointer here, not edited in place per the ledger's append-only contract: read its "every applied fix" as "nearly every applied fix; a fix confined to files outside every selected owner drifts only the candidate hash and continues in-chain", matching the normative text it records. RED-baseline (applied, red for the right reason): the new over-budget fixture was added BEFORE the validator fix and the suite went red showing the unfixed validator accept the three-receipt budget-one ledger as ready_for_human_decision; after the one-line bound the case rejects with the named token and the full suite is green. Packet-verifiability findings about unshippable suite output remain dispositioned against the documented reviewer-verification-scope boundary with rerun oracles in the MR/PR body. |
|
|
466
|
+
| The assent-triggered blocked outcome states its interim classification with an explicit verb, not an elliptical fragment | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#form and classifies the turn `interim` | `updated` | Owner key `product-rd-workflow/SKILL.md`. Debt-repayment chain for the disclosed unreviewed pinned-phrase restoration commit 4e65dbc (patch-identical pre-rebase form 4b30746): replayed against its immutable base b0fe08b, `git show b0fe08b:skills/product-rd-workflow/SKILL.md` carries "form and classifies the turn `interim`." and the 4e65dbc diff drops "and classifies the", leaving the malformed "form, turn `interim`" — the independent review of this repayment chain confirmed the verb loss as the only defect of that diff still unrepaired at review time (the fused "reconfirm A" sentence boundary was repaired by 5fc25fb and the ledger-table break by the note relocation in 26b773e, both verified against the current file state); head restores the classification verb from the b0fe08b text of the same sentence, whose later specificity/carrier revisions elsewhere in the sentence are untouched. |
|
|
462
467
|
|
|
463
468
|
Supersede note (this round, before landing): the two crypto-erase rows above ("go-microservice-architecture" and "python-service-architecture") describe an earlier candidate state; the landed text attributes the CE do-not-use conditions to SP 800-88 Rev.1 §2.6 with Rev.2 (2025) superseding and continuing the framework — per the source-verification ledger row "CE conditions text location". The rows' RED-baseline probes and firing-path anchors are unaffected.
|
|
469
|
+
|
|
470
|
+
Round 073-receipt-bundling rows (new table so the entry renders as a table row after the supersede-note paragraph above):
|
|
471
|
+
|
|
472
|
+
| Upstream rule | Downstream owner | Expected executable behavior | Status (updated, unchanged, routed, or not-applicable) | Evidence |
|
|
473
|
+
| --- | --- | --- | --- | --- |
|
|
474
|
+
| Deterministic-gate process claims ride the packet as candidate-SHA-bound receipts: `gate_receipt.py` mints against the committed candidate (clean tree, HEAD, argv, exit code, output hash, bounded tail) and verifies differentially, so a pre-fix RED stops being unverifiable testimony | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#clean tree required, HEAD commit recorded | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (Reviewer verification scope — the "Standing backlog" clause is superseded in place by the landed channel) + scripts/gate_receipt.py + scripts/test_gate_receipt.sh (registered in the fast lane). Observed failure: the packet-verifiability class — historical-process claims graded implementer testimony across the prior round's review chains, terminally dispositioned accepted-limitation repeatedly, with the backlog row naming receipt embedding as the mechanical fix. Remedy re-derived against current code (a deferred registration is hypothesis, not spec): the packet already binds `--review-plan-file` evidence via review_context_sha256 and the v3 ledger already binds sibling files by hash, so no controller or validator change — the missing pieces were the receipt artifact class and its mint/verify tool, which is what lands. Trust model stated in the landed text so it cannot be oversold: candidate-bound falsifiable consistency evidence, not runner authentication; CI re-running gates stays the deterministic authority; unreceipted process claims remain testimony. RED baseline (applied mutations, differential): tampered output hash → rc1 output_hash_mismatch; tampered recorded exit code → rc1 exit_code_mismatch; foreign key → rc1 exact-key-set; wrong checked-out candidate → rc2 named no-verdict (not a false red); nondeterministic-output gate → full rerun rc1, --exit-only rc0 with scope token; control legs green. Same-round sibling disposition, no diff landed: the two-place numeric-contract backlog item (guarded-file whitelist + version-bump receipt + parse-values-from-prose) was re-probed against the current baseline and closed already-covered — the pinned/sync declared-pair registry owns the whitelist half and test_routing_pointer_integrity.sh's doc-vs-executor threshold parity check owns the parse half; that incumbent fired live twice this round against a draft duplicate (executor marker loss; a pinned firing-path phrase reworded), so the duplicate was deleted per same-class convergence-by-deletion and the final candidate leaves SKILL.md, description-authoring.md, and the checker cap logic byte-identical to base. |
|
|
475
|
+
| Load-bearing prose contracts are pinned declaratively: each contract-anchors.tsv row demands its pinned literal exactly once in the owner file, so deletion, semantic inversion, numeric falsification, or decoy duplication of a verdict-taxonomy discriminator, a stop-condition predicate, or an externally verified numeric tier turns the repo check red instead of passing clean | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/check-contract-anchors.sh + contract-anchors.tsv (9 anchors) + the check-ccl-skills.sh delegation block (beside-the-validator resolution, pinned green grammar, fail-closed rc mapping) + the fast-lane suite). Observed failure: three enforcement-gap findings from the prior round's review chains, deferred needs_human_decision and scheduled into this round by the maintainer — reproduced against the round base before any implementation: deleting the fault-origin discriminator sentence, inverting the materially-differing/evidenced-cause stop predicates, and falsifying externally verified values (14.4→12.4, 97→87) each left the full check clean_ok while a control mutation (broken reference link) went red, proving the instrument could fail. RED-baseline (applied, differential): the same probe mutations now red the gate with per-anchor attribution; suite mutants M1–M8 red for their named reason (missing/duplicate/file-missing/empty-table/malformed-row/short-literal/duplicate-id/table-missing), benign neighbors B1–B3 green, D1 names only the broken anchor, and the whole suite goes red under an always-green checker stub. Scope stated honestly: anchors make contract-wording and pinned-value changes conscious (same-MR table edit), not externally re-verified — external-truth re-verification stays with the documented reviewer-verification-scope boundary, and stop-predicate semantics testability is routed to the evals layer (deferred, bound to that round's entry). Registered-remedy narrowing recorded: the N×M forbidden-token matrix is not landed (three mechanized anti-patterns, the 1×1 pending/clean exclusion, and the sync registry already own every observed shape; recurrence re-opens it) and numeric copy-prohibition narrowed to both-sides anchors (entry + reference pinned together) because repo values are already single-owner. |
|
|
476
|
+
| A pinned-phrase gate family must be provably able to go red for the right reason: one applied deletion mutation per family runs the FULL shipped checker against a committed fixture clone and demands that family's own red token, with the unmutated control run green on every family token; and every anti-patterns panel section must carry its Grep recipe | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in the heavy-lane walk + the fast-lane panel structural check test_antipattern_grep_panel.sh). Observed failure: none of the ~40 required_phrase pins across the four inline gate families had ever been proven able to red — the always-green-oracle class this repository already met as the degenerate-fixture worktree-pruning precedent and the empty-glob false-green lesson (an emptied phrase list or quoting regression would certify silently). RED-baseline (applied, differential): walk legs W1–W5 each delete one currently-pinned phrase in a committed fixture clone and the full checker reds with exactly that family's token (project-assessment, task-retro/teammate-trigger, test-case-first, product-rd anchor, and the new contract-anchor delegation), ~70s total; the control leg is green with all five family green tokens and a loop-count floor guards list vacuity; the panel check reds on a stripped Grep line naming the right section and stays green for a benign non-anti-pattern section. Environment pitfall recorded for reuse: CCL_SKILL_BASE_REF must stay unset around nested gate runs — it leaks into child validators' synthetic self-test repos where HEAD always resolves and flips their no-base→degraded legs into false passes. Rule side already-covered: the killing-mutation walk, benign-near-miss precision rows, and oracle-validation duties are owned by testing-strategy and the dual-track Self-audit section — no prose added anywhere; the firing path for this failure class is now these suites in the fast/heavy lanes. Grep pattern-compile validation deliberately discarded (panel recipes mix GNU-BRE commands with prose instructions by design; a compile check would false-red on regex-dialect differences without protecting a real contract). |
|
|
477
|
+
|
|
478
|
+
| Eval-first authoring fires pre-draft: a NEW skill needs three-plus scenarios and a no-skill baseline before body text, and a non-failing control stops the draft | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#must define eval/pressure scenarios, baselines, acceptance criteria pre-draft | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. 075/R3 round (S9). Both external sources re-verified primary-source on 2026-08-31: Anthropic "Skill authoring best practices" (https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices, section "Evaluation and iteration" / "Build evaluations first") — create evaluations BEFORE writing, build three scenarios, establish the baseline without the Skill, write minimal instructions, and no built-in runner exists; superpowers plugin 6.3.0, skills/writing-skills (SKILL.md sections "The Iron Law" and "Micro-Test Wording Before Full Scenarios"; testing-skills-with-subagents.md section "RED Phase") — a failing test first for new skills AND edits, 3+ pressure scenarios, and a no-guidance control that does not exhibit the failure means stop, do not author. The commit-time RED-baseline contract is untouched in this diff (the semantic-control leg); the change moves the firing point onto the pre-draft transition per the firing-point-placement corollary, with mechanics merged into the existing canonical bullet in validation-and-landing.md rather than appended as a new rule. Zero-loss map for the rewritten Workflow step-2 sentence: the subjective/high-impact category list survives (frontend/client shortened to client, same referent), pressure scenarios widen to eval/pressure scenarios, acceptance criteria survive verbatim, and before-editing tightens to pre-draft; nothing dropped, NEW-skill coverage and baselines are the additions. |
|
|
479
|
+
|
|
480
|
+
| Deterministic anchors pin stop-predicate wording while paired body-compliance probes grade classification on the gate's own continuing:/blocked: markers | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#probe subset on this machine before landing | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Closes the 071-chainB-r1f1 semantic-testability follow-up deferred by 074 to R3. Registered form (invert predicate, gate stays clean) no longer reproduces: the entrypoint anchor gate, the register firing-path anchor, and the size ratchet each redded a separate applied polarity/reword probe in a disposable checkout at the round base, with a pristine control leg green first. Evolved form reproduced and is the RED leg: a word-compensated additive neutralization (appending an advisory-continue sentence while deleting equal unpinned words) exits 0 with ccl_skill_check_clean_ok. With-change leg: four paired prd-* probes graded 4/4 on the pristine body after instrument fixes (marker-decoration tolerance with a mention-vs-verdict grammar; shared-scaffolding single-variable isolation), and the probe set demonstrably can fail (first run graded 2/4 on real output, one miss being a genuine same-case-two-classifications observation); run reports committed under eval/evidence/stop-predicate-probes-2026-08-31/. Honest boundary recorded in the f4 layering section (eval-routing.md points there): the same applied neutralization mutants did not flip live agent behavior either (two mutants by two probes, small N) — probes carry behavior drift, not buried-sentence tripwires. |
|
|
481
|
+
|
|
482
|
+
| Negative controls and coverage-gap probes are first-class routing-bank rows (expected "none" sentinel, acceptable alternates, bait neighbors), with absorbed / ownership_split as labeled outcomes and clarify / low-confidence / replica-agreement as first-class report metrics | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#已修复的路由 miss 必须把其 utterance 冻结成 bank task; bank-evidence: command:skills/skill-extraction-workflow/scripts/eval-routing-bank.rb | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Live at the round base: grading the new gap probes with two replicas showed one replica absorbing each probe into a coordinator skill while the other rejected it — the runner's absorbed + ownership_split labels fired on real grader output before the expectations were corrected with acceptable[]; integrity-lane mutants (sentinel in must_not, acceptable restating expected, unknown acceptable target, acceptable∩must_not, empty-string fields) each turned test_routing_bank_integrity.sh red on its own named assertion with the unmutated control green, and the validator self-proof section now replays those mutants on every run. Durable artifacts (exact invocation, grader model, candidate fingerprints, raw per-replica verdicts, red/green transcript, gate exits): eval/evidence/routing-negative-controls-2026-08-31/. |
|
|
483
|
+
|
|
484
|
+
| A before/after routing comparison may vary only ONE routing variable (one description, or one skill's indivisible routing face) for its delta to be attributable | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#一次改前/改后对照只准动**一个路由变量** | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source-side paired single-description A/B is the observed working mechanism (borrow round; sanitized provenance in the private alias archive); no mis-attribution incident observed in this repository yet, so the clause lands as protocol item 5 with the existing four items unchanged as the paired control. |
|
|
485
|
+
|
|
486
|
+
| Cross-skill / cross-reference routing pointers in body text must carry the routing quadruple (trigger / scope / output / return point); a bare "refer to X if useful" pointer is never a landing shape | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/description-authoring.md#routing pointer in body text must carry the routing quadruple | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Borrowed from an adopting skill pack where unbounded pointers were the dominant dead-routing shape (two adopters verified in source; sanitized provenance in the private alias archive); landed in the routing-surface authoring reference because the entrypoint is size-ratcheted level — the reference is the required pre-edit reading for routing-surface work, and eval-routing.md's silent-skip row points back at it; the description-side Skip-when idiom already satisfies the quadruple and is named as the unchanged control. |
|
|
487
|
+
| Review-finding fixes are held un-applied until the full review+challenge chain has run on the frozen candidate: under the 1+1 budget apply-now is never fundable after the review round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (enumeration item 1 + cadence Round 2). Observed failure: a prior round applied its review fixes before the challenge; the tracked chain broke (challenge binds to the round-1 candidate), the round lost its double-receipt terminal, and closure required a user-granted continuation chain. The prior wording stated the rule only as a fundability conditional whose arithmetic the agent under pressure never ran; the operative unconditional form (hold all fixes; challenge on the frozen candidate; land the batch after the chain) is now explicit at both firing points. RED baseline: the recorded chain-break incident is the without-change failure; the with-change compliance surface is the explicit hold rule at the enumeration walked when a round returns findings. |
|
|
488
|
+
| A frozen eval case is sacred: deleting or re-scoping a bank task or golden trace requires a same-round `case-retired:`/`case-rescoped:` register adjudication row, and average improvement never offsets a frozen-case loss | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#平均改善不得抵消单条冻结案例的失守 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/eval-routing.md (冻结案例神圣 bullet), scripts/test_frozen_case_sanctity.sh (fast lane, registered), docs/f4-skill-effectiveness-harness.md (pointer line). Rule semantics: a previously-passing frozen case that degrades — including to unsure/INCONCLUSIVE — is a regression, and its sacredness attaches per case, so no aggregate improvement offsets it; the mechanism's source-verification record lives in the round's private archive, and transferred evidence does not exempt the behavioral row. RED baseline (replayed, throwaway clone at the round base): deleting the non-pinned bank case ctrl-unit-test passed the pre-change surface silently (test_routing_bank_integrity.sh exit 0) and reddens the new gate (exit 1 naming the id and the required adjudication row); re-scope and golden-trace-deletion mutants red for the right reason; adjudicated-deletion and untouched-tree control legs green; no-base and unresolvable-base legs print the explicit skip token. |
|
|
489
|
+
| Rationalization tables are built from excuses captured verbatim in baseline/pressure runs, never invented: counter only what a run actually said, and new observed excuses accrete counter rows | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/validation-and-landing.md#counter only what a run actually said | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/validation-and-landing.md (eval-first steps 2 and 4) with the sourcing clause mirrored into references/rule-consolidation.md's discipline-slip form row. The clause is scope-bounded to discipline-slip failures (prohibition-form counters measurably backfire on shape/omission failures per the form table it points at); the source-verification record lives in the round's private archive. RED baseline (fair tempting scenario, headless fresh-context, small-N 2x2, fully separated): asked to harden a merge-authorization gate with no run data supplied, the no-clause arm invented 12+ Excuse-Reality rows in both reps with zero mention of captured evidence; the with-clause arm produced zero invented rows in both reps and stopped to request actual run transcripts. |
|
|
490
|
+
| The draft-time security axis walk screens skill text as a prompt: leakage-inducing, permission-overreaching, or unsafe-automation wording is named at the axis-1 instance list and may appear only as a labelled anti-example | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#text a draft must not carry except as an explicitly labelled anti-example | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (pre-cover axis (1) instance list). Semantic control: axis (1) already required a security/authority pre-cover with at least one negative case per applicable axis; this names three instances (prompt leakage, overreach, unsafe automation) inside the existing obligation rather than adding a new gate, so the walked enumeration, the challenge mandate, and every other axis are unchanged; existing anti-example discussions stay legal via the labelled-anti-example carve-out. |
|
|
491
|
+
|
|
492
|
+
| Reference files carry a delta-ratcheted line budget: a new or crossing `skills/*/references/**/*.md` over 500 physical lines blocks, an already-over reference may shrink or stay level but never grow, the append-only ledger is excluded by gate design, and a long new reference without `##` structure draws a navigation advisory | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#stays inside the reference line budget | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in scripts/check-size-budget.sh (reference ratchet + counters + `reference_line_budget_blocking_ok` / `_failed` / `_unevaluated` verdict tokens), scripts/test_check_ccl_size_budget.sh (legs g0-g9, registered in the fast lane through the existing size-budget suite entry), and references/attention-budget-ratchet.md (the five design invariants every size/budget gate must satisfy: stable proxy estimator, anti-false-green sentinel, zero tolerance for new debt, legacy-shrink-only, missing-baseline-is-never-a-pass — plus the write-side reference norms and the per-clause verdicts on the external source). Write-side gap was measured before implementing, not assumed: 338 reference files, 4 over 500 lines, 104 between 101 and 300, and zero carrying a table-of-contents block. Paired RED at the round base in a throwaway checkout: a new 703-line reference passed the pre-change gate silently (exit 0, `entrypoint_size_blocking_ok`) and reds the new one (exit 1, naming path, head_lines=703 and the 500 allowance); on the live corpus, appending one line to the 691-line source-to-skill-extraction.md blocks as `over-limit reference grew`, deleting three lines passes with `reference_line_budget_legacy_ok allowed_lines=691`, appending 200 ledger rows passes with the `ledger_excluded` visibility token, and the untouched tree is green with over_limit_count_delta=+0 (the corpus is frozen, not retroactively reddened). Oracle self-proof (applied mutants on a copied gate, differential, control green): disabling the over-budget branch, allowing legacy growth, an off-by-one budget, removing the ledger exclusion, downgrading the unknown-base partial to a print, dropping the reference verdict from the clean-exit aggregate, and promoting the navigation advisory to a block each red on their own named leg; a first mutant attempt that broke the program was discarded as red-for-the-wrong-reason and re-applied semantically. Author-dogfood leg: the gate blocked this round's own entrypoint edit at +56 then +18 body words until the landing was funded by consolidation, which is the invariant working on its author. Review-chain repairs folded into the same round commit (tracked chain 078-r6, 1 review + 1 challenge, all four findings applied after the chain closed on the frozen candidate): line endings are normalized to LF before the line and heading counts, so a CR-delimited 501-line file can no longer read as one line and earn a clean verdict; the navigation advisory counts EXACT H2 headings, so an H3-only long reference still draws it; the head census guards every read and degrades its COUNTER to unknown rather than aborting the program before the per-file fail-closed verdicts run; and the legacy freeze gained a level-edit leg so a `>` silently becoming `>=` cannot pass. Four further mutants (H3-counts-as-section, no line-ending normalization, level-edit-blocks, census-error-reads-zero) each red on their own new leg with the control green, the last of them only after its leg was added — it survived the first re-walk, which is why the walk was re-owed after the fixes. Registered-claim narrowing recorded (a deferred registration is hypothesis, not spec): the official 500-line figure is SKILL.md-scoped and already covered more strictly by the body-word ratchet, the registered table-of-contents mandate is NOT PRESENT in the primary source (the official remedy is one-level references) so it landed as a `##`-structure advisory on repo-internal evidence only, and the three-model test matrix is not mechanized because this repo ships no model-pinned skills. |
|
|
493
|
+
| Core Rules own each invariant and a Workflow step only points at it: same-facet text living in both surfaces is converged toward the canonical surface rather than restated, and the sweep enumerates candidates instead of fixing whichever one a diff happens to touch | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#do not make normal users route through a source name | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The Step 6 cross-section facet-ownership check already owned this rule and fires on any edit touching Core Rules or a step; this round ran it as a five-candidate enumeration rather than a spot fix (the user challenged an earlier framing that treated the duplications as a word-budget offset ledger). Verdicts: content placement (Core Rules content-placement bullet vs two Step 5 bullets) converged to a pointer; the sibling-generalization mini-map field list (Core Rules owner-generalization group vs Step 4) converged to a pointer keeping only the step-order clause; capability naming (Core Rules naming bullet vs Step 5 vs two Step 6 checklist lines) converged to one Step 5 pointer plus one merged Step 6 residual-search check; representative sampling (Core Rules full-ask prohibition vs Step 3 labeling duty) judged complementary and left unchanged; the source-register row schema (Step 3 vs references/source-register.md) left unchanged as out of this contract's Core-Rules-versus-step scope and useful where an author builds rows. Zero-loss obligation map for the four rewritten passages: entrypoint-owns-trigger/routing/non-negotiables and references-own-detail both survive in the Core Rules content-placement bullet; the mini-map field list, the `update`/`unchanged`/`route-to-shared` vocabulary, and the smallest-common-owner routing survive verbatim in the Core Rules owner-generalization bullet, with `then add stack-specific implementation notes only where needed` kept in the step; the capability-name examples, the never-name-after-source clause, and the do-not-route-users-through-a-source-name clause all survive in the step, and provenance-labelling survives in the Core Rules naming and provenance bullets; the Step 6 residual-search list gains the page-name and scenario-label terms the two merged lines carried separately, and keeps `absent from executable guidance or clearly marked as provenance`. Net effect on the frozen entrypoint: base_body_words=16759 head_body_words=16750 (-9), so the round funded its own additions and left the entrypoint smaller than it found it. |
|
|
494
|
+
| A base-relative gate's design-time premise is measured against EVERY base its landing faces resolve, never the current round's base alone: the set is read off CI's own base-resolution expression rather than guessed - one face per pull-request target branch plus the pushed branch's previous tip on a push build - and each resolution is a separate run of the author-dogfood leg, a difference that accumulated before the gate existed surfaces as one violation on the face nobody measured, and the repair is to shrink the frozen surface rather than add a cross-base exemption | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#enumerate the landing faces, never assume one | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key; the changed file is `references/dual-track-review-gate.md`. **Observed failure (RED, recorded incident + re-computable).** The reference line ratchet landed in the prior round measured only against the round's own base. Reproduction on the pre-fix candidate: `CCL_SKILL_BASE_REF=origin/main bash skills/skill-extraction-workflow/scripts/check-size-budget.sh .` printed `reference_line_block: ... dual-track-review-gate.md: over-limit reference grew base_lines=662 head_lines=668` and `reference_line_budget_blocking_failed`, while the same command with `CCL_SKILL_BASE_REF=origin/dev` printed `reference_line_budget_blocking_ok`. The gate behaved exactly as documented; the defect was that leg (a) said `the SAME base resolution CI uses` (singular), so the author measured one face and the six lines that older rounds had added to a frozen surface only became visible on the promotion face. **Compliance (with-change).** Same two commands on the head candidate: base=origin/main `base_lines=662 head_lines=661 allowed_lines=662` and base=origin/dev `base_lines=668 head_lines=661 allowed_lines=668`, both ending `reference_line_budget_blocking_ok`. Running BOTH faces is this round's own dogfood of the rule it lands. **Zero-loss obligation map for the five consolidations that funded the addition** (all within the same file per the frozen-reference funding rule; 668 to 660 lines, and 125978 to 126158 bytes - the line unit the ratchet measures fell, the byte count rose by the amount the new obligation text exceeds the recovered duplication, which is stated here rather than hidden). (1) The four-line raw-CLI preamble collapses to one line carrying all three of its propositions - diagnostics only, never review or challenge evidence, never a replacement for the owner wrapper on a non-wording lane. (2) The standalone do-not-iterate-to-zero-findings paragraph merges into the convergence-bar paragraph, keeping the design-tradeoff clause, the pre-existing clause, the over-correct-or-scope-creep clause, and both contrasts - the bar differs from zero findings AND from no-new-P0-P1. (3) The R0-evidence value menu was stated verbatim three times; two occurrences become pointers to the review-pass row that keeps the menu, matching the pointer form the adjacent Item-9 field already used. (4) The Rules bullet restating the behavioral-evidence table's own two rows is dropped; its one clause absent from the table, that an author cannot self-assert semantic-control, moves into the semantic-control cell itself. (5) The trailing what-a-fallback-is-worth paragraph folds into the reviewer-ladder item that already owns that question, keeping the ad-hoc-run bar, the remediation-versus-evidence distinction, and the interim rule. No obligation was dropped and no paragraph was re-wrapped to buy lines. **A sixth consolidation was attempted and reverted, which is the reusable finding.** The premise-verification bullet in the self-audit section reads as a verbatim restatement of the two paragraphs above it, and deleting it looked free; `register_firing_path_unresolved` then failed because an earlier round's register row anchors its firing path on that exact line. A rule line can be another row's evidence, so an append-only ledger makes some prose non-deletable: check the anchor set before treating any rule line as redundant, and restore rather than EXEMPT when the deletion was to fund your own budget. **Other owners.** `references/attention-budget-ratchet.md` is `unchanged: already-covered` with a real firing path - its five-invariant preamble already says the budget gates are `the budget-gate instantiation of the design-time operability check in dual-track-review-gate.md - run that check's four legs too`, so the tightened leg (a) reaches ratchet authors through that pointer and restating it there would be the same-facet drift the Step 6 check forbids. `scripts/check-size-budget.sh` is `not-applicable`: the gate is correct as shipped and judges whatever base it is handed, so multi-base topology belongs to the caller, not the script. **Residual risk, stated rather than hidden.** The firing path is a walked design-time enumeration in prose, not a mechanical multi-base run; `Makefile` still defaults `CCL_SKILL_DEFAULT_BASE_REF` to the integration branch, so a local check still measures one face unless the author enumerates. A mechanical all-faces target is deferred with an owner - it would be a new gate surface owing its own four legs, oracle self-proof and suite registration, which is a round of its own rather than a rider on this one. |
|
|
495
|
+
| A design-time obligation over a SET is discharged by a manifest a reviewer can diff against the set's authoritative source, never by an asserted walk: the manifest names one entry per member with the value that member resolved to and the check's verdict on it, and a set with no finite manifest is a blocking residual routed to the existing non-blocking or risk-owner-deferral exits rather than reported as coverage | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#recorded manifest, never by a walk you assert | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key; this is the post-review fix batch of the same round, kept as its own commit so the externally reviewed candidate stays identifiable on the branch. **Observed failure (RED).** The first formulation of the preceding row's rule said to run the design-time leg `once per resolution CI can produce`. Both lanes of the dual-track chain independently reached the same defect on the frozen candidate: the review lane found the required set is not statically enumerable because an unrestricted pull-request trigger can resolve any target branch, so an author must either guess a subset and falsely claim completion or cannot satisfy the rule; the challenge lane found the same accumulated cross-base violation can therefore still ship despite apparent compliance, since the unchanged per-base gate cannot detect the omitted face. Two independent lanes converging on self-certifiability is the self-adjudication shape this file already names - the classification verb had no named output that produces the classification. **Fix.** The obligation now takes a manifest derived from the workflow itself, one entry per landing branch carrying the resolved base ref and the gate verdict, which a reviewer can diff against the workflow; and an unenumerable set is routed to the exits the premise-check leg already defines instead of being claimed as covered. No new mechanism is introduced - the fix converts an author-adjudicated claim into a reviewer-checkable artifact using exits that already exist. **Chain state.** Wrapper-fixed budget of one review plus one challenge, both bound to the same frozen candidate digest `70fd34678f99c444bf5b7c808a283380e99201eada6d71d076fd08e2f2ec1789`; fixes were held un-applied across both rounds. The final round itself returned findings, so the honest terminal state is `continuation_authorization_required` and this batch is post-review, not reviewed. A third review finding asked for the workflow file and captured command outputs inside the packet; it is dispositioned per this file's reviewer-verification-scope rule - deterministic-gate claims are verified by CI re-running the gates on the branch, never accepted from the implementer's prose - with the packet-composition miss recorded as a process defect for the next round rather than re-litigated here. **Supersedes two cells of the preceding row.** (i) Its compliance figures read `head_lines=661`, drafted before a later edit in the same round took the file to 660; the correct values are base=origin/main `base_lines=662 head_lines=660 allowed_lines=662` and base=origin/dev `base_lines=668 head_lines=660 allowed_lines=668`, both `reference_line_budget_blocking_ok`. (ii) Its rule cell describes the face set as read off CI's base-resolution expression; the manifest form in this row is the one that governs. The correction is recorded here rather than by editing that row: the ledger is append-only and the gate enforces it mechanically - a row edited after its own round committed no longer survives at HEAD, its round loses its only row, and the gate fails closed. That is the append-only contract catching an in-place fix, which is what supersede-by-pointer exists for. |
|
|
@@ -56,7 +56,7 @@ If the validator reports `missing_required_command`, keep the failure visible an
|
|
|
56
56
|
|
|
57
57
|
## Behavioral Validation
|
|
58
58
|
|
|
59
|
-
- For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology before
|
|
59
|
+
- For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology — **and the firing point is BEFORE drafting the body, not only before finalizing**. Eval-first authoring for a NEW skill (or a new hard-rule section): (1) write the evaluation scenarios first — **at least three** for a new skill (both the vendor's published authoring guide and the high-star practice pack converge on three-plus scenarios before body text; a single-rule edit may scope down to that rule's own scenario); (2) run them WITHOUT the skill and record the observed failures verbatim — and for a discipline-slip failure (the agent knows the rule and skips it under pressure), capture the agent's rationalizations word-for-word: each verbatim excuse is the raw material for one rationalization-vs-reality row and one red-flag line in the skill text (the discipline-slip form in `rule-consolidation.md`'s form-by-failure table); an invented hypothetical excuse does not qualify — counter only what a run actually said, and don't add rows for excuses no run produced; **a no-skill control that does not exhibit the failure is a stop signal — do not author guidance for a failure you cannot observe** (record the null finding instead; this is the pre-draft face of "Evidence must come before new rules"); (3) draft the **minimal** content that addresses the observed failures, then re-run the same scenarios WITH the skill; (4) when a later run, review round, or live miss surfaces a NEW rationalization for an existing discipline gate, add its explicit counter row to that gate's table and re-run the tempting scenario — counter tables accrete from observed excuses across rounds, never from imagination. The code-level RED-GREEN-REFACTOR method (write the failing case first, watch a fresh agent violate the rule WITHOUT the skill, then add the skill and watch it comply) is owned by `superpowers:writing-skills` + `superpowers:test-driven-development` — **if installed, route there; otherwise apply the RED-baseline rule inline** (manually record the without-change failure and the with-change compliance). This is the skill-authoring face of **eval-driven development** (for a behavior/routing change, run the scenario before you finalize; never special-case the scenario just to make it pass) — borrow the *principle*, not a claim of production-grade eval rigor.
|
|
60
60
|
- **What makes a `RED-baseline` valid is executed-and-recorded vs narrated — not recorded vs live.** Any evidence form (before-after diff, golden trace, or pressure scenario) is valid when it actually records the without-change failure AND the with-change compliance with a locator + expected-vs-actual (see `dual-track-review-gate.md`). Prose that merely *describes* an expected failure without running it is not a baseline; a pressure scenario you actually executed and recorded is.
|
|
61
61
|
- **A claimed impossibility is falsified in-env before it lands (authoring and reviewing alike; relocated from `SKILL.md`).** Deferring work behind a conservative-sounding stub ("not implemented", "needs a future tested wrapper") is the avoidance form of a blocked-verification claim: it *feels* safe but ships an unverified impossibility as durable behavior — distinct from a real `unavailable`/`pending`-with-remediation+residual-risk record, which is what remains after a safe attempt was genuinely impossible (the attempt itself stays within the safety boundaries the `SKILL.md` rule names). Failure shape: a fallback reviewer lane shipped as a permanent fail-closed stub on an unverified "permission probe not implemented" premise that a ~60-second live tool run refuted, re-enabling the lane.
|
|
62
62
|
- **A headless code-writing RED/GREEN needs a *fair* violation-tempting scenario + an *independently-valid* objective measure — a capable agent complies on a clean task.** When the rule governs how an agent WRITES code, a clear well-specified prompt usually makes even the no-rule baseline produce compliant code (zero delta → no RED), so a clean-task baseline proves nothing. Surface the RED with **realistic inherited pressure** — a real legacy/house convention the rule must override, or a genuinely ambiguous spec — **NOT an explicit instruction to emit the anti-pattern**: leading the baseline directly into the violation launders a constructed failure into "RED" and is fabrication (the same defect as the rule above), so record why the prompt is fair and mark any direct-leading scenario synthetic/advisory, not RED. Score both runs with an **objective measure that detects the ANTI-PATTERN** — the rule's own checker counts ONLY if it was independently validated first (held-out positive/negative fixtures + a documented residual boundary, so RED genuinely fails); a same-change checker that merely whitelists the expected GREEN form makes GREEN trivially true. Run the agent **cross-model / fresh-context** so the baseline isn't primed by your session. If even the fair tempting baseline complies, that is an honest finding — the rule's marginal value is in edge/legacy cases, not the common one — record it, don't manufacture a RED.
|
|
@@ -1359,6 +1359,36 @@ for required_phrase in \
|
|
|
1359
1359
|
done
|
|
1360
1360
|
echo "test_case_first_gate_ok"
|
|
1361
1361
|
|
|
1362
|
+
# Contract-anchor gate: declarative wording-existence pins for load-bearing
|
|
1363
|
+
# prose contracts that structural checks cannot see (verdict-taxonomy
|
|
1364
|
+
# discriminators, stop-condition predicates, externally verified numeric
|
|
1365
|
+
# tiers). Checker and its contract-anchors.tsv table resolve NEXT TO THIS
|
|
1366
|
+
# VALIDATOR (one-versioned-unit rule, same as the sync-pointer gate — an
|
|
1367
|
+
# in-tree copy could be doctored to exit 0). Red = a pinned contract sentence
|
|
1368
|
+
# drifted, was deleted, or became ambiguous; the remedy is printed by the
|
|
1369
|
+
# checker (restore the wording, or update the anchor row in the same MR for an
|
|
1370
|
+
# intentional contract change). Self-proof: test_check_contract_anchors.sh
|
|
1371
|
+
# (fast) and test_pinned_phrase_mutation_walk.sh (heavy).
|
|
1372
|
+
contract_anchor_script="$checker_scripts_dir/check-contract-anchors.sh"
|
|
1373
|
+
if [[ -L "$contract_anchor_script" || ! -f "$contract_anchor_script" ]]; then
|
|
1374
|
+
echo "contract_anchor_infra_failed: check-contract-anchors.sh missing or not a regular file beside the validator: $contract_anchor_script" >&2
|
|
1375
|
+
exit 2
|
|
1376
|
+
fi
|
|
1377
|
+
contract_anchor_rc=0
|
|
1378
|
+
contract_anchor_out="$(bash "$contract_anchor_script" "$root")" || contract_anchor_rc=$?
|
|
1379
|
+
[ -n "$contract_anchor_out" ] && printf '%s\n' "$contract_anchor_out"
|
|
1380
|
+
if [ "$contract_anchor_rc" -eq 1 ]; then
|
|
1381
|
+
echo "contract_anchor_gate_blocking_failed: pinned contract wording drifted (diagnostics above)" >&2
|
|
1382
|
+
exit 1
|
|
1383
|
+
elif [ "$contract_anchor_rc" -ne 0 ]; then
|
|
1384
|
+
echo "contract_anchor_infra_failed: rc=$contract_anchor_rc (contract-anchor gate could not run — fail-closed)" >&2
|
|
1385
|
+
exit 2
|
|
1386
|
+
fi
|
|
1387
|
+
if ! printf '%s\n' "$contract_anchor_out" | grep -qE '^contract_anchor_gate_ok \([0-9]+ anchors\)$'; then
|
|
1388
|
+
echo "contract_anchor_infra_failed: green output grammar missing (expected: contract_anchor_gate_ok (N anchors))" >&2
|
|
1389
|
+
exit 2
|
|
1390
|
+
fi
|
|
1391
|
+
|
|
1362
1392
|
# Post-cleanup register check: after a project-specific extraction batch is
|
|
1363
1393
|
# migrated out of the shared skill tree, the source-register.md template must
|
|
1364
1394
|
# not silently grow new project-specific dated entries. The rule in
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Contract-anchor gate: declarative wording-existence pinning for load-bearing
|
|
3
|
+
# prose contracts that no structural check would otherwise protect.
|
|
4
|
+
#
|
|
5
|
+
# Problem class (observed, 074 RED-baseline probes): a skill reference's
|
|
6
|
+
# load-bearing contract sentence — a verdict-taxonomy discriminator, a
|
|
7
|
+
# stop-condition predicate, an externally verified numeric tier — can be
|
|
8
|
+
# deleted, semantically inverted, or numerically falsified while every
|
|
9
|
+
# structural gate stays green, because structural gates check shape and
|
|
10
|
+
# references, never the survival of specific contract wording.
|
|
11
|
+
#
|
|
12
|
+
# Mechanism: a sidecar TSV table (contract-anchors.tsv, same directory) lists
|
|
13
|
+
# one anchor per row: id <TAB> repo-relative path <TAB> pinned literal <TAB> note
|
|
14
|
+
# - the literal must occur EXACTLY ONCE in the named file (fixed-string match).
|
|
15
|
+
# 0 occurrences => contract_anchor_missing (drift/deletion) exit 1
|
|
16
|
+
# >1 occurrences => contract_anchor_duplicate (decoy/ambiguity;
|
|
17
|
+
# an anchor that matches twice can no longer prove which copy is the
|
|
18
|
+
# contract — same rule as the sync-pointer registry) exit 1
|
|
19
|
+
# - a missing anchored file is a broken contract, not infrastructure:
|
|
20
|
+
# contract_anchor_file_missing exit 1
|
|
21
|
+
# Table integrity is fail-closed (exit 2, infra): unreadable/empty table
|
|
22
|
+
# (an empty list scanning nothing must never certify), malformed row,
|
|
23
|
+
# duplicate id, or a literal under 16 characters (too weak to be unique —
|
|
24
|
+
# same floor as firing-path anchors).
|
|
25
|
+
#
|
|
26
|
+
# All rows are scanned before any exit: anchor-drift findings (rc 1) and
|
|
27
|
+
# row-integrity findings (rc 2) are each collected across the whole table, so
|
|
28
|
+
# one run reports every problem; when both classes are present the run exits 2
|
|
29
|
+
# (a broken table means no verdict can be trusted). Green output grammar is
|
|
30
|
+
# pinned for the caller:
|
|
31
|
+
# contract_anchor_gate_ok (N anchors)
|
|
32
|
+
# Exemptions: none, and deliberately no environment override — an anchor is
|
|
33
|
+
# removed or reworded only by editing the table in the same MR that changes
|
|
34
|
+
# the pinned wording (reject-and-instruct failure message points there).
|
|
35
|
+
#
|
|
36
|
+
# Intentional-change recipe (printed on failure): edit the contract sentence
|
|
37
|
+
# AND its table row together; the diff then shows the contract change
|
|
38
|
+
# explicitly instead of a silent drift.
|
|
39
|
+
set -u
|
|
40
|
+
|
|
41
|
+
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)
|
|
42
|
+
root="${1:-.}"
|
|
43
|
+
table="$script_dir/contract-anchors.tsv"
|
|
44
|
+
if [[ "${2:-}" == "--table" && -n "${3:-}" ]]; then
|
|
45
|
+
table="$3"
|
|
46
|
+
fi
|
|
47
|
+
|
|
48
|
+
if [[ ! -d "$root" ]]; then
|
|
49
|
+
echo "contract_anchor_root_missing: $root" >&2
|
|
50
|
+
exit 2
|
|
51
|
+
fi
|
|
52
|
+
if [[ ! -f "$table" ]]; then
|
|
53
|
+
echo "contract_anchor_table_missing: $table" >&2
|
|
54
|
+
exit 2
|
|
55
|
+
fi
|
|
56
|
+
|
|
57
|
+
anchor_count=0
|
|
58
|
+
fail_count=0
|
|
59
|
+
infra_count=0
|
|
60
|
+
# Bash 3.2-safe id-uniqueness set (stock macOS bash has no associative arrays):
|
|
61
|
+
# newline-delimited membership string, ids are TSV fields so they carry no tabs
|
|
62
|
+
# and no newlines.
|
|
63
|
+
seen_ids=$'\n'
|
|
64
|
+
line_no=0
|
|
65
|
+
while IFS= read -r raw || [[ -n "$raw" ]]; do
|
|
66
|
+
line_no=$((line_no + 1))
|
|
67
|
+
[[ -z "$raw" || "$raw" == \#* ]] && continue
|
|
68
|
+
IFS=$'\t' read -r id path literal note <<<"$raw"
|
|
69
|
+
if [[ -z "${id:-}" || -z "${path:-}" || -z "${literal:-}" || -z "${note:-}" ]]; then
|
|
70
|
+
echo "contract_anchor_row_malformed: line $line_no of $table (need id<TAB>path<TAB>literal<TAB>note)" >&2
|
|
71
|
+
infra_count=$((infra_count + 1))
|
|
72
|
+
continue
|
|
73
|
+
fi
|
|
74
|
+
if [[ "$note" == *$'\t'* ]]; then
|
|
75
|
+
echo "contract_anchor_row_malformed: line $line_no of $table (extra tab-separated field)" >&2
|
|
76
|
+
infra_count=$((infra_count + 1))
|
|
77
|
+
continue
|
|
78
|
+
fi
|
|
79
|
+
case "$seen_ids" in
|
|
80
|
+
*$'\n'"$id"$'\n'*)
|
|
81
|
+
echo "contract_anchor_duplicate_id: $id (line $line_no repeats an earlier row's id)" >&2
|
|
82
|
+
infra_count=$((infra_count + 1))
|
|
83
|
+
continue
|
|
84
|
+
;;
|
|
85
|
+
esac
|
|
86
|
+
seen_ids="$seen_ids$id"$'\n'
|
|
87
|
+
if (( ${#literal} < 16 )); then
|
|
88
|
+
echo "contract_anchor_literal_too_short: $id (${#literal} chars, need >=16)" >&2
|
|
89
|
+
infra_count=$((infra_count + 1))
|
|
90
|
+
continue
|
|
91
|
+
fi
|
|
92
|
+
anchor_count=$((anchor_count + 1))
|
|
93
|
+
target="$root/$path"
|
|
94
|
+
fix_line=" fix: restore the contract wording, or — for an intentional contract change — update this anchor row in $(basename "$table") in the same MR"
|
|
95
|
+
if [[ ! -f "$target" ]]; then
|
|
96
|
+
echo "contract_anchor_file_missing: $id $path" >&2
|
|
97
|
+
echo " fix: restore the anchored file, or — for an intentional move/retirement — update or remove this anchor row in $(basename "$table") in the same MR" >&2
|
|
98
|
+
fail_count=$((fail_count + 1))
|
|
99
|
+
continue
|
|
100
|
+
fi
|
|
101
|
+
occurrences=$(grep -oF -- "$literal" "$target" | wc -l | tr -d '[:space:]')
|
|
102
|
+
if [[ "$occurrences" -eq 0 ]]; then
|
|
103
|
+
echo "contract_anchor_missing: $id in $path" >&2
|
|
104
|
+
echo " pinned literal: $literal" >&2
|
|
105
|
+
echo "$fix_line" >&2
|
|
106
|
+
fail_count=$((fail_count + 1))
|
|
107
|
+
elif [[ "$occurrences" -gt 1 ]]; then
|
|
108
|
+
echo "contract_anchor_duplicate: $id in $path ($occurrences occurrences; the anchor can no longer prove which copy is the contract)" >&2
|
|
109
|
+
echo " fix: keep the pinned wording in exactly one place (dedupe the copy), or repin the anchor row in $(basename "$table") to a longer unique literal in the same MR" >&2
|
|
110
|
+
fail_count=$((fail_count + 1))
|
|
111
|
+
fi
|
|
112
|
+
done < "$table"
|
|
113
|
+
|
|
114
|
+
if [[ "$infra_count" -gt 0 ]]; then
|
|
115
|
+
echo "contract_anchor_table_invalid: $infra_count integrity error(s) in $table (no verdict from a broken table — fail-closed)" >&2
|
|
116
|
+
exit 2
|
|
117
|
+
fi
|
|
118
|
+
if [[ "$anchor_count" -eq 0 ]]; then
|
|
119
|
+
echo "contract_anchor_table_empty: $table has no data rows (an empty anchor set must never certify)" >&2
|
|
120
|
+
exit 2
|
|
121
|
+
fi
|
|
122
|
+
if [[ "$fail_count" -gt 0 ]]; then
|
|
123
|
+
echo "contract_anchor_gate_failed: $fail_count of $anchor_count anchors" >&2
|
|
124
|
+
exit 1
|
|
125
|
+
fi
|
|
126
|
+
echo "contract_anchor_gate_ok ($anchor_count anchors)"
|