@ccoalm/ccl-skills 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +4 -3
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +23 -0
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +178 -4
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +127 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +2 -1
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +1 -1
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +1 -0
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +11 -1
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +16 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/source-map.md +1 -0
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +8 -8
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/dispatch-owner-skills.md +9 -1
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/problem-resolution-and-learning.md +2 -0
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +1 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/tag-and-prod-pipeline-gate.md +9 -0
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +17 -20
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +9 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +26 -44
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +16 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +3 -1
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +59 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +2 -2
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +3 -1
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +35 -0
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +16 -0
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +35 -6
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +476 -0
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +81 -1
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +28 -0
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
  44. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
  45. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
  46. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
  47. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
  48. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
  49. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +336 -0
  50. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
  51. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_pointer_integrity.sh +41 -1
  52. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +141 -28
  53. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +139 -38
  54. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/SKILL.md +1 -1
  55. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +9 -8
  56. package/dist/assets/release.json +110 -45
  57. package/package.json +1 -1
@@ -459,5 +459,64 @@ and inverted the sense (production, not product), and the coordinator now shares
459
459
  | A compression that drops a scope qualifier widens the rule it hosts: the external-pack routing clause routes only a MISSING method/tool-layer capability, and restoring the dropped word is the fix, not rewording around it | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#M needs the functional-equivalent check | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: a size-budget compression rewrote the reference-only routing clause from routing a missing method/tool-layer capability to routing that layer categorically, which would displace locally covered P-verdict capabilities and contradict the functional-equivalent check landed in the same round; caught by the supplementary post-delta challenge round enumerating compressed sentences (same class as the metered model/tool qualifier drop fixed in the sibling owner). Fix: the missing-capability qualifier restored; byte offsets from this branch's own pointer sentence, whose semantics live in the owning reference. RED baseline (replayed): word-level diff against the base revision showed the dropped qualifier before the fix and shows it restored after; the class sweep over all three owner entrypoints found no further load-bearing drops. |
460
460
  | Stop reporting keeps its specificity qualifiers: the entrypoint demands the concrete stop reason and the exact evidence checked, and budget offsets come from relocating clauses whose semantics already live verbatim in the owning reference | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#state the concrete stop reason and the exact evidence checked | `updated` | Owner key `product-rd-workflow/SKILL.md`. Observed failure: a size-budget compression dropped the-concrete/the-exact from the stop-reporting sentence, licensing generic stop reports; the challenger graded it load-bearing, the implementer's word-sweep had graded it neutral, and the maintainer's standing delegation resolves such token disputes by the repository's fail-closed obligation standard, so the qualifiers are restored. Byte offset: the stale-source parenthetical is removed from the entrypoint because its full sentence lives verbatim in references/pre-final-continuation-gate.md (Status-source reconciliation) which the same sentence already cites — relocation, not compression. RED baseline (replayed): grep for the restored phrase zero-hit on the pre-fix entrypoint, exactly one hit after; the removed parenthetical greps once in the owning reference. |
461
461
  | The dual-track reviewer's verification scope is a documented boundary: content semantics belong to the reviewer, deterministic-gate claims to CI, historical-process claims are testimony unless receipt-bound — ruled on once so packet-verifiability findings stop recurring per round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#a finding that only restates this boundary is dispositioned against this rule, never re-litigated per round | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; the change lands in references/dual-track-review-gate.md (new Reviewer verification scope section). Observed failure: the packet-verifiability finding class recurred across four supplementary review rounds and roughly a dozen occurrences in this round's chains — every reviewer independently rediscovered that the packet cannot carry the deterministic oracles, and every round paid the same finding again because the boundary was undocumented. The maintainer confirmed the operating reality (all consumers and reviewers are agents; the human role is authority, not readership), so the boundary is now standing text agents can disposition against, with the receipt-embedding backlog item named in place. RED baseline (replayed): grep for the boundary phrase zero-hit before this change, exactly one hit after, on an added normative list line. |
462
+ | The Agent-autonomous review budget is summed across review chains, never per chain: in a self-hosted skill repository a finding fix that touches a selected-owner tree breaks the tracked chain by design (nearly every fix in such a round; a fix confined to files outside every selected owner drifts only the candidate hash and continues in-chain), so below the cap the break is recovered by a ledger-counted restart (batched dispositions first, full-context first packet, plan frozen with the candidate, final chain still able to hold the review-plus-challenge ready floor), and at the cap the designed terminal is disposition plus the honest terminal record — `continuation_authorization_required` when the final round itself returned findings, otherwise an interim record naming the last reviewed candidate and all later deltas — while restarting chains until a clean pass, or counting a restart as fresh budget, is the named contract violation | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#Sum spent rounds across all chains before opening one more | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; changes land in references/dual-track-review-gate.md — new self-hosted-chain rule with a four-step walked enumeration, the continuation bullet's dead-end sentence scoped to the at-cap case, an anti-pattern merge — and in references/extraction-quickstart.md 3e, whose two rerun clauses gain the cross-chain budget qualifier). Observed failure (production receipts, per-host archive, mechanism at the same commit as the current base): one round ran 20+ reviewer rounds, the next 12 restarted chains / 21 reviewer invocations to land a three-line diff, none human-authorized; receipts show every chain restarted at index 1 with a new candidate hash after fixes, while a control chain on an unchanged candidate ran review plus two challenges without invalidation. Root cause is a corpus-level contradiction, confirmed by frozen-criteria elicitation runs (n=3 per arm, isolated cwd, arms byte-identical to versioned text): the pre-change gate section alone elicits the STRICT reading three of three — zero autonomous restarts, interim-then-human after any break — while the quickstart page simultaneously mandated "re-run both on the updated candidate before landing" and the wrapper mechanically accepts fresh chains; jointly unsatisfiable, resolved in production by improvised unbounded restarts. RED-baseline: the occurred production failure plus the three-of-three strict/mandated contradiction on the pre-change corpus; post-change runs elicit the landed semantics three of three (cross-chain summing to the three-round cap, ledger-counted restart below it, terminal disposition at it, no laundering) with no contradiction-rationalizing text. Semantics delta declared honestly: below-cap restarts move from ask-human (strict reading) to Agent-autonomous ledger-counted rounds — a deliberate loosening grounded in the review-efficiency adjudication, the wrapper's three-round design, and two informed merges of rounds that ended in the terminal-disposition shape; the maintainer can revert to the strict reading by decision. Supersedes by pointer the recovery clause of the earlier four-process-controls row ("recovered by an interim checkpoint … plus a human continuation authorization"): that recovery is now the at-cap path only, and the row stays unedited per the append-only contract. Oracle: delete this row on the candidate and run the repo check with the base ref set — it prints the impact-chain missing token. |
463
+ | "Candidate edits do not reset Agent authority" promises no continuation: the selected-owner digest hashes each selected owner package's current working tree and owners derive from candidate paths, so a candidate edit inside any selected owner package invalidates every prior receipt and the next tracked round fails `review_chain_invalid` — for a self-hosted skill-repo candidate that is every applied fix, and a restarted chain re-enters the same cumulative Agent budget | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: file:skills/code-review/references/staged-review-contract.md#invalidates every prior receipt | `updated` | Owner key `code-review/SKILL.md` (entrypoint unchanged; the change adds two consequence bullets to the Agent review chain section of references/staged-review-contract.md). Observed failure: the section's tolerance sentence ("older candidate hashes; they remain consumed") reads as continuation-after-fix, while the stable-binding predicate in review_gate.py (this owner's script) compares controller digest, owner-selection source, owner names, and selected-owner digest on every prior receipt, with the digest computed over the owner packages' live working tree — so the documented tolerance is unreachable exactly when the candidate lives inside its owner package; archived receipts from one extraction round show 12 chains each restarted at index 1 with a new candidate hash after fixes, and a control chain on an unchanged candidate continuing three rounds. Declaration-contradicted-by-implementation class: the correction documents the implemented refusal instead of changing it — binding semantics, trust model, wrapper, and validator behavior are untouched. RED-baseline: grep for the consequence phrase is zero-hit before this change and exactly one hit after, on an added normative list line; the mechanism is reproducible from the chain-validation predicates plus the archived receipts. |
464
+ | The extraction lane's autonomous budget is one review plus one challenge, pinned in exactly two executable places and derived everywhere else: the wrapper passes the single value, the closeout validator computes every numeric bound from two module constants, and prose surfaces name the wrapper-fixed budget instead of repeating numerals — under this budget a fix-restart is never fundable, so the hold-fixes branch is the standing path, the self-hosted chain-break conflict becomes unreachable without touching the binding trust model, and a two-receipt final-round-findings chain validates as `continuation_authorization_required`, closing the cross-chain machine-terminal gap | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (dual-track bullet numerals synced; frontmatter untouched; severe-entrypoint byte budget held at net zero by equivalent shortenings in the same bullet). The budget-size choice was maintainer-delegated in-session after two cost escalations (a review loop can burn two days on one issue) and restores the earlier one-review-plus-at-most-one-challenge adjudication for this repository; the challenge remains mandatory for every non-wording shared-skill change. Changes: extraction_review_gate.sh passes the fixed budget 1 and its guard message matches; validate_extraction_review_state.py derives round bounds, remaining counts, per-round state legitimacy, and the continuation predicate from WRAPPER_CHALLENGE_BUDGET/WRAPPER_AUTONOMOUS_ROUNDS (three hidden hardcodes found and converted during the green run: completion remaining, review_state round semantics, the continuation receipt count); both regression suites' fixtures converted from three-round to two-round shapes with occurrence semantics preserved (multi-finding rounds keep sweep-triggering occurrence counts). RED-baseline (applied): reverting the wrapper to the old budget in a throwaway edit turns test_extraction_review_gate.sh red at the budget-argument assertion and the restore is green; the validator suite was red at each hidden hardcode until converted, then fully green; catalog byte gate and implementation-gates suite green on the final candidate. Supersedes by pointer the wording of the two rows above where they cite a three-round cap or an unfundable-restart arithmetic tied to it: their production evidence and anchors are unaffected, and the budget-agnostic four-step enumeration they land is unchanged — only the numeral moved. Elicitation runs were re-taken against the final section with a corrected prompt (the scenario's own budget parenthetical had contradicted the attached rule text; the stale batch is archived unscored). |
465
+ | The closeout validator must reject a controller chain longer than the wrapper can mint: without an upper bound, a caller-supplied third receipt under budget one claims a negative remaining count and reaches completion validation as a ready-state budget bypass; and every prose surface advertising the retired budget is a laundering affordance, so the stale phrases are closed and pinned closed by documentation assertions | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (unchanged this batch; changes land in validate_extraction_review_state.py — the receipts loop now fails `controller chain exceeds the wrapper budget` past WRAPPER_AUTONOMOUS_ROUNDS — its regression suite, the gate suite's documentation assertions, dual-track-review-gate.md, extraction-quickstart.md, and the register rows above). Provenance: the post-budget batch of this candidate's own two-round review chain — the final challenge (receipt archived per-host) surfaced the over-budget bypass and the stale third-round/two-challenge phrases; the review round surfaced the stale phrases independently, the uncertified-post-batch gap (closed by the pending-branch certification sentence in the terminal-disposition step), and a residual absolute in the consequence row above — superseded by pointer here, not edited in place per the ledger's append-only contract: read its "every applied fix" as "nearly every applied fix; a fix confined to files outside every selected owner drifts only the candidate hash and continues in-chain", matching the normative text it records. RED-baseline (applied, red for the right reason): the new over-budget fixture was added BEFORE the validator fix and the suite went red showing the unfixed validator accept the three-receipt budget-one ledger as ready_for_human_decision; after the one-line bound the case rejects with the named token and the full suite is green. Packet-verifiability findings about unshippable suite output remain dispositioned against the documented reviewer-verification-scope boundary with rerun oracles in the MR/PR body. |
466
+ | The assent-triggered blocked outcome states its interim classification with an explicit verb, not an elliptical fragment | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#form and classifies the turn `interim` | `updated` | Owner key `product-rd-workflow/SKILL.md`. Debt-repayment chain for the disclosed unreviewed pinned-phrase restoration commit 4e65dbc (patch-identical pre-rebase form 4b30746): replayed against its immutable base b0fe08b, `git show b0fe08b:skills/product-rd-workflow/SKILL.md` carries "form and classifies the turn `interim`." and the 4e65dbc diff drops "and classifies the", leaving the malformed "form, turn `interim`" — the independent review of this repayment chain confirmed the verb loss as the only defect of that diff still unrepaired at review time (the fused "reconfirm A" sentence boundary was repaired by 5fc25fb and the ledger-table break by the note relocation in 26b773e, both verified against the current file state); head restores the classification verb from the b0fe08b text of the same sentence, whose later specificity/carrier revisions elsewhere in the sentence are untouched. |
462
467
 
463
468
  Supersede note (this round, before landing): the two crypto-erase rows above ("go-microservice-architecture" and "python-service-architecture") describe an earlier candidate state; the landed text attributes the CE do-not-use conditions to SP 800-88 Rev.1 §2.6 with Rev.2 (2025) superseding and continuing the framework — per the source-verification ledger row "CE conditions text location". The rows' RED-baseline probes and firing-path anchors are unaffected.
469
+
470
+ Round 073-receipt-bundling rows (new table so the entry renders as a table row after the supersede-note paragraph above):
471
+
472
+ | Upstream rule | Downstream owner | Expected executable behavior | Status (updated, unchanged, routed, or not-applicable) | Evidence |
473
+ | --- | --- | --- | --- | --- |
474
+ | Deterministic-gate process claims ride the packet as candidate-SHA-bound receipts: `gate_receipt.py` mints against the committed candidate (clean tree, HEAD, argv, exit code, output hash, bounded tail) and verifies differentially, so a pre-fix RED stops being unverifiable testimony | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#clean tree required, HEAD commit recorded | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (Reviewer verification scope — the "Standing backlog" clause is superseded in place by the landed channel) + scripts/gate_receipt.py + scripts/test_gate_receipt.sh (registered in the fast lane). Observed failure: the packet-verifiability class — historical-process claims graded implementer testimony across the prior round's review chains, terminally dispositioned accepted-limitation repeatedly, with the backlog row naming receipt embedding as the mechanical fix. Remedy re-derived against current code (a deferred registration is hypothesis, not spec): the packet already binds `--review-plan-file` evidence via review_context_sha256 and the v3 ledger already binds sibling files by hash, so no controller or validator change — the missing pieces were the receipt artifact class and its mint/verify tool, which is what lands. Trust model stated in the landed text so it cannot be oversold: candidate-bound falsifiable consistency evidence, not runner authentication; CI re-running gates stays the deterministic authority; unreceipted process claims remain testimony. RED baseline (applied mutations, differential): tampered output hash → rc1 output_hash_mismatch; tampered recorded exit code → rc1 exit_code_mismatch; foreign key → rc1 exact-key-set; wrong checked-out candidate → rc2 named no-verdict (not a false red); nondeterministic-output gate → full rerun rc1, --exit-only rc0 with scope token; control legs green. Same-round sibling disposition, no diff landed: the two-place numeric-contract backlog item (guarded-file whitelist + version-bump receipt + parse-values-from-prose) was re-probed against the current baseline and closed already-covered — the pinned/sync declared-pair registry owns the whitelist half and test_routing_pointer_integrity.sh's doc-vs-executor threshold parity check owns the parse half; that incumbent fired live twice this round against a draft duplicate (executor marker loss; a pinned firing-path phrase reworded), so the duplicate was deleted per same-class convergence-by-deletion and the final candidate leaves SKILL.md, description-authoring.md, and the checker cap logic byte-identical to base. |
475
+ | Load-bearing prose contracts are pinned declaratively: each contract-anchors.tsv row demands its pinned literal exactly once in the owner file, so deletion, semantic inversion, numeric falsification, or decoy duplication of a verdict-taxonomy discriminator, a stop-condition predicate, or an externally verified numeric tier turns the repo check red instead of passing clean | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/check-contract-anchors.sh + contract-anchors.tsv (9 anchors) + the check-ccl-skills.sh delegation block (beside-the-validator resolution, pinned green grammar, fail-closed rc mapping) + the fast-lane suite). Observed failure: three enforcement-gap findings from the prior round's review chains, deferred needs_human_decision and scheduled into this round by the maintainer — reproduced against the round base before any implementation: deleting the fault-origin discriminator sentence, inverting the materially-differing/evidenced-cause stop predicates, and falsifying externally verified values (14.4→12.4, 97→87) each left the full check clean_ok while a control mutation (broken reference link) went red, proving the instrument could fail. RED-baseline (applied, differential): the same probe mutations now red the gate with per-anchor attribution; suite mutants M1–M8 red for their named reason (missing/duplicate/file-missing/empty-table/malformed-row/short-literal/duplicate-id/table-missing), benign neighbors B1–B3 green, D1 names only the broken anchor, and the whole suite goes red under an always-green checker stub. Scope stated honestly: anchors make contract-wording and pinned-value changes conscious (same-MR table edit), not externally re-verified — external-truth re-verification stays with the documented reviewer-verification-scope boundary, and stop-predicate semantics testability is routed to the evals layer (deferred, bound to that round's entry). Registered-remedy narrowing recorded: the N×M forbidden-token matrix is not landed (three mechanized anti-patterns, the 1×1 pending/clean exclusion, and the sync registry already own every observed shape; recurrence re-opens it) and numeric copy-prohibition narrowed to both-sides anchors (entry + reference pinned together) because repo values are already single-owner. |
476
+ | A pinned-phrase gate family must be provably able to go red for the right reason: one applied deletion mutation per family runs the FULL shipped checker against a committed fixture clone and demands that family's own red token, with the unmutated control run green on every family token; and every anti-patterns panel section must carry its Grep recipe | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in the heavy-lane walk + the fast-lane panel structural check test_antipattern_grep_panel.sh). Observed failure: none of the ~40 required_phrase pins across the four inline gate families had ever been proven able to red — the always-green-oracle class this repository already met as the degenerate-fixture worktree-pruning precedent and the empty-glob false-green lesson (an emptied phrase list or quoting regression would certify silently). RED-baseline (applied, differential): walk legs W1–W5 each delete one currently-pinned phrase in a committed fixture clone and the full checker reds with exactly that family's token (project-assessment, task-retro/teammate-trigger, test-case-first, product-rd anchor, and the new contract-anchor delegation), ~70s total; the control leg is green with all five family green tokens and a loop-count floor guards list vacuity; the panel check reds on a stripped Grep line naming the right section and stays green for a benign non-anti-pattern section. Environment pitfall recorded for reuse: CCL_SKILL_BASE_REF must stay unset around nested gate runs — it leaks into child validators' synthetic self-test repos where HEAD always resolves and flips their no-base→degraded legs into false passes. Rule side already-covered: the killing-mutation walk, benign-near-miss precision rows, and oracle-validation duties are owned by testing-strategy and the dual-track Self-audit section — no prose added anywhere; the firing path for this failure class is now these suites in the fast/heavy lanes. Grep pattern-compile validation deliberately discarded (panel recipes mix GNU-BRE commands with prose instructions by design; a compile check would false-red on regex-dialect differences without protecting a real contract). |
477
+
478
+ | Eval-first authoring fires pre-draft: a NEW skill needs three-plus scenarios and a no-skill baseline before body text, and a non-failing control stops the draft | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#must define eval/pressure scenarios, baselines, acceptance criteria pre-draft | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. 075/R3 round (S9). Both external sources re-verified primary-source on 2026-08-31: Anthropic "Skill authoring best practices" (https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices, section "Evaluation and iteration" / "Build evaluations first") — create evaluations BEFORE writing, build three scenarios, establish the baseline without the Skill, write minimal instructions, and no built-in runner exists; superpowers plugin 6.3.0, skills/writing-skills (SKILL.md sections "The Iron Law" and "Micro-Test Wording Before Full Scenarios"; testing-skills-with-subagents.md section "RED Phase") — a failing test first for new skills AND edits, 3+ pressure scenarios, and a no-guidance control that does not exhibit the failure means stop, do not author. The commit-time RED-baseline contract is untouched in this diff (the semantic-control leg); the change moves the firing point onto the pre-draft transition per the firing-point-placement corollary, with mechanics merged into the existing canonical bullet in validation-and-landing.md rather than appended as a new rule. Zero-loss map for the rewritten Workflow step-2 sentence: the subjective/high-impact category list survives (frontend/client shortened to client, same referent), pressure scenarios widen to eval/pressure scenarios, acceptance criteria survive verbatim, and before-editing tightens to pre-draft; nothing dropped, NEW-skill coverage and baselines are the additions. |
479
+
480
+ | Deterministic anchors pin stop-predicate wording while paired body-compliance probes grade classification on the gate's own continuing:/blocked: markers | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#probe subset on this machine before landing | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Closes the 071-chainB-r1f1 semantic-testability follow-up deferred by 074 to R3. Registered form (invert predicate, gate stays clean) no longer reproduces: the entrypoint anchor gate, the register firing-path anchor, and the size ratchet each redded a separate applied polarity/reword probe in a disposable checkout at the round base, with a pristine control leg green first. Evolved form reproduced and is the RED leg: a word-compensated additive neutralization (appending an advisory-continue sentence while deleting equal unpinned words) exits 0 with ccl_skill_check_clean_ok. With-change leg: four paired prd-* probes graded 4/4 on the pristine body after instrument fixes (marker-decoration tolerance with a mention-vs-verdict grammar; shared-scaffolding single-variable isolation), and the probe set demonstrably can fail (first run graded 2/4 on real output, one miss being a genuine same-case-two-classifications observation); run reports committed under eval/evidence/stop-predicate-probes-2026-08-31/. Honest boundary recorded in the f4 layering section (eval-routing.md points there): the same applied neutralization mutants did not flip live agent behavior either (two mutants by two probes, small N) — probes carry behavior drift, not buried-sentence tripwires. |
481
+
482
+ | Negative controls and coverage-gap probes are first-class routing-bank rows (expected "none" sentinel, acceptable alternates, bait neighbors), with absorbed / ownership_split as labeled outcomes and clarify / low-confidence / replica-agreement as first-class report metrics | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#已修复的路由 miss 必须把其 utterance 冻结成 bank task; bank-evidence: command:skills/skill-extraction-workflow/scripts/eval-routing-bank.rb | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Live at the round base: grading the new gap probes with two replicas showed one replica absorbing each probe into a coordinator skill while the other rejected it — the runner's absorbed + ownership_split labels fired on real grader output before the expectations were corrected with acceptable[]; integrity-lane mutants (sentinel in must_not, acceptable restating expected, unknown acceptable target, acceptable∩must_not, empty-string fields) each turned test_routing_bank_integrity.sh red on its own named assertion with the unmutated control green, and the validator self-proof section now replays those mutants on every run. Durable artifacts (exact invocation, grader model, candidate fingerprints, raw per-replica verdicts, red/green transcript, gate exits): eval/evidence/routing-negative-controls-2026-08-31/. |
483
+
484
+ | A before/after routing comparison may vary only ONE routing variable (one description, or one skill's indivisible routing face) for its delta to be attributable | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#一次改前/改后对照只准动**一个路由变量** | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source-side paired single-description A/B is the observed working mechanism (borrow round; sanitized provenance in the private alias archive); no mis-attribution incident observed in this repository yet, so the clause lands as protocol item 5 with the existing four items unchanged as the paired control. |
485
+
486
+ | Cross-skill / cross-reference routing pointers in body text must carry the routing quadruple (trigger / scope / output / return point); a bare "refer to X if useful" pointer is never a landing shape | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/description-authoring.md#routing pointer in body text must carry the routing quadruple | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Borrowed from an adopting skill pack where unbounded pointers were the dominant dead-routing shape (two adopters verified in source; sanitized provenance in the private alias archive); landed in the routing-surface authoring reference because the entrypoint is size-ratcheted level — the reference is the required pre-edit reading for routing-surface work, and eval-routing.md's silent-skip row points back at it; the description-side Skip-when idiom already satisfies the quadruple and is named as the unchanged control. |
487
+ | Review-finding fixes are held un-applied until the full review+challenge chain has run on the frozen candidate: under the 1+1 budget apply-now is never fundable after the review round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (enumeration item 1 + cadence Round 2). Observed failure: a prior round applied its review fixes before the challenge; the tracked chain broke (challenge binds to the round-1 candidate), the round lost its double-receipt terminal, and closure required a user-granted continuation chain. The prior wording stated the rule only as a fundability conditional whose arithmetic the agent under pressure never ran; the operative unconditional form (hold all fixes; challenge on the frozen candidate; land the batch after the chain) is now explicit at both firing points. RED baseline: the recorded chain-break incident is the without-change failure; the with-change compliance surface is the explicit hold rule at the enumeration walked when a round returns findings. |
488
+ | A frozen eval case is sacred: deleting or re-scoping a bank task or golden trace requires a same-round `case-retired:`/`case-rescoped:` register adjudication row, and average improvement never offsets a frozen-case loss | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#平均改善不得抵消单条冻结案例的失守 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/eval-routing.md (冻结案例神圣 bullet), scripts/test_frozen_case_sanctity.sh (fast lane, registered), docs/f4-skill-effectiveness-harness.md (pointer line). Rule semantics: a previously-passing frozen case that degrades — including to unsure/INCONCLUSIVE — is a regression, and its sacredness attaches per case, so no aggregate improvement offsets it; the mechanism's source-verification record lives in the round's private archive, and transferred evidence does not exempt the behavioral row. RED baseline (replayed, throwaway clone at the round base): deleting the non-pinned bank case ctrl-unit-test passed the pre-change surface silently (test_routing_bank_integrity.sh exit 0) and reddens the new gate (exit 1 naming the id and the required adjudication row); re-scope and golden-trace-deletion mutants red for the right reason; adjudicated-deletion and untouched-tree control legs green; no-base and unresolvable-base legs print the explicit skip token. |
489
+ | Rationalization tables are built from excuses captured verbatim in baseline/pressure runs, never invented: counter only what a run actually said, and new observed excuses accrete counter rows | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/validation-and-landing.md#counter only what a run actually said | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/validation-and-landing.md (eval-first steps 2 and 4) with the sourcing clause mirrored into references/rule-consolidation.md's discipline-slip form row. The clause is scope-bounded to discipline-slip failures (prohibition-form counters measurably backfire on shape/omission failures per the form table it points at); the source-verification record lives in the round's private archive. RED baseline (fair tempting scenario, headless fresh-context, small-N 2x2, fully separated): asked to harden a merge-authorization gate with no run data supplied, the no-clause arm invented 12+ Excuse-Reality rows in both reps with zero mention of captured evidence; the with-clause arm produced zero invented rows in both reps and stopped to request actual run transcripts. |
490
+ | The draft-time security axis walk screens skill text as a prompt: leakage-inducing, permission-overreaching, or unsafe-automation wording is named at the axis-1 instance list and may appear only as a labelled anti-example | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#text a draft must not carry except as an explicitly labelled anti-example | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (pre-cover axis (1) instance list). Semantic control: axis (1) already required a security/authority pre-cover with at least one negative case per applicable axis; this names three instances (prompt leakage, overreach, unsafe automation) inside the existing obligation rather than adding a new gate, so the walked enumeration, the challenge mandate, and every other axis are unchanged; existing anti-example discussions stay legal via the labelled-anti-example carve-out. |
491
+
492
+ | Reference files carry a delta-ratcheted line budget: a new or crossing `skills/*/references/**/*.md` over 500 physical lines blocks, an already-over reference may shrink or stay level but never grow, the append-only ledger is excluded by gate design, and a long new reference without `##` structure draws a navigation advisory | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#stays inside the reference line budget | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in scripts/check-size-budget.sh (reference ratchet + counters + `reference_line_budget_blocking_ok` / `_failed` / `_unevaluated` verdict tokens), scripts/test_check_ccl_size_budget.sh (legs g0-g9, registered in the fast lane through the existing size-budget suite entry), and references/attention-budget-ratchet.md (the five design invariants every size/budget gate must satisfy: stable proxy estimator, anti-false-green sentinel, zero tolerance for new debt, legacy-shrink-only, missing-baseline-is-never-a-pass — plus the write-side reference norms and the per-clause verdicts on the external source). Write-side gap was measured before implementing, not assumed: 338 reference files, 4 over 500 lines, 104 between 101 and 300, and zero carrying a table-of-contents block. Paired RED at the round base in a throwaway checkout: a new 703-line reference passed the pre-change gate silently (exit 0, `entrypoint_size_blocking_ok`) and reds the new one (exit 1, naming path, head_lines=703 and the 500 allowance); on the live corpus, appending one line to the 691-line source-to-skill-extraction.md blocks as `over-limit reference grew`, deleting three lines passes with `reference_line_budget_legacy_ok allowed_lines=691`, appending 200 ledger rows passes with the `ledger_excluded` visibility token, and the untouched tree is green with over_limit_count_delta=+0 (the corpus is frozen, not retroactively reddened). Oracle self-proof (applied mutants on a copied gate, differential, control green): disabling the over-budget branch, allowing legacy growth, an off-by-one budget, removing the ledger exclusion, downgrading the unknown-base partial to a print, dropping the reference verdict from the clean-exit aggregate, and promoting the navigation advisory to a block each red on their own named leg; a first mutant attempt that broke the program was discarded as red-for-the-wrong-reason and re-applied semantically. Author-dogfood leg: the gate blocked this round's own entrypoint edit at +56 then +18 body words until the landing was funded by consolidation, which is the invariant working on its author. Review-chain repairs folded into the same round commit (tracked chain 078-r6, 1 review + 1 challenge, all four findings applied after the chain closed on the frozen candidate): line endings are normalized to LF before the line and heading counts, so a CR-delimited 501-line file can no longer read as one line and earn a clean verdict; the navigation advisory counts EXACT H2 headings, so an H3-only long reference still draws it; the head census guards every read and degrades its COUNTER to unknown rather than aborting the program before the per-file fail-closed verdicts run; and the legacy freeze gained a level-edit leg so a `>` silently becoming `>=` cannot pass. Four further mutants (H3-counts-as-section, no line-ending normalization, level-edit-blocks, census-error-reads-zero) each red on their own new leg with the control green, the last of them only after its leg was added — it survived the first re-walk, which is why the walk was re-owed after the fixes. Registered-claim narrowing recorded (a deferred registration is hypothesis, not spec): the official 500-line figure is SKILL.md-scoped and already covered more strictly by the body-word ratchet, the registered table-of-contents mandate is NOT PRESENT in the primary source (the official remedy is one-level references) so it landed as a `##`-structure advisory on repo-internal evidence only, and the three-model test matrix is not mechanized because this repo ships no model-pinned skills. |
493
+ | Core Rules own each invariant and a Workflow step only points at it: same-facet text living in both surfaces is converged toward the canonical surface rather than restated, and the sweep enumerates candidates instead of fixing whichever one a diff happens to touch | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#do not make normal users route through a source name | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The Step 6 cross-section facet-ownership check already owned this rule and fires on any edit touching Core Rules or a step; this round ran it as a five-candidate enumeration rather than a spot fix (the user challenged an earlier framing that treated the duplications as a word-budget offset ledger). Verdicts: content placement (Core Rules content-placement bullet vs two Step 5 bullets) converged to a pointer; the sibling-generalization mini-map field list (Core Rules owner-generalization group vs Step 4) converged to a pointer keeping only the step-order clause; capability naming (Core Rules naming bullet vs Step 5 vs two Step 6 checklist lines) converged to one Step 5 pointer plus one merged Step 6 residual-search check; representative sampling (Core Rules full-ask prohibition vs Step 3 labeling duty) judged complementary and left unchanged; the source-register row schema (Step 3 vs references/source-register.md) left unchanged as out of this contract's Core-Rules-versus-step scope and useful where an author builds rows. Zero-loss obligation map for the four rewritten passages: entrypoint-owns-trigger/routing/non-negotiables and references-own-detail both survive in the Core Rules content-placement bullet; the mini-map field list, the `update`/`unchanged`/`route-to-shared` vocabulary, and the smallest-common-owner routing survive verbatim in the Core Rules owner-generalization bullet, with `then add stack-specific implementation notes only where needed` kept in the step; the capability-name examples, the never-name-after-source clause, and the do-not-route-users-through-a-source-name clause all survive in the step, and provenance-labelling survives in the Core Rules naming and provenance bullets; the Step 6 residual-search list gains the page-name and scenario-label terms the two merged lines carried separately, and keeps `absent from executable guidance or clearly marked as provenance`. Net effect on the frozen entrypoint: base_body_words=16759 head_body_words=16750 (-9), so the round funded its own additions and left the entrypoint smaller than it found it. |
494
+ | A base-relative gate's design-time premise is measured against EVERY base its landing faces resolve, never the current round's base alone: the set is read off CI's own base-resolution expression rather than guessed - one face per pull-request target branch plus the pushed branch's previous tip on a push build - and each resolution is a separate run of the author-dogfood leg, a difference that accumulated before the gate existed surfaces as one violation on the face nobody measured, and the repair is to shrink the frozen surface rather than add a cross-base exemption | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#enumerate the landing faces, never assume one | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key; the changed file is `references/dual-track-review-gate.md`. **Observed failure (RED, recorded incident + re-computable).** The reference line ratchet landed in the prior round measured only against the round's own base. Reproduction on the pre-fix candidate: `CCL_SKILL_BASE_REF=origin/main bash skills/skill-extraction-workflow/scripts/check-size-budget.sh .` printed `reference_line_block: ... dual-track-review-gate.md: over-limit reference grew base_lines=662 head_lines=668` and `reference_line_budget_blocking_failed`, while the same command with `CCL_SKILL_BASE_REF=origin/dev` printed `reference_line_budget_blocking_ok`. The gate behaved exactly as documented; the defect was that leg (a) said `the SAME base resolution CI uses` (singular), so the author measured one face and the six lines that older rounds had added to a frozen surface only became visible on the promotion face. **Compliance (with-change).** Same two commands on the head candidate: base=origin/main `base_lines=662 head_lines=661 allowed_lines=662` and base=origin/dev `base_lines=668 head_lines=661 allowed_lines=668`, both ending `reference_line_budget_blocking_ok`. Running BOTH faces is this round's own dogfood of the rule it lands. **Zero-loss obligation map for the five consolidations that funded the addition** (all within the same file per the frozen-reference funding rule; 668 to 660 lines, and 125978 to 126158 bytes - the line unit the ratchet measures fell, the byte count rose by the amount the new obligation text exceeds the recovered duplication, which is stated here rather than hidden). (1) The four-line raw-CLI preamble collapses to one line carrying all three of its propositions - diagnostics only, never review or challenge evidence, never a replacement for the owner wrapper on a non-wording lane. (2) The standalone do-not-iterate-to-zero-findings paragraph merges into the convergence-bar paragraph, keeping the design-tradeoff clause, the pre-existing clause, the over-correct-or-scope-creep clause, and both contrasts - the bar differs from zero findings AND from no-new-P0-P1. (3) The R0-evidence value menu was stated verbatim three times; two occurrences become pointers to the review-pass row that keeps the menu, matching the pointer form the adjacent Item-9 field already used. (4) The Rules bullet restating the behavioral-evidence table's own two rows is dropped; its one clause absent from the table, that an author cannot self-assert semantic-control, moves into the semantic-control cell itself. (5) The trailing what-a-fallback-is-worth paragraph folds into the reviewer-ladder item that already owns that question, keeping the ad-hoc-run bar, the remediation-versus-evidence distinction, and the interim rule. No obligation was dropped and no paragraph was re-wrapped to buy lines. **A sixth consolidation was attempted and reverted, which is the reusable finding.** The premise-verification bullet in the self-audit section reads as a verbatim restatement of the two paragraphs above it, and deleting it looked free; `register_firing_path_unresolved` then failed because an earlier round's register row anchors its firing path on that exact line. A rule line can be another row's evidence, so an append-only ledger makes some prose non-deletable: check the anchor set before treating any rule line as redundant, and restore rather than EXEMPT when the deletion was to fund your own budget. **Other owners.** `references/attention-budget-ratchet.md` is `unchanged: already-covered` with a real firing path - its five-invariant preamble already says the budget gates are `the budget-gate instantiation of the design-time operability check in dual-track-review-gate.md - run that check's four legs too`, so the tightened leg (a) reaches ratchet authors through that pointer and restating it there would be the same-facet drift the Step 6 check forbids. `scripts/check-size-budget.sh` is `not-applicable`: the gate is correct as shipped and judges whatever base it is handed, so multi-base topology belongs to the caller, not the script. **Residual risk, stated rather than hidden.** The firing path is a walked design-time enumeration in prose, not a mechanical multi-base run; `Makefile` still defaults `CCL_SKILL_DEFAULT_BASE_REF` to the integration branch, so a local check still measures one face unless the author enumerates. A mechanical all-faces target is deferred with an owner - it would be a new gate surface owing its own four legs, oracle self-proof and suite registration, which is a round of its own rather than a rider on this one. |
495
+ | A design-time obligation over a SET is discharged by a manifest a reviewer can diff against the set's authoritative source, never by an asserted walk: the manifest names one entry per member with the value that member resolved to and the check's verdict on it, and a set with no finite manifest is a blocking residual routed to the existing non-blocking or risk-owner-deferral exits rather than reported as coverage | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#recorded manifest, never by a walk you assert | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key; this is the post-review fix batch of the same round, kept as its own commit so the externally reviewed candidate stays identifiable on the branch. **Observed failure (RED).** The first formulation of the preceding row's rule said to run the design-time leg `once per resolution CI can produce`. Both lanes of the dual-track chain independently reached the same defect on the frozen candidate: the review lane found the required set is not statically enumerable because an unrestricted pull-request trigger can resolve any target branch, so an author must either guess a subset and falsely claim completion or cannot satisfy the rule; the challenge lane found the same accumulated cross-base violation can therefore still ship despite apparent compliance, since the unchanged per-base gate cannot detect the omitted face. Two independent lanes converging on self-certifiability is the self-adjudication shape this file already names - the classification verb had no named output that produces the classification. **Fix.** The obligation now takes a manifest derived from the workflow itself, one entry per landing branch carrying the resolved base ref and the gate verdict, which a reviewer can diff against the workflow; and an unenumerable set is routed to the exits the premise-check leg already defines instead of being claimed as covered. No new mechanism is introduced - the fix converts an author-adjudicated claim into a reviewer-checkable artifact using exits that already exist. **Chain state.** Wrapper-fixed budget of one review plus one challenge, both bound to the same frozen candidate digest `70fd34678f99c444bf5b7c808a283380e99201eada6d71d076fd08e2f2ec1789`; fixes were held un-applied across both rounds. The final round itself returned findings, so the honest terminal state is `continuation_authorization_required` and this batch is post-review, not reviewed. A third review finding asked for the workflow file and captured command outputs inside the packet; it is dispositioned per this file's reviewer-verification-scope rule - deterministic-gate claims are verified by CI re-running the gates on the branch, never accepted from the implementer's prose - with the packet-composition miss recorded as a process defect for the next round rather than re-litigated here. **Supersedes two cells of the preceding row.** (i) Its compliance figures read `head_lines=661`, drafted before a later edit in the same round took the file to 660; the correct values are base=origin/main `base_lines=662 head_lines=660 allowed_lines=662` and base=origin/dev `base_lines=668 head_lines=660 allowed_lines=668`, both `reference_line_budget_blocking_ok`. (ii) Its rule cell describes the face set as read off CI's base-resolution expression; the manifest form in this row is the one that governs. The correction is recorded here rather than by editing that row: the ledger is append-only and the gate enforces it mechanically - a row edited after its own round committed no longer survives at HEAD, its round loses its only row, and the gate fails closed. That is the append-only contract catching an in-place fix, which is what supersede-by-pointer exists for. |
496
+ | A published version is immutable, so the source tree's version pointer must never sit below the highest already-released version; the invariant is checked at merge time against two independent records the repository owns - the release-tag set and the version the merge target already declares - because the tag record is deletable and no depth of fetch recovers a tag removed from the remote, and because the commit that lowers the pointer is typically an unrelated change whose rebase resolved a both-sides-changed hunk backwards | `release-coordination` | Enforced by `scripts/check-release-version.py`, wired into the `test-repo-gates` leaf CI runs. This row carries no impact-chain declaration: the rule's owner is `release-coordination`, which is outside the curated upstream-owner set, and the mechanism has no surface inside this register's owner package | updated | Observed miss: an extraction commit about review budgets rewrote all three version sites from 0.9.0 to 0.8.0 while 0.9.0 was already published and immutable; a human restored it later. Paired RED on the replayed tree: the four pre-existing PR-time repo gates each exit 0 with clean tokens, the new gate exits 1 naming the declared version, the released version, its tag, and all three sites to repair. Six applied mutations of the gate are each killed differentially by their own leg (numeric ordering, tagless-is-unevaluated, the below-floor comparison, the nested lockfile site, the base floor, the unreadable-file catch); the control tree is clean at 18 legs. The base floor was added post-chain after both reviewer lanes independently attacked the tag record's deletability; it makes the gate partly base-relative, so its design-time landing faces are recorded rather than asserted: pull_request into dev (base floor 0.10.0), pull_request into main (0.10.0), push on dev previous tip (0.10.0), push on main previous tip (0.9.0), and an unresolvable base (floor absent, tag floor alone) - all five exit 0 on the candidate, and the verdict is stable across them because the tag floor dominates. Residual recorded rather than closed: a version published without a tag, or the loss of both records, lowers the floor with them; closing that would put a registry query inside a merge-time gate. `skills/release-coordination/references/tag-and-prod-pipeline-gate.md` carries the rule beside the existing do-not-force-move-a-published-tag paragraph |
497
+ | A repository control that ENCODES a reading of an external contract carries that reading's expiry beside itself — the primary-source clause, the date and the tool/host version it was verified against, and the condition that invalidates it — because without them the next reader cannot separate a deliberate decision from a fossil, and re-verification costs the whole external read again | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: file:skills/skill-extraction-workflow/references/external-practice-controls.md#must be treated as carrying an unverified reading | updated | RED-baseline of the recorded-incident kind, observed in this round against the pre-rule state: the packed-artifact verifier carried a machine-enforced ban on a `version` key in plugin.json with no source, date or expiry, so reading the control could not distinguish a decision from a fossil - and applying the new rule to it is what surfaced that the ban read only one of the two plugin manifests, leaving the load-bearing one unguarded across every release shipped so far. The paired observation is that the same control had passed the full gate suite and both reviewer lanes in earlier rounds without that gap being visible. Merged into `external-practice-controls.md`'s inherited-reading section as its record half; the predicate half (own the predicate, not the upstream's vocabulary) already lived there. Scoped narrowly to external-contract claims: attribution's who-said-it tier stays with `attribution-verification.md`. Two live instances found while reading, both annotated: the packed-artifact verifier's ban on a `version` key in plugin.json, and the new release-version gate's dependence on `actions/checkout` fetch-depth. Both reviewer lanes then caught the first annotation failing the rule it illustrates - it carried a date but no host version - which is the rule's own dogfood arriving as a finding; both annotations now pin the verified host or action major and state which upstream release re-opens them. Annotating the first surfaced that the ban read only the Codex manifest, leaving the Claude one — the manifest whose version actually pins host updates — unguarded; the ban now covers both. The owning entrypoint `skills/skill-extraction-workflow/SKILL.md` is deliberately `unchanged`: both rules merge into references its Core Rules already route to - the inherited-reading section for the record half, and the form-by-failure table its rule-consolidation pointer already owns - so restating either at the entrypoint would be the same-facet drift the Step 6 cross-section check forbids. Also landed: two rows in `rule-consolidation.md`'s form-by-failure table (plan-validate-execute for batch/destructive steps; freedom tier matched to operation fragility), both verified against the Anthropic skill-authoring best-practices page on 2026-09-01 |
498
+ | A gate whose verdict is computed per OWNER but caused by one ROW must name the offending row, not only the owner: an owner-scoped diagnostic sends the author to inspect the row they just wrote, which is usually the correct one, while an unrelated row that the round merely EDITED is the actual offender — editing an already-committed ledger row makes it an added row of this round, so its own original anchor stops resolving | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: command:skills/skill-extraction-workflow/scripts/impact-chain-gate.rb | updated | Observed in the preceding round: a correct new row failed the owner because an unbounded string replace had also modified a pre-existing ledger row; three iterations rewrote the good anchor before debug output patched into the gate found the real offender, and the round then recorded a non-existent encoding defect as its root cause. The diagnostic now lists each failing row's first cell. SUPERSEDES that round's claim that a non-ASCII firing-path anchor can never validate: `locator_valid` resolves blobs through `blob_at`, which reads without forcing an encoding, while only the separate `raw_blob_at` forces binary. Paired fixtures holding every other variable fixed — uniqueness, the 16-character bar, list-line shape, whitelisted verb — show a Chinese anchor validating normally, and that positive control is now a suite leg (`case-ref-non-ascii-anchor-accepted`). Four applied mutations kill differentially: reverting the diagnostic to owner-only kills the offending-row leg; forcing `blob_at` to binary — the implementation the previous round wrongly believed existed — kills the Han pair; dropping the control-character scrub kills the control-byte leg; and special-casing the one Han literal kills the NFC pair. That last mutation is why the acceptance proof is a differential TABLE rather than one passing literal: the challenge lane observed that a single-literal leg is satisfied by a special-cased implementation, so each non-ASCII anchor is paired with an ASCII control matched on everything the gate predicates on — uniqueness, the 16-character bar, list-line shape, a whitelisted normative verb — and the two verdicts must agree across Han, precomposed NFC, and decomposed NFD. The diagnostic scrubs C0/C1 controls, bidi overrides and zero-width characters before truncating, because it renders contributor-controlled ledger text onto a terminal: both reviewer lanes independently found that raw ANSI or carriage-return bytes could erase or forge the surrounding diagnostic. Suite at 88 cases; its own case-count guard caught every addition rather than absorbing them silently. `skills/skill-extraction-workflow/SKILL.md` unchanged because the anchor contract it states was already correct |
499
+ | A landed conclusion — a claim that a capability is unavailable or impossible, or a DIAGNOSIS of why an observed failure happened — is a hypothesis until an operation that could have falsified it has been run; the two admissible operations are the suspected mechanism exercised on the path that actually ran with the reaching call site named, and a paired control differing in exactly ONE variable with every other precondition of the tested predicate enumerated and confirmed equal, and an unfalsified cause is labelled `hypothesis` and kept off the shared surfaces | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#until an operation that could have falsified it; revalidate-when: a long-session probe puts the tempting hit mid-task rather than in a one-shot answer, so the with-change half of this row can be recorded instead of left inconclusive | updated | `skill-extraction-workflow/SKILL.md` is the owner key; supporting changes in `skill-extraction-workflow/references/validation-and-landing.md` (method detail) and this register. Observed failure (RED) in the preceding round: a gate rejected a ledger anchor, a `force_encoding(BINARY)` call found by grep in the gate script was recorded as the cause, and that cause landed in a merged pull-request body, a durable note and the round charter; the suspected line was never on the checked path and the real cause was an unbounded string replace in the round's own edit. The pre-rule tree obligated a falsifying attempt only for unavailability/impossibility conclusions and for done/covered/converged claims, so a causal diagnosis fell between the two and rode through a full dual-track gate. Applied differential control in THIS round, scoped honestly: the new `defect-diagnosis` anchor line was first written with "does not enter", `impact-chain-gate.rb` rejected that owner because "does not" is excluded from its normative vocabulary, and changing that single token to "must not" with the rest of the line, the row and the diff held constant turned the same command green. That is the executed-path proof of this round's own reading of that predicate and evidence that the anchor gate is sensitive - it is NOT evidence that the rule moves agent behavior, and the independent review was right to say so. The RED half of this row is therefore the recorded incident alone; the with-change half is NOT established, and the row is honest about that rather than resting on the mutation. BEHAVIORAL ATTRIBUTION NOT ESTABLISHED - read this before treating the row as efficacy evidence: both halves of the baseline are executed and recorded, but the paired control complied too, so no delta is demonstrated. Detail: with the changed text as the operating layer an agent kept the same tempting cause hypothesis-grade, named the competing causes and asked for the discriminating observation - but the paired control carrying the base text of the same passages did the same, so no delta was demonstrated. Probe detail: two headless arms differing only in whether the changed or the base text of these passages was the operating layer, scored against a rubric frozen before either run, three independently designed scenarios were run, and the control passed all three: a neutral question, one pressing for a confident cause paragraph to paste into a pull request, and - the shape that targets this rule most directly - a synthetic repository with a known-by-construction ground truth, where a grep-visible force_encoding sits on a dead path while the real cause is a uniqueness predicate, run once as an investigation task and once as a closeout task whose stated cause was inherited, false, and already followed by a green pipeline. In the closeout arm the control verified the inherited cause and corrected it unprompted. The first probe pair additionally ran with host auto-memory and repository instructions enabled and was contaminated - the control cited this repository own earlier round - so only the reruns with both disabled count; the synthetic-repository fixture and its frozen rubric are reproducible by a third party. The probe is therefore recorded inconclusive rather than as a demonstrated behavioral delta: a bounded task does not reproduce the context load the observed failure happened under, and the honest reading of three passing controls is that the base text already yields the wanted behavior on tasks this size - the rule earns its place on the recorded incident and on two independent review lanes endorsing its wording, not on a measured delta. The standing RED is the recorded incident above; what would settle the with-change half is a long-session probe where the tempting hit appears mid-task. Landed by consolidation rather than addition: the entrypoint's body-word count falls from 16750 to 16749 with the merged rule in place, the collapsed passages being ones a cited reference already carries verbatim |
500
+ | A hypothesised cause is stated together with the observation that would FALSIFY it, and that observation is collected first: a search hit, a log line or a plausible implementation detail proves the text exists, not that it ran on the failing path, so the diagnosis must name the reaching call site or build a one-variable paired control before the cause may become a fix, a commit or merge-request body, a durable note, or a report | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/defect-diagnosis/SKILL.md#the observation that would falsify it | updated | `defect-diagnosis/SKILL.md` is the owner key. Downstream executable owner of the preceding row's upstream rule. Its hypothesise step previously asked only for the observation expected IF THE HYPOTHESIS IS TRUE, which the observed failure satisfied exactly - the grep hit was that expected observation - so the step licensed the confirmationist probe rather than blocking it. The upstream rule governs what may enter a landing; this row governs the diagnostic step itself, which is where the executed-path form is actually applied. Same RED as the row above, plus the paired-control half: three probes built in the preceding round to test the wrong cause each differed in more than one precondition of the predicate under test - anchor length against the threshold, uniqueness within the file, vocabulary membership - so no verdict among them was attributable |
501
+ | A stack dev owner that is absent from the lifecycle coordinator's dispatch enumeration is unreachable from every multi-stage delivery in that stack, and the coordinator's own no-owner fallback clause then silently reverses the ownership a previous round established on the executor side | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#merely because it is a command; bank-evidence: file:eval/routing-tasks.jsonl#route-nodejs-multistage-to-product-rd | `updated` | Owner key `product-rd-workflow/SKILL.md`. RED (measured on origin/dev): `grep -rc nodejs-service-dev skills/product-rd-workflow/` returned 0 across the whole package while the Development-routing line named the Go and Python dev owners and then fell back with "a CLI in a language with no such dev owner stays with `terminal-cli-dev`" — so the Node CLI ownership landed by rows 371/372 was reversed by this clause. Candidate adds the Node owner to the routing enumeration and to the CLI carve-out list, adds a stack-owner map to the product-rd dispatch-owner-skills reference, fills the Node architecture vacuum in the delivery-lifecycle reference and the problem-resolution-and-learning reference per the maintainer's decision not to create a Node architecture sibling, and funds every addition inside the severe-debt entrypoint by in-file consolidation (head_bytes= 76200 vs base 76200). Correction to an earlier draft of this round: that draft moved the two CLI carve-out sentences into the reference, but the routing pointer-integrity suite pins both branches on the entrypoint, so both were restored verbatim and the budget repaid elsewhere in the same line. Independent review then found the round's routing-bank case does NOT guard the coordinator omission — a multi-stage Node request selects the coordinator from its own wording and stays green with the Node owner removed from the dispatch map — so that fixture is advisory only. The actual probe is three checks pinned over the stack enumeration, the CLI carve-out list, and the design-gate architecture destination; each was verified by an APPLIED mutation that turned exactly its own check RED with no non-owning assertion failing, against a green unmutated control. |
502
+ | A stack implementation owner with no `*-architecture` sibling must name the owner that absorbs its architecture decisions, or the stack is reachable for implementation and unreachable for design | `nodejs-service-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/nodejs-service-dev/SKILL.md#Do not borrow the Go or Python architecture skill | `updated` | Owner key `nodejs-service-dev/SKILL.md`. RED (measured on origin/dev): the package named neither sibling stack owner and carried no generalization section (0 hits for `go-microservice-dev`/`python-service-dev`/`Generalization`), and `AsyncLocalStorage` resolved only inside the maintainer source map — a file whose own preamble says it is "not required reading for ordinary Node.js implementation work", so the extracted constraint had no executable landing. Candidate adds the architecture-owner statement, a Bun/Deno runtime boundary, a Generalization Discipline section carrying the inward and outward sibling duty, the request-context rule (primary source: Node async_context docs, which recommend the built-in store over custom `async_hooks`), and an outbound-HTTP-client section grounded in undici's own docs (global dispatcher backs built-in `fetch`; per-origin pools default to unlimited connections; connect/headers/body timeouts are layered with no overall deadline; an unconsumed response body holds its pooled connection). Version-sensitive numbers are deliberately not hardcoded. Independent review round 2 raised a P1 on the first draft of the context rule: prescribing a declared default for an absent store is unsafe when the stored value is security-bearing, because a tenant, subject, or permission scope that defaults executes the operation under the wrong identity. The rule now splits by what the value authorizes — diagnostic values may default, security-bearing ones fail closed — and states that the store is a propagation mechanism, never the authorization decision. This was the draft-time security axis the pre-cover sweep should have caught before handoff, not the reviewer. |
503
+ | A defect-routing table that enumerates stack owners must carry every stack that has an owner, and must say where a stack's architecture defects go when that stack has no architecture sibling | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/defect-diagnosis/SKILL.md#must not be filed under a sibling stack's architecture skill | `updated` | Owner key `defect-diagnosis/SKILL.md`. RED (measured on origin/dev): 0 hits for `nodejs-service-dev` in the package while the prevention-routing table listed Go and Python implementation and architecture rows, so a Node defect had no durable landing place. Same-class member of the sweep this round performed after the set-difference predicate surfaced four owners with the identical shape. |
504
+ | A test-layer owner that hands implementation mechanics to stack skills must name every stack owner, or the layer choice lands with no executor for that stack | `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/testing-strategy/SKILL.md#packaged-artifact verification must use | `updated` | Owner key `testing-strategy/SKILL.md`. RED (measured on origin/dev): 0 hits for `nodejs-service-dev` while the post-layer-choice pointers named Go, Python, app, miniapp, web, terminal, and LLM owners. The addition is funded inside this severe-debt entrypoint by consolidating the phrase "after the test layer is chosen", which was repeated on all 8 stack-pointer lines, into one statement on the scope line: net -2 bytes vs base with the repeats removed. |
505
+ | A skill that hands a host stack back to its owner must cover every stack that has one, including the disposition of that stack's architecture decisions | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/SKILL.md#never to a Node architecture sibling | `updated` | Owner key `llm-inference-integration/SKILL.md`. RED (measured on origin/dev): 0 hits for `nodejs-service-dev` while the hand-back clause named the Go and Python architecture and dev owners on identical terms. Same-class member of this round's sweep. |
506
+ | A test-code implementation pointer list is a routing obligation, not a convenience enumeration: a stack missing from it has approved test cases and no owner to implement them | `test-artifact-management` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/test-artifact-management/SKILL.md#must go to the owning stack skill | `updated` | Owner key `test-artifact-management/SKILL.md`. RED (measured on origin/dev): 0 hits for `nodejs-service-dev` in the six-owner implementation pointer. Same-class member of this round's sweep; the line is also restated as a normative routing obligation rather than a suggestion. |
507
+ | Member-join obligation inheritance enumerated from a remembered list of obligation types reproduces the miss: the list that shipped named four routing-surface forms and omitted the coordinator's dispatch enumeration and the shared body section, which are exactly what went missing on the next join — so the predicate must be a mechanical set difference over where an established sibling is named | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/source-to-skill-extraction.md#do not enumerate the inherited obligations from a remembered list | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: row 371 ran the join-side sweep for `nodejs-service-dev` and dispositioned the remaining class obligations as unchanged/not-applicable, recording `not-applicable: nodejs has no *-architecture sibling` — which restates a routing vacuum as a reason. One round later the coordinator enumeration, the sibling generalization loop, and four peer routing tables were all still missing the member. RED baseline (measured, both directions): on origin/dev the new predicate `comm -23 <(grep -rl python-service-dev …) <(grep -rl nodejs-service-dev …)` emitted every one of the five files this round had to change (the lifecycle router entrypoint plus its delivery-lifecycle and problem-resolution references, and both sibling stack dev entrypoints); on the candidate all five are absent from the difference, and the 30 remaining paths are candidates requiring disposition, four of which were confirmed real and swept this round. The predicate replaces the type list rather than extending it, per the same-class-recurrence rule that a proxy predicate be re-expressed over an invariant the control owns. The reference stays level at 691 lines by folding the two new clauses into the existing bullet. The same owner's routing pointer-integrity suite gains four coordinator-dispatch assertions and a line-scoped assertion helper; its self-pin count moves 21 -> 24, so silently deleting one is itself caught. (Two earlier drafts of this row recorded 25 and then 23; each described a candidate that a later round superseded, and the number is corrected here rather than left to describe a tree that no longer exists.) The adversarial challenge rejected the first form of these checks: whole-file substring greps pass when the Node tokens are moved into unrelated prose while being deleted from the line that carries the obligation, and the architecture check passed for a sentence that declared the vacuum without naming any destination. The assertions are now scoped to the carrying line and require the complete relationship (both owner enumerations on the routing line; both the platform owners and the Node implementation owner on the design-gate line). Verified by applied mutation: moving the token out of the routing line into stray prose leaves the OLD whole-file form green (the hole, reproduced) and turns exactly the new line-scoped assertion RED; stripping the destination from the architecture sentence turns exactly its two assertions RED; the unmutated control and the restored tree are green, with no non-owning assertion failing in either mutant. A later review round on this same candidate reported that the diff DELETED the sibling owner's causal-falsification rule. The report's direction was right and its attribution was not: the rule had landed on the target branch after this round's worktree was cut, so `git diff <target>` rendered "the branch is behind" as "the branch deleted it" — the stale-branch shape, on a real collision set (both rounds changed the same two files). Resolution followed the update-before-merge recipe rather than a text edit: pin the target, compute the pre-update merge-base, list the collision set, rebase (the branch was never pushed, so no shared history was rewritten), and resolve the ledger conflict by keeping BOTH sides' rows in append-only order rather than taking one side. Content-layer verification then confirmed the sibling round's rule and both of its ledger rows survive on the rebased candidate, this round's seven rows and its routing rule are intact, and all eighteen deleted lines are this round's own intended rewrites. Recorded because `--stat` cannot see an in-file revert, and because a finding whose ATTRIBUTION is wrong can still name a real defect: the fix belonged to the baseline, not to the prose the finding pointed at. The landing round then produced two findings of ONE shape — an assertion weaker than the contract it states — and they were swept as a class rather than patched individually: the per-requirement form asserted each token against the whole set of selector-matching lines, so two requirements could be satisfied on two different lines while the contract said one line must carry both, and neither architecture assertion pinned the positive ownership clause, so deleting it while keeping both owner tokens left the architecture decision ownerless. The four assertions collapse into two that filter the matched lines through every requirement in turn, with the ownership clause pinned alongside the two owners. Applied-mutation evidence, both legs measured: deleting the ownership clause while keeping both owner tokens leaves the pre-fix form green and turns the new assertion RED; splitting the requirements across two selector-matching lines does the same. The first control leg for the ownership mutation was invalid — it included the newly added requirement in the supposedly OLD form, so both arms went red and proved nothing — and was rerun with the actual pre-fix requirement set before the result was recorded. The landing round also closed a conflict the review chain surfaced and an earlier draft of this round had deferred: `testing-strategy` claimed `terminal-cli-dev` for "command-line ... implementation", which collides with the CLI carve-out that the coordinator and every stack dev owner carry, so a Node, Go, or Python CLI test could route to a skill that does not own the implementation. Measured as pre-existing on the target branch — the claim and both sibling stack pointers are older than this round — it was resolved by DELETING the over-broad claim rather than adding a fourth copy of the hand-off rule: the line now scopes to the terminal interface contract, PTY/ANSI/keyboard rendering, and full-screen TUI testing, which costs 13 bytes less inside a severe-debt entrypoint and leaves the hand-off stated once, where it is already pinned. A same-line assertion pins the narrowed scope, verified by an applied mutation that restores the old wording and turns exactly that assertion RED. Separately, `frozen_at_sha: root` was raised in three rounds and stays refuted on measurement rather than on argument: 143 of the 158 existing routing cases carry that exact value, so the new cases follow the bank's dominant convention, not a deviation from it. |
508
+ | Supersedes by pointer the two budget rows above where they state the extraction lane is two rounds and that a restart is never fundable: the lane is one review plus one challenge on the frozen candidate PLUS one succession challenge owed exactly when the held fix batch moved the candidate, and the trigger is the candidate hash rather than any disposition label the author writes | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in references/dual-track-review-gate.md, references/extraction-quickstart.md, scripts/validate_extraction_review_state.py, scripts/review_ledger_binding.py, and both fast-lane suites). Observed failure, found by reading the gate against itself: the anti-pattern clause demanded "always do at least one re-challenge after a non-trivial fix-up" while the cadence plus the hold-fixes rule made that round unreachable under the two-round budget — an obligation no reader could satisfy, which reads as satisfied. Underneath it, the closeout validator required every counted round to bind the ledger candidate, so once the held batch was applied the ledger could not close on what lands at all: the batch left the gate's jurisdiction and the only remaining lane was human MR review, which this same document disqualifies as dual-track. Predicate choice recorded because the obvious one is wrong: "a P0/P1 was dispositioned fixed" is a self-adjudication clause of exactly the shape this reference already forbids, so the trigger is the candidate hash the control owns — changing a label cannot fake it, changing bytes is the thing being detected, and a round whose findings were all accepted or pre-existing pays nothing. Wrapper unchanged (22 lines, no allowlist, so the new flag transits; it is stateless and cannot count chains). Per-chain `challenge_budget` stays 1, so all 14 historical receipts in specs/*/evidence/review remain valid byte-for-byte; the lane bound is the new constant pair. RED-baseline (applied, differential): with the validator reverted to base, the succession-ready fixture reds at "controller receipt 1 does not bind the ledger candidate" — the exact invariant this round re-phases — and the restored validator greens; the pre-existing over-budget case stayed red for the right reason after the bound moved, which is what caught a non-succession third receipt slipping through, fixed by requiring any post-chain round to be a succession. |
509
+ | A caller-built ledger that nothing reads at merge time cannot enforce anything: the landing candidate is bound to the evidence that inspected it by a merge-side gate that recomputes the packet with the controller's own freeze, and evidence is kept outside the reviewed paths so committing the ledger cannot move the hash the ledger records | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md` (contract sentence updated net-negative under the entrypoint word ratchet; mechanics land in references/staged-review-contract.md and scripts/review_gate.py; the merge-side half and its CI wiring belong to the extraction lane's row above). Observed failure, measured not assumed: grep for extraction_review_state / closeout / receipt / review_chain across .github/workflows/ and check-ccl-skills.sh returned zero references, so the whole dual-track ledger was agent-run and agent-supplied with no merge-time reader — adding a third round alone would have changed nothing mechanically. Contract change: a fix touching a selected owner's SKILL.md or references/**.md moves selected_skills_sha256 and ends the chain by construction (verified in review_gate.py, not inferred: a changed candidate alone only adds a material_candidate_change self-review trigger and does NOT end the chain, and the owner digest excludes scripts/, so script-only fixes continue in-chain), therefore one succeeding chain may open at index 1 in challenge mode against a moved candidate. The owner digest is the single binding allowed to move — controller digest, owner-selection source, owner names, scope digest, and stage/depth/risk-tags/budget must all still match, an unmoved candidate is rejected as a repeat round, and prior focuses carry forward. RED-baseline (applied, differential): six new controller cases red against the unchanged controller and green after, with 257 pre-existing cases unaffected; the rejection cases were re-anchored after they were caught passing for the wrong reason — the old controller emits the same review_chain_invalid code for these arguments, so reason_code alone could not tell the mechanism from its absence and each now pins the succession diagnostic. The merge-side consumer of this contract carries its own suite in the extraction lane. |
510
+ | Supersedes by pointer the merge-side row above on two counts its own review found: the bound paths include the workflow directory, because with only the skill tree bound the CI step that runs the gate could be deleted without moving the candidate; and the gate has no evidence branch keyed on a self-declared field, because this gate cannot authenticate that a controller minted any file it reads | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py, scripts/check-ccl-skills.sh, scripts/test_review_ledger_binding.sh, references/dual-track-review-gate.md). Observed failure: this change's own review lane, rounds 1 and 2 against the frozen candidate, returned seven findings; four are dispositioned here. (a) A run with no base printed an unevaluated notice and exited 0, so a base-wiring mistake would have read as a passing required check, and the diagnostic named an environment variable the code never read — now fail-closed by default, with the variable actually read and an explicit --allow-unevaluated for events that genuinely have no base. (b) The wording-only branch accepted any file carrying the candidate hash and a non-empty proof field: hand-writable, so deleted rather than patched, and the cost is recorded — a wording-only change now owes the same evidence. (c) Nothing invoked the gate against a real checkout; a smoke now runs from the repository gate suite, deliberately dropping an inherited CCL_SKILL_BASE_REF because that checker runs against synthetic clones inside other suites where the leaked base resolves against the wrong repository — a hazard this repository has recorded before and which reproduced here on the first wiring attempt. (d) A stale sentence still told broad extractions to stop at two rounds. RED-baseline (applied, differential): each case reds the suite on its own assertion before the fix — the no-base case at the exit status, the receipt-shaped file at acceptance, the real-checkout smoke at packet freeze — and the fourteen-case suite is green after. |
511
+ | Supersedes by pointer the succession row above: succession is one-shot. A succession receipt may neither be continued inside its own chain nor become the next succession's predecessor, so the relaxation cannot be daisy-chained into unbounded autonomous rounds | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md` (entrypoint unchanged this round; lands in scripts/review_gate.py and its suite). Observed failure: the adversarial round of this change's own review lane found that a succession chain was minted at index 1 with a positive budget and nothing made it terminal — its receipt could enter the ordinary prior-result path for index 2, and that chain's terminal receipt could then be another succession's predecessor, composing without bound. The controller-side cap matters independently of the extraction lane's ledger bound, because other lanes consume the same controller. Both directions are now refused with their own diagnostics. RED-baseline (applied, differential): two new cases red against the pre-fix controller and green after, with the full suite at 259 passing and no pre-existing case disturbed. Bootstrap limit recorded rather than worked around: a round that edits the controller itself cannot use succession on its own predecessor, because the predecessor cannot preserve a controller digest the round just moved — observed live on this very candidate, which is the rule working, not a defect. |
512
+ | Supersedes by pointer the merge-side rows above with the residual those rows did not state: a gate that lives inside the candidate cannot authenticate itself, so the pinned CI step and the pinned fail-closed branch raise the bar without closing it, and the terminal authority is the platform's required-check configuration plus human review of the gate's own diff | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/extraction-quickstart.md#still owes the two-round chain before it can land | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/contract-anchors.tsv, scripts/review_ledger_binding.py, references/extraction-quickstart.md). Observed failure: the review and challenge rounds of this candidate's own second chain both found that a pull request may delete the workflow step or replace its command while keeping the required job name, and that the gate runs the candidate's own validator, so handcrafted receipts plus a validator that exits zero would pass. Neither is closable from inside the candidate. What lands is the honest pair: a contract anchor on the fail-closed branch so hollowing it also edits a registry another required check verifies — the workflow step is deliberately NOT pinned, because that registry addresses skill files and a cross-tree row reds the anchor checker against its own synthetic fixtures, which is how the attempt was caught — and a docstring that states what the gate proves — that what merges is the candidate an external round inspected — rather than implying it proves the round was honest. The wording-only path was also reconciled: the quickstart promised a single-review row with no ledger while the gate accepts only a validator-checked ledger, and the doc now records the cost rather than leaving the contradiction. RED-baseline (applied): the anchor gate reds on deletion of the pinned literal and is green at 10 anchors on the final candidate. |
513
+ | Supersedes by pointer the succession row above: inherited challenge focuses live in their own receipt field, because `prior_challenge_focuses` carries in-chain arithmetic that consumers derive from the round index, and a succession round arriving at index 1 with inherited focuses is a count the index cannot explain | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md` (entrypoint unchanged this round; lands in scripts/review_gate.py and its suite). Observed failure: the review round of this candidate's own second chain found that no test proved the completion checkpoint accepts the chain-index-1 challenge a succession produces. Adding that case turned it red immediately — the checkpoint rejected every succession round, because it computes the expected focus count as index minus two and the succession round carried its predecessor's focuses at index one. The ledger's terminal step was therefore unreachable for exactly the chain shape the third round exists to produce, and the defect would have surfaced only while closing a real ledger. Inherited focuses now populate `predecessor_challenge_focuses`; the distinctness rule still tests the union, so a succession cannot reuse a focus the ended chain already spent. RED-baseline (applied, differential): the new completion case red before the field split and green after, with the full suite at 260 passing and no pre-existing case disturbed. |
514
+ | The merge gate resolves its base to a commit id before that value reaches any git command, because an option-shaped base is read by git as an option and `git diff` with no revision compares the index to the working tree — which in a clean checkout reports no paths and passes the gate having compared nothing | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Observed failure: the adversarial round of this candidate's third review chain constructed the input — `CCL_SKILL_BASE_REF=--quiet` — and traced it to a clean pass. The environment path is the one that matters because no argument parser stands in front of it, which is exactly how a base arrives in CI. RED-baseline (applied, differential): three unresolvable bases (`--quiet`, `--name-only`, a ref that does not exist) each red the suite before the fix and are refused with a named diagnostic after, while a resolvable revision expression still reaches the normal verdict. The same round's second case is also pinned: the suite now compares `git status --porcelain --ignored` across a gate run, so a regression in the bytecode or packet-cleanup discipline reds instead of passing a hash back. |
515
+ | Succession terminality is the predecessor receipt's own arithmetic — challenge index, remaining rounds, and the allowed flag — not its round index alone, because a forged receipt can carry a terminal index while every other field still says the chain has rounds left | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md` (entrypoint unchanged this round; lands in scripts/review_gate.py and its suite). Observed failure: the adversarial round of this candidate's third review chain showed that a tracked challenge receipt with budget one and index two, but `challenge_index=0` and a positive remaining count, satisfied the terminality test and could be succeeded — extending the lane past the rounds its chain had actually spent. RED-baseline (applied, differential): the forged-terminal predecessor case reds against the pre-fix controller and is refused after, with the suite at 261 passing and no pre-existing case disturbed. |
516
+ | The merge gate reads committed evidence only — it enumerates from HEAD and refuses a dirty evidence tree — because what merges is the committed tree, so a ledger read from the working tree can be evidence about something nobody can find after the merge | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py, its suite, and .github/workflows/ci.yml). Observed failure: the fourth review chain of this candidate found that the gate globbed the working tree and passed whatever it found to the validator without proving any of it was committed, and separately that scoping the CI step to pull requests leaves a skipped step in a merge-queue run — where a skipped step does not fail its job. Both land: enumeration comes from `git ls-tree HEAD`, an uncommitted evidence tree is refused outright, and the step now also runs for `merge_group`. The residual is stated rather than closed: direct pushes to a protected branch are refused by the platform, not by this gate. Non-blocking item deferred with its reason: the controller returns the predecessor chain id without comparing it to the new chain id, so a self-succession is refused one layer up by the closeout validator rather than at mint time; fixing it would move the controller digest and forfeit this round's ability to close its own ledger with the succession round it introduces, so it is recorded here for the next round that touches the controller. RED-baseline (applied, differential): the uncommitted-evidence case reds before the fix and is refused with its own diagnostic after; the option-shaped-base and tree-perturbation cases from the previous batch stay green. |
517
+ | A gate that reports no-change must say so rather than surfacing an empty packet as a freeze error: the repository-gate smoke diffs against the parent commit, and the commit that lands the evidence touches no reviewed path, so the honest answer is `no-change`, not a broken gate | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/check-ccl-skills.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and scripts/check-ccl-skills.sh). Observed failure: self-inflicted and caught by this repository's own suites rather than by review — committing the closeout evidence made the parent-commit diff empty over the reviewed paths, the smoke asked for a candidate, the packet freeze refused an empty packet, and the checker exited before its later verdicts, reddening two unrelated suites (route drift and sync pointers) that only read those verdicts. The shape is worth keeping: a diagnostic that cannot distinguish "nothing to check" from "the check is broken" turns one benign state into a cascade. RED-baseline (applied): the smoke reds against the pre-fix gate on the evidence commit and is green after, with route-drift and sync-pointer suites recovering in the same run. |
518
+ | An assertion about a subject with two legitimate outputs must accept both, or it is an assertion about repository state rather than about the subject: the real-checkout case accepted only a candidate hash, so the commit that landed the evidence — which touches no reviewed path — turned the benign no-change answer into a red suite | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/test_review_ledger_binding.sh). Observed failure: the succession round of this candidate's own lane raised it as P2 and it fired within the same sitting — the ledger commit made the parent-commit diff empty over the reviewed paths, and the assertion reddened on an output the gate is designed to produce. The shape generalizes past this case: an oracle that admits one of its subject's several valid outputs measures the fixture, not the behaviour, and goes red on a change that is correct. RED-baseline (applied): the assertion reds on the ledger commit before the fix and is green after, with the gate's own behaviour unchanged in both runs. |
519
+ | A base-relative gate compares against the fork point, not the base branch's tip: measured against the tip, every unrelated merge on the target branch restates what this branch is and voids evidence that is still correct | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Observed failure, live and not hypothetical: while this round's own pull request was open, the target branch gained an unrelated merge, and the gate immediately reported that no evidence bound the landing candidate — thirty files appeared changed, all of them somebody else's. Measured against the fork point the candidate was byte-identical to the one the ledger already bound, which is what proved the ledger right and the comparison point wrong. The operational consequence had this shipped unfixed is worth naming: a busy target branch would demand a fresh review lane every time anyone else merged, which is a treadmill rather than a gate. RED-baseline (applied, differential): the suite advances its synthetic base branch with an unrelated commit and asserts the candidate is unchanged; that case reds against the tip-relative gate and is green against the fork-point one, with the rest of the suite unaffected. |
520
+ | A probe that cannot tell "this checkout cannot be measured" from "the thing being measured is broken" is deleted, not patched again: the repository checker runs against synthetic fixtures inside other suites, and a smoke that reds there fails suites that have nothing to do with it | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/check-ccl-skills.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/check-ccl-skills.sh). Supersedes by pointer the row above that added this smoke. Observed failure, three times in one round and each time in a suite that does not own the gate: the catalog suite, then route-drift and sync-pointers together, then the source-register lifecycle suite — every one of them runs the checker against a synthetic repository where the smoke legitimately cannot operate, and the third failure additionally exposed that its capture was not `set -e` safe, so the checker died silently mid-run before printing the verdicts those suites read. Two patches had already narrowed the predicate (skip without a controller, skip without a parent commit) and a third would have narrowed it again, which is the signal this repository already records: when the same class recurs, question whether the capability should exist. It should not. The gate's own suite owns the real-checkout path with a case that runs it against the checkout it ships in, and the CI step is the enforcement point — verified green on this candidate's own pull request. What is lost is stated rather than glossed: nothing else runs the gate during `make test-repo-gates`, so a break in it surfaces at the CI step rather than locally. RED-baseline (applied): the lifecycle suite reds against the smoke-bearing checker and is green after its removal, with the gate's own eighteen-case suite unchanged in both runs. |
521
+ | The merge gate binds every tracked path minus exactly what this round ADDS under a round's evidence directory, and refuses a candidate tree that is not committed: a whitelist binds only the paths some round happened to review, so unreviewed executable content rode along on a valid ledger; a written-down `specs/` exclusion would additionally hide edits to the committed review history itself; and a packet frozen from a dirty working tree produces a hash no clean checkout recomputes | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py, its suite, and .github/workflows/ci.yml). Observed failure: the succession round of the previous round's own lane raised the whitelist half and it was deferred with its reason -- every fix moves the candidate and voids the ledger, and that lane's budget was spent -- so it was recorded for the next round that touches this gate. This round's own review and challenge then each found that the first inversion traded one hole for another, and both land here rather than as further deferrals. The review found that the frozen packet includes untracked files, so a scratch file inside the widened set produces a candidate hash only that working copy can reproduce: the author records it in the ledger and the merge-side run then reports that nothing binds the landing candidate, refusing valid work. The challenge found that excluding all of `specs/` excludes the committed review history, so a pull request could delete or rewrite an earlier round's plan and receipts with no evidence required. The exclusion is therefore computed from the round's own diff rather than written down: only added paths under a round's evidence directory stay outside, because a receipt inside the bound set would move the hash it records, while every modification and deletion under `specs/` is bound like any other file. The succession round then broke that shape too -- an arbitrary added file under an evidence directory, a script included, was excluded for the same reason -- which is the third occurrence of one class: the rule kept naming a LOCATION and letting the location stand in for `this is a receipt`. Rather than narrow the path a fourth time, the predicate moved to an invariant this gate owns: a path is excluded only when its committed blob parses as a JSON object carrying a 64-hex `candidate_sha256`, so a script, a fixture, or an unbound JSON file committed there is bound like anything else. Backward compatibility was measured, not assumed: recomputed at the previous round's own fork point, the old and new path sets produce the identical candidate hash its committed ledger records. Supersedes by pointer the merge-queue half of the row above, which recorded that the CI step now also runs for `merge_group`: the workflow's `on:` never subscribed to that event, so in a merge-queue run the workflow would not start at all and the condition read as coverage while providing none. Restoring the trigger was rejected rather than done, because a merge_group HEAD combines several queued pull requests while each committed ledger binds one individual candidate, so no ledger binds the aggregate and every otherwise-valid queued request would be refused -- the trigger would make the sentence true and the system worse. The unreachable branch is removed and the real coverage boundary is stated where the step lives. RED-baseline (applied, differential, two mutants each attributed to its own cases and nothing else): restoring the whole-subtree exclusion reds exactly the three committed-history cases -- rewriting an earlier plan, rewriting an earlier receipt, deleting an earlier receipt; removing the committed-tree refusal reds exactly the three dirty-tree cases; degenerating the receipt predicate to always-true reds exactly the three smuggled-file cases; three unbound executable paths (a root Makefile, a README, a release script) each red against the original whitelist and are refused after; the added-evidence case stays green throughout, which is what proves the self-reference exclusion survived. Control is 34 passing with no case disturbed. A fourth round then broke the content predicate too -- a JSON file carrying any 64-hex `candidate_sha256` is accepted as a receipt -- and that one is NOT fixed, deliberately. Four shapes of this exclusion have now been broken in four rounds, and every one of them was a proxy for `this is a controller-generated receipt` over a file the candidate itself supplies, which is the already-recorded boundary that a gate living inside the candidate cannot authenticate what it reads. This repository's own standard is that a class recurring across rounds is a question about the design rather than a fifth patch, so the residual is recorded for a person: accept it, or replace the exclusion mechanism outright -- binding the tree as of the commit before the evidence lands would need no exclusion predicate at all. What did close is real: history can no longer be rewritten unnoticed, and a script or binary can no longer ride in under an evidence directory. |
522
+ | A succession may not carry the chain id of the chain it succeeds, and the controller refuses it at mint rather than leaving the refusal to the closeout validator: the validator only sees a lane it reads whole, while the controller mints one receipt at a time, so a caller that never closes a ledger never reaches that check | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md` (entrypoint unchanged this round; lands in scripts/review_gate.py and its suite). Observed failure: the previous round recorded this as a non-blocking deferral with its reason -- fixing it would have moved the controller digest and forfeited that round's ability to close its own ledger with the succession round it introduced. The severity recorded then is the one that holds now, and it is narrower than it first reads: this is not an open bypass, because `validate_extraction_review_state.py` already refuses a succession whose chain id equals the wrapper chain's. What lands is the same refusal at the point the receipt is made, which is the only place it applies to a controller run that never reaches a closeout. The equality direction is not inferred: the existing validator refusal uses the same predicate and the same words, so the intended semantics is that the two ids must differ. RED-baseline (applied, differential): a succession minted with its predecessor's own chain id reds against the pre-fix controller and is refused with its own diagnostic after, with the suite moving from 261 to 262 passing and no pre-existing case disturbed. |
@@ -255,9 +255,9 @@ Rules:
255
255
 
256
256
  The SKILL.md class-wide COMPLETE-set rule (example of a class-wide change: "every stack `*-dev`/`*-architecture` should advertise a localized-refactor trigger"; example of a member-pair landing: Python+Go) fires on change-side sweeps. This section is its join-side dual.
257
257
 
258
- - When a NEW skill joins an already-swept class (e.g. a new stack `*-dev` joining the stack-implementation-owner class), enumerate the class-wide obligations its siblings already carry — advertised triggers, skip-leg reciprocity on routing counterparties, pinned pointer-integrity anchors, and bank fixtures — and inherit each, or record a per-obligation `not-applicable` reason, before the member lands.
258
+ - When a NEW skill joins an already-swept class (e.g. a new stack `*-dev` joining the stack-implementation-owner class), **do not enumerate the inherited obligations from a remembered list of obligation types — derive them mechanically.** Take the file set where an established sibling is named and subtract the file set where the newcomer is named: `comm -23 <(grep -rl '<sibling>' skills/ eval/ | sort) <(grep -rl '<newcomer>' skills/ eval/ | sort)`. Every path in that difference is a candidate obligation site and gets a disposition — `inherit` or `not-applicable: <reason>` — before the member lands. Run it against each established sibling, not just one. **The set difference is the predicate; any type list is only illustration** — the forms it surfaces include advertised triggers, skip-leg reciprocity on routing counterparties, pinned pointer-integrity anchors, bank fixtures, the lifecycle **COORDINATOR's** dispatch enumerations (a router's development-routing list, a technical-design-gate owner table, a problem-routing table), and **body-level structural sections the siblings share** — notably a cross-sibling generalization clause, which is N-way: the newcomer must name every peer *and* every peer must name the newcomer, or the loop breaks in the direction nobody reads. A remembered type list is how the miss recurs: the list that shipped with this section named four routing-surface forms, and the two it omitted — the coordinator's dispatch enumeration and the shared body section — are the ones that went missing on the next join.
259
259
  - A sibling-generalization map that only answers the copied-content question ("does the new skill duplicate sibling text?") must not be counted as obligation-inheritance coverage; the two questions are independent, and the join-side miss survives a clean copied-content map.
260
- - Validation: the closeout map lists each inherited obligation with a status, exactly as the change-side sweep lists each member. Failure shape: a new stack implementation owner landed with "siblings unchanged because the new skill routes to their existing contracts" while lacking the CLI carve-out reciprocity and the localized-refactor trigger pair its siblings carry; only independent review caught it.
260
+ - Validation: the closeout map lists each path from the set difference with a status, exactly as the change-side sweep lists each member. Failure shapes: (a) a new stack implementation owner landed with "siblings unchanged because the new skill routes to their existing contracts" while lacking the CLI carve-out reciprocity and localized-refactor trigger pair its siblings carry; (b) the same owner, one round later, was still absent from the lifecycle coordinator's development-routing enumeration — so multi-stage deliveries in that stack reached the coordinator with no owner to dispatch to, and the coordinator's own fallback clause ("a CLI in a language with no dev owner stays with the default CLI owner") silently reversed the ownership the previous round had just established. Both were found by review or audit, never by the join-side map.
261
261
 
262
262
  ## Capability Naming And Provenance
263
263
 
@@ -56,9 +56,11 @@ If the validator reports `missing_required_command`, keep the failure visible an
56
56
 
57
57
  ## Behavioral Validation
58
58
 
59
- - For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology before writing or finalizing the skill. The code-level RED-GREEN-REFACTOR method (write the failing case first, watch a fresh agent violate the rule WITHOUT the skill, then add the skill and watch it comply) is owned by `superpowers:writing-skills` + `superpowers:test-driven-development` — **if installed, route there; otherwise apply the RED-baseline rule inline** (manually record the without-change failure and the with-change compliance). This is the skill-authoring face of **eval-driven development** (for a behavior/routing change, run the scenario before you finalize; never special-case the scenario just to make it pass) — borrow the *principle*, not a claim of production-grade eval rigor.
59
+ - For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology — **and the firing point is BEFORE drafting the body, not only before finalizing**. Eval-first authoring for a NEW skill (or a new hard-rule section): (1) write the evaluation scenarios first — **at least three** for a new skill (both the vendor's published authoring guide and the high-star practice pack converge on three-plus scenarios before body text; a single-rule edit may scope down to that rule's own scenario); (2) run them WITHOUT the skill and record the observed failures verbatim — and for a discipline-slip failure (the agent knows the rule and skips it under pressure), capture the agent's rationalizations word-for-word: each verbatim excuse is the raw material for one rationalization-vs-reality row and one red-flag line in the skill text (the discipline-slip form in `rule-consolidation.md`'s form-by-failure table); an invented hypothetical excuse does not qualify — counter only what a run actually said, and don't add rows for excuses no run produced; **a no-skill control that does not exhibit the failure is a stop signal — do not author guidance for a failure you cannot observe** (record the null finding instead; this is the pre-draft face of "Evidence must come before new rules"); (3) draft the **minimal** content that addresses the observed failures, then re-run the same scenarios WITH the skill; (4) when a later run, review round, or live miss surfaces a NEW rationalization for an existing discipline gate, add its explicit counter row to that gate's table and re-run the tempting scenario — counter tables accrete from observed excuses across rounds, never from imagination. The code-level RED-GREEN-REFACTOR method (write the failing case first, watch a fresh agent violate the rule WITHOUT the skill, then add the skill and watch it comply) is owned by `superpowers:writing-skills` + `superpowers:test-driven-development` — **if installed, route there; otherwise apply the RED-baseline rule inline** (manually record the without-change failure and the with-change compliance). This is the skill-authoring face of **eval-driven development** (for a behavior/routing change, run the scenario before you finalize; never special-case the scenario just to make it pass) — borrow the *principle*, not a claim of production-grade eval rigor.
60
60
  - **What makes a `RED-baseline` valid is executed-and-recorded vs narrated — not recorded vs live.** Any evidence form (before-after diff, golden trace, or pressure scenario) is valid when it actually records the without-change failure AND the with-change compliance with a locator + expected-vs-actual (see `dual-track-review-gate.md`). Prose that merely *describes* an expected failure without running it is not a baseline; a pressure scenario you actually executed and recorded is.
61
61
  - **A claimed impossibility is falsified in-env before it lands (authoring and reviewing alike; relocated from `SKILL.md`).** Deferring work behind a conservative-sounding stub ("not implemented", "needs a future tested wrapper") is the avoidance form of a blocked-verification claim: it *feels* safe but ships an unverified impossibility as durable behavior — distinct from a real `unavailable`/`pending`-with-remediation+residual-risk record, which is what remains after a safe attempt was genuinely impossible (the attempt itself stays within the safety boundaries the `SKILL.md` rule names). Failure shape: a fallback reviewer lane shipped as a permanent fail-closed stub on an unverified "permission probe not implemented" premise that a ~60-second live tool run refuted, re-enabling the lane.
62
+ - **The falsifying operation has exactly two admissible forms, and a search hit is neither (detail for the `SKILL.md` landed-conclusion rule).** (1) **Executed path** — run the suspected mechanism on the path that actually failed and collect an observation that only THIS cause predicts. Naming the call site is the entry bar, not the operation: **reachability alone rules out non-execution and nothing else** — a wrapper can demonstrably run while a downstream stall is what caused the failure, so "I watched the suspected code execute" clears no cause. The probe must produce a discriminating consequence (the mechanism's own signature in the output, a value only it would set, a timing or state it alone explains); if the run cannot separate the suspect from the alternatives still on the table, it has not falsified anything and form (2) is required. (2) **Single-variable control** — a paired probe whose arms differ in exactly ONE variable. Enumerate EVERY precondition of the predicate under test (each threshold, each uniqueness or vocabulary requirement, each surrounding-state assumption) and confirm both arms equal on all but the varied one; an arm differing in two preconditions yields a verdict attributable to neither, and it fails silently because the verdict still reads as decisive. A shared surface here is the source-register row, the commit or merge-request body, a durable note or memory, and the final response's causal account. The control leg a deferred registration owes (`source-to-skill-extraction.md`) is this rule's narrow instance, and the differential attribution a `RED-baseline` row owes (`dual-track-review-gate.md`) is its encoded form — both assume the arms are otherwise equal, which is the part that has to be enumerated rather than assumed.
63
+ Failure shape: a gate rejected a ledger anchor; a `force_encoding(BINARY)` call found by grep in the gate script was recorded as the cause and landed in a merged PR body, a durable memory note, and a round charter. The suspected line was never on the checked path — that check resolves its blobs through a different helper — and the real cause was an unbounded string replace in the round's own edit, which had modified a pre-existing ledger row. Three paired probes built to test the theory each differed in more than one precondition (anchor length against the threshold, uniqueness within the file, vocabulary membership), so no verdict was attributable; the first of them produced the wrong cause. Withdrawing it cost corrections on three surfaces, against about a minute for the executed-path question.
62
64
  - **A headless code-writing RED/GREEN needs a *fair* violation-tempting scenario + an *independently-valid* objective measure — a capable agent complies on a clean task.** When the rule governs how an agent WRITES code, a clear well-specified prompt usually makes even the no-rule baseline produce compliant code (zero delta → no RED), so a clean-task baseline proves nothing. Surface the RED with **realistic inherited pressure** — a real legacy/house convention the rule must override, or a genuinely ambiguous spec — **NOT an explicit instruction to emit the anti-pattern**: leading the baseline directly into the violation launders a constructed failure into "RED" and is fabrication (the same defect as the rule above), so record why the prompt is fair and mark any direct-leading scenario synthetic/advisory, not RED. Score both runs with an **objective measure that detects the ANTI-PATTERN** — the rule's own checker counts ONLY if it was independently validated first (held-out positive/negative fixtures + a documented residual boundary, so RED genuinely fails); a same-change checker that merely whitelists the expected GREEN form makes GREEN trivially true. Run the agent **cross-model / fresh-context** so the baseline isn't primed by your session. If even the fair tempting baseline complies, that is an honest finding — the rule's marginal value is in edge/legacy cases, not the common one — record it, don't manufacture a RED.
63
65
  - **Optional real-agent RED-baseline (F4 Tier-3).** For a routing-surface or hub-skill change you can execute the baseline through the in-repo F4 Tier-3 harness (`scripts/eval-golden-trace.rb`; contract in `eval-routing.md`) instead of a hand-recorded scenario: run it manually **twice** — first on the checkout WITHOUT your change, then WITH it (the harness does **not** auto-checkout or auto-diff; you run both and preserve both reports as the without→with evidence). Caveats that keep it honest: (a) it is **advisory + non-deterministic + small-N** (a few hub traces, structured *routing* assertions only, statuses PASS/FAIL/INCONCLUSIVE) — a without-change run that returns PASS or INCONCLUSIVE does **not** establish a RED; only an actually-observed miss does; (b) prefer a **pre-existing frozen trace** — a trace you author alongside the change is self-authored evidence (the runner checks `frozen_at_sha` ancestry but does **not** detect a trace added/tuned in the same change), so the reviewer must confirm it is not trivially fail-before/pass-after; (c) it stays **optional and never mandated** — a properly executed-and-recorded pressure scenario remains a valid `RED-baseline` for most changes.
64
66
  - Create at least one pressure scenario: a realistic prompt where the agent should use the skill and avoid a known failure mode.
@@ -1359,6 +1359,36 @@ for required_phrase in \
1359
1359
  done
1360
1360
  echo "test_case_first_gate_ok"
1361
1361
 
1362
+ # Contract-anchor gate: declarative wording-existence pins for load-bearing
1363
+ # prose contracts that structural checks cannot see (verdict-taxonomy
1364
+ # discriminators, stop-condition predicates, externally verified numeric
1365
+ # tiers). Checker and its contract-anchors.tsv table resolve NEXT TO THIS
1366
+ # VALIDATOR (one-versioned-unit rule, same as the sync-pointer gate — an
1367
+ # in-tree copy could be doctored to exit 0). Red = a pinned contract sentence
1368
+ # drifted, was deleted, or became ambiguous; the remedy is printed by the
1369
+ # checker (restore the wording, or update the anchor row in the same MR for an
1370
+ # intentional contract change). Self-proof: test_check_contract_anchors.sh
1371
+ # (fast) and test_pinned_phrase_mutation_walk.sh (heavy).
1372
+ contract_anchor_script="$checker_scripts_dir/check-contract-anchors.sh"
1373
+ if [[ -L "$contract_anchor_script" || ! -f "$contract_anchor_script" ]]; then
1374
+ echo "contract_anchor_infra_failed: check-contract-anchors.sh missing or not a regular file beside the validator: $contract_anchor_script" >&2
1375
+ exit 2
1376
+ fi
1377
+ contract_anchor_rc=0
1378
+ contract_anchor_out="$(bash "$contract_anchor_script" "$root")" || contract_anchor_rc=$?
1379
+ [ -n "$contract_anchor_out" ] && printf '%s\n' "$contract_anchor_out"
1380
+ if [ "$contract_anchor_rc" -eq 1 ]; then
1381
+ echo "contract_anchor_gate_blocking_failed: pinned contract wording drifted (diagnostics above)" >&2
1382
+ exit 1
1383
+ elif [ "$contract_anchor_rc" -ne 0 ]; then
1384
+ echo "contract_anchor_infra_failed: rc=$contract_anchor_rc (contract-anchor gate could not run — fail-closed)" >&2
1385
+ exit 2
1386
+ fi
1387
+ if ! printf '%s\n' "$contract_anchor_out" | grep -qE '^contract_anchor_gate_ok \([0-9]+ anchors\)$'; then
1388
+ echo "contract_anchor_infra_failed: green output grammar missing (expected: contract_anchor_gate_ok (N anchors))" >&2
1389
+ exit 2
1390
+ fi
1391
+
1362
1392
  # Post-cleanup register check: after a project-specific extraction batch is
1363
1393
  # migrated out of the shared skill tree, the source-register.md template must
1364
1394
  # not silently grow new project-specific dated entries. The rule in
@@ -1414,6 +1444,11 @@ if [ -n "$root_worktree_toplevel" ]; then
1414
1444
  fi
1415
1445
  ruby "$impact_chain_gate" "$root"
1416
1446
 
1447
+ # No merge-side ledger-binding probe runs here, deliberately: this checker also
1448
+ # runs against synthetic fixtures inside other suites, where such a probe cannot
1449
+ # operate and reddens suites that do not own it. That gate's own suite covers the
1450
+ # real-checkout path and the CI step enforces it. Do not re-add one here.
1451
+
1417
1452
  # Whole-ledger firing-path resolution. The impact-chain gate above is
1418
1453
  # diff-scoped by design, so a row's firing path is machine-checked exactly once
1419
1454
  # — at the commit that added it. Nothing re-checks it afterwards, while the
@@ -0,0 +1,126 @@
1
+ #!/usr/bin/env bash
2
+ # Contract-anchor gate: declarative wording-existence pinning for load-bearing
3
+ # prose contracts that no structural check would otherwise protect.
4
+ #
5
+ # Problem class (observed, 074 RED-baseline probes): a skill reference's
6
+ # load-bearing contract sentence — a verdict-taxonomy discriminator, a
7
+ # stop-condition predicate, an externally verified numeric tier — can be
8
+ # deleted, semantically inverted, or numerically falsified while every
9
+ # structural gate stays green, because structural gates check shape and
10
+ # references, never the survival of specific contract wording.
11
+ #
12
+ # Mechanism: a sidecar TSV table (contract-anchors.tsv, same directory) lists
13
+ # one anchor per row: id <TAB> repo-relative path <TAB> pinned literal <TAB> note
14
+ # - the literal must occur EXACTLY ONCE in the named file (fixed-string match).
15
+ # 0 occurrences => contract_anchor_missing (drift/deletion) exit 1
16
+ # >1 occurrences => contract_anchor_duplicate (decoy/ambiguity;
17
+ # an anchor that matches twice can no longer prove which copy is the
18
+ # contract — same rule as the sync-pointer registry) exit 1
19
+ # - a missing anchored file is a broken contract, not infrastructure:
20
+ # contract_anchor_file_missing exit 1
21
+ # Table integrity is fail-closed (exit 2, infra): unreadable/empty table
22
+ # (an empty list scanning nothing must never certify), malformed row,
23
+ # duplicate id, or a literal under 16 characters (too weak to be unique —
24
+ # same floor as firing-path anchors).
25
+ #
26
+ # All rows are scanned before any exit: anchor-drift findings (rc 1) and
27
+ # row-integrity findings (rc 2) are each collected across the whole table, so
28
+ # one run reports every problem; when both classes are present the run exits 2
29
+ # (a broken table means no verdict can be trusted). Green output grammar is
30
+ # pinned for the caller:
31
+ # contract_anchor_gate_ok (N anchors)
32
+ # Exemptions: none, and deliberately no environment override — an anchor is
33
+ # removed or reworded only by editing the table in the same MR that changes
34
+ # the pinned wording (reject-and-instruct failure message points there).
35
+ #
36
+ # Intentional-change recipe (printed on failure): edit the contract sentence
37
+ # AND its table row together; the diff then shows the contract change
38
+ # explicitly instead of a silent drift.
39
+ set -u
40
+
41
+ script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)
42
+ root="${1:-.}"
43
+ table="$script_dir/contract-anchors.tsv"
44
+ if [[ "${2:-}" == "--table" && -n "${3:-}" ]]; then
45
+ table="$3"
46
+ fi
47
+
48
+ if [[ ! -d "$root" ]]; then
49
+ echo "contract_anchor_root_missing: $root" >&2
50
+ exit 2
51
+ fi
52
+ if [[ ! -f "$table" ]]; then
53
+ echo "contract_anchor_table_missing: $table" >&2
54
+ exit 2
55
+ fi
56
+
57
+ anchor_count=0
58
+ fail_count=0
59
+ infra_count=0
60
+ # Bash 3.2-safe id-uniqueness set (stock macOS bash has no associative arrays):
61
+ # newline-delimited membership string, ids are TSV fields so they carry no tabs
62
+ # and no newlines.
63
+ seen_ids=$'\n'
64
+ line_no=0
65
+ while IFS= read -r raw || [[ -n "$raw" ]]; do
66
+ line_no=$((line_no + 1))
67
+ [[ -z "$raw" || "$raw" == \#* ]] && continue
68
+ IFS=$'\t' read -r id path literal note <<<"$raw"
69
+ if [[ -z "${id:-}" || -z "${path:-}" || -z "${literal:-}" || -z "${note:-}" ]]; then
70
+ echo "contract_anchor_row_malformed: line $line_no of $table (need id<TAB>path<TAB>literal<TAB>note)" >&2
71
+ infra_count=$((infra_count + 1))
72
+ continue
73
+ fi
74
+ if [[ "$note" == *$'\t'* ]]; then
75
+ echo "contract_anchor_row_malformed: line $line_no of $table (extra tab-separated field)" >&2
76
+ infra_count=$((infra_count + 1))
77
+ continue
78
+ fi
79
+ case "$seen_ids" in
80
+ *$'\n'"$id"$'\n'*)
81
+ echo "contract_anchor_duplicate_id: $id (line $line_no repeats an earlier row's id)" >&2
82
+ infra_count=$((infra_count + 1))
83
+ continue
84
+ ;;
85
+ esac
86
+ seen_ids="$seen_ids$id"$'\n'
87
+ if (( ${#literal} < 16 )); then
88
+ echo "contract_anchor_literal_too_short: $id (${#literal} chars, need >=16)" >&2
89
+ infra_count=$((infra_count + 1))
90
+ continue
91
+ fi
92
+ anchor_count=$((anchor_count + 1))
93
+ target="$root/$path"
94
+ fix_line=" fix: restore the contract wording, or — for an intentional contract change — update this anchor row in $(basename "$table") in the same MR"
95
+ if [[ ! -f "$target" ]]; then
96
+ echo "contract_anchor_file_missing: $id $path" >&2
97
+ echo " fix: restore the anchored file, or — for an intentional move/retirement — update or remove this anchor row in $(basename "$table") in the same MR" >&2
98
+ fail_count=$((fail_count + 1))
99
+ continue
100
+ fi
101
+ occurrences=$(grep -oF -- "$literal" "$target" | wc -l | tr -d '[:space:]')
102
+ if [[ "$occurrences" -eq 0 ]]; then
103
+ echo "contract_anchor_missing: $id in $path" >&2
104
+ echo " pinned literal: $literal" >&2
105
+ echo "$fix_line" >&2
106
+ fail_count=$((fail_count + 1))
107
+ elif [[ "$occurrences" -gt 1 ]]; then
108
+ echo "contract_anchor_duplicate: $id in $path ($occurrences occurrences; the anchor can no longer prove which copy is the contract)" >&2
109
+ echo " fix: keep the pinned wording in exactly one place (dedupe the copy), or repin the anchor row in $(basename "$table") to a longer unique literal in the same MR" >&2
110
+ fail_count=$((fail_count + 1))
111
+ fi
112
+ done < "$table"
113
+
114
+ if [[ "$infra_count" -gt 0 ]]; then
115
+ echo "contract_anchor_table_invalid: $infra_count integrity error(s) in $table (no verdict from a broken table — fail-closed)" >&2
116
+ exit 2
117
+ fi
118
+ if [[ "$anchor_count" -eq 0 ]]; then
119
+ echo "contract_anchor_table_empty: $table has no data rows (an empty anchor set must never certify)" >&2
120
+ exit 2
121
+ fi
122
+ if [[ "$fail_count" -gt 0 ]]; then
123
+ echo "contract_anchor_gate_failed: $fail_count of $anchor_count anchors" >&2
124
+ exit 1
125
+ fi
126
+ echo "contract_anchor_gate_ok ($anchor_count anchors)"