@ccoalm/ccl-skills 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +21 -12
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/references/diagnosis-playbook.md +42 -1
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +7 -4
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-context-freshness.md +4 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-tool-dispatch.md +2 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/inference-capacity-operations.md +20 -1
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/llm-client-gateway.md +11 -0
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +3 -2
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/retrieval-agent-safety.md +5 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +13 -9
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/references/multi-agent-delegation-playbook.md +12 -1
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +14 -41
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +11 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/correction-routing-map.md +22 -0
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/coverage-exhaustion-traps.md +7 -0
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +26 -0
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +2 -2
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +2 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +6 -0
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +4 -2
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +22 -0
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/incident-postmortem-extraction.md +8 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +70 -0
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +11 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +11 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/entrypoint_form_census.py +169 -0
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/reference-access-census.sh +157 -0
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +454 -103
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +8 -0
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_form_census.sh +174 -0
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_reference_access_census.sh +209 -0
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +331 -0
  34. package/dist/assets/release.json +56 -31
  35. package/package.json +1 -1
@@ -177,6 +177,14 @@ When the extraction is incident-driven, the main `extraction-quickstart.md` flow
177
177
  - **Stopping at the first controllable layer** — landing only the test or review rule when the contract / schema / capability owner is the real earliest practical prevention point. Continue the chain; list later layers as secondary controls.
178
178
  - **Re-extracting the same incident twice** — happens when the first extraction landed only the procedural patch. Treat the second pass as a failure-class extraction, supersede the procedural rule, and record why the first pass was insufficient.
179
179
 
180
+ ## Success reviews — the sustain half of learning
181
+
182
+ An incident is one source class; ordinary work that went right is the larger one, and the entrypoint's result-classification rule already requires a `stable success` to carry mechanism, non-luck evidence, reuse conditions, firing point, and owner. Two obligations that rule leaves implicit, each with a primary source:
183
+
184
+ - **Record the success mechanism as work-as-done, not as rule compliance.** Things go right mostly because the agent adjusted its work to the actual conditions, not because a rule was followed to the letter; a `stable success` row must name the adjustment that produced the outcome (which condition was read, what was varied, what was checked) and only then whether an existing rule fired. A row that says "the rule was applied" names compliance, not the mechanism, and cannot transfer when conditions differ. (Hollnagel & Leonhardt, *From Safety-I to Safety-II*, EUROCONTROL 2013: "the reason that things go right is not people behave as they are told to, but that people can adjust their work so that it matches the conditions"; "we cannot make sure things go right just by preventing them from going wrong".)
185
+ - **A sustained practice carries its own re-examination trigger.** Repeated success with a procedure accumulates experience with it and starves the alternative of a fair trial, so a sustain row must name the condition under which the practice is re-tried against an alternative (a changed host capability, a cheaper tool, a second consecutive workaround, a cost row that stops improving) — reuse conditions say where it still holds; the trigger says when to stop assuming it does. (Levitt & March, "Organizational Learning", *Annual Review of Sociology* 14, 1988: competency trap and superstitious learning — the disconfirming observation the classification rule already requires is the guard against the second.)
186
+ - Every retro asks the sustain question next to the improve question, but a sustain row must be written only when stable-success evidence meets the entry bar (mechanism + non-luck evidence + reuse conditions + firing point + owner); otherwise the retro records an explicit `no-new-lesson` or `unstable/insufficient evidence` disposition on the sustain side, never a fabricated success. The pairing has evidence in its own settings: reviewing successes and failures together improved subsequent performance more than reviewing failures alone in a quasi-field experiment on navigation training (Ellis & Davidi, *Journal of Applied Psychology* 90(5), 2005), and the Army's after-action review ends with two lists, sustain and improve (TC 25-20, 1993). The entrypoint's LARGE-session sustain axis and the DO-CONFIRM card's sustain row are the firing points.
187
+
180
188
  ## What gets committed
181
189
 
182
190
  A successful incident extraction usually produces, in one commit or a tight series:
@@ -37,6 +37,7 @@ Two wording rules for whichever form wins:
37
37
 
38
38
  - **No nuance clauses.** "Don't X unless it matters" reopens the negotiation — appending a single nuance clause to a winning recipe degraded it from consistent to noisy in the same tests. Write a real exception as its own conditional on an observable predicate.
39
39
  - **Exemption clauses don't scope.** "This limit doesn't apply to code blocks" still suppresses code blocks; if part of the output must be exempt, restructure the rule so it cannot reach that part.
40
+ - **A whole-file form claim carries a recorded ruler, committed before the edit it will judge.** A claim about a file's form as a whole ("this entrypoint over-uses prohibitions", "the form table is not applied to its own bullets") is a measurement, and a measurement whose method lives only in the round that took it cannot be recomputed: a figure recorded bare is not comparable to the next round's figure, so neither a trend nor an improvement may be claimed from the pair, in either direction. Record the method as a runnable instrument that states its own definitions — what counts as one rule, as a prohibition, as a named baseline failure — and commit it, with its baseline reading, before the edit whose effect it will report. `scripts/entrypoint_form_census.py` is this package's instrument; it encodes no threshold, because a ruler that also scores makes itself the argument for its own reading. Failure shape: a round measured this entrypoint's prohibitive tokens and recorded the count alone; the round that came to act on it counted the same file by its own method, got a different figure, and could not tell a real change from a change of ruler. **The density of imperative-negative vocabulary is not itself a measure of form.** A required-slot rule, a conditional keyed to an observable predicate, and a positive recipe all legitimately carry that vocabulary inside them, so the count names a set to classify one rule at a time, never a set to rewrite — read that way it is the same category error as reading a mention count as an open count. Failure shape: a round inferred from a high prohibitive-token-to-rationale ratio that this entrypoint did not apply its own form table; walking the resulting set rule by rule found the great majority already in a form the table endorses, and the inference was withdrawn rather than acted on.
40
41
 
41
42
  Boundary vs the owning Core Rule's salience mechanism: that rule governs how JOINT requirements are structured (walked enumeration at the firing point; merge-over-append); this table governs the form of a SINGLE rule's text once its landing spot is chosen.
42
43
 
@@ -522,3 +522,73 @@ Round 073-receipt-bundling rows (new table so the entry renders as a table row a
522
522
  | A succession may not carry the chain id of the chain it succeeds, and the controller refuses it at mint rather than leaving the refusal to the closeout validator: the validator only sees a lane it reads whole, while the controller mints one receipt at a time, so a caller that never closes a ledger never reaches that check | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh | `updated` | Owner key `code-review/SKILL.md` (entrypoint unchanged this round; lands in scripts/review_gate.py and its suite). Observed failure: the previous round recorded this as a non-blocking deferral with its reason -- fixing it would have moved the controller digest and forfeited that round's ability to close its own ledger with the succession round it introduced. The severity recorded then is the one that holds now, and it is narrower than it first reads: this is not an open bypass, because `validate_extraction_review_state.py` already refuses a succession whose chain id equals the wrapper chain's. What lands is the same refusal at the point the receipt is made, which is the only place it applies to a controller run that never reaches a closeout. The equality direction is not inferred: the existing validator refusal uses the same predicate and the same words, so the intended semantics is that the two ids must differ. RED-baseline (applied, differential): a succession minted with its predecessor's own chain id reds against the pre-fix controller and is refused with its own diagnostic after, with the suite moving from 261 to 262 passing and no pre-existing case disturbed. |
523
523
  | The merge gate binds a landing candidate larger than one review packet through a committed landing partition manifest: path partitions whose changed files together equal the candidate's exactly once, each recomputed with the controller's own freeze and bound by its own validated ledger, so the reviewer's byte ceiling is no longer the pull request's ceiling | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py, its suite, references/dual-track-review-gate.md, references/extraction-quickstart.md, and a .github/workflows/ci.yml comment). Observed failure: the gate defined the landing candidate as one packet hash, and the controller caps a packet at what one reviewer can read whole, so a candidate larger than that could not be frozen and no ledger could ever bind it -- a release whose whole diff was three times the ceiling had to land as eight separate merges, each splitting the pull request where the review side already permitted splitting the packet. The two identities are different sizes: base..HEAD has no natural byte limit, a reviewer's input does. A manifest committed under a round's evidence directory names the partitions and, for each, the hash that `--print-candidate --paths` already answers; the gate refuses any manifest whose parts do not add up to the whole -- a changed file in no partition, a changed file in two, a partition whose recorded hash no longer reproduces, a base other than the fork point, an aggregate hash that does not reproduce its partitions, or a partition path shaped like a pathspec. The manifest carries a top-level 64-hex `candidate_sha256` (the aggregate identity) so it satisfies the existing receipt predicate and committing it moves no partition; the exclusion predicate is unchanged and the accepted caller-controlled-evidence residual is not widened. `--print-manifest --partition ...` renders the manifest with every hash computed by the gate, so the canonical form lives in one place. Merge-queue aggregation of several pull requests into one HEAD is a different aggregate and stays unsolved, as the workflow comment now states. RED-baseline (applied, differential): the partition cases red 18 against the pre-fix gate with the existing 35 undisturbed; five in-place mutants -- coverage equality, disjointness, aggregate recomputation, base equality, partition-hash recomputation -- each red exactly their own cases (2, 2, 1, 1, 1) and nothing else, restored and verified after each. The candidate that triggered the round, measured at 623,458 bytes, renders as six freezable partitions. |
524
524
  | The partition manifest refuses a wildcard partition path and requires the partition union to EQUAL the reviewed changed set, not merely contain it: git reads `*`, `?`, `[` and `\\` as glob syntax even in a non-magic pathspec, so a wildcard partition chooses its own coverage, and under a narrowed `--paths` scope a partition can reach changed files outside the reviewed set with no uncovered file and no overlap to refuse | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Observed failure: the round's own dual-track lane found both -- the independent review reported that `validate_partition_path` rejected only leading pathspec magic while `lane-*` passed through to git as a glob, and that `partition_coverage` checked only `changed_all - owner`, so with `--paths skills/a` a partition naming `.` covered changed files outside the scope and passed; the adversarial challenge independently hit the wildcard class on the same frozen candidate. Both are the same shape: the manifest was allowed to influence what git enumerated on its behalf. Fix: refuse the metacharacters as a path (never handed to git), and refuse a union larger than the reviewed set with the surplus named. Held until the challenge ran, then applied as one batch that moved the candidate, so the lane owes and runs one succession challenge bound to what lands. RED-baseline (applied, differential): the four wildcard shapes plus the rendering case red against the pre-fix gate and are refused after; the out-of-scope case reds against the pre-fix gate with a freeze error on the oversized `.` partition and is refused before freezing after; disabling the wildcard check reds exactly the five wildcard cases and disabling the equality check reds exactly the out-of-scope case, suite otherwise at 60 passing. |
525
+ | A multi-component failure is localized to one boundary in a single instrumented run — entry/exit data and the received env/config logged at every component boundary, run once, the first wrong boundary owns the search — before any hypothesis fans out across the chain, because locating is the expensive phase | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#must be localized to one boundary in a single instrumented run | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external primary (SRE troubleshooting chapter: simplify and reduce, inject known data at component boundaries, bisection over the component chain) plus an independent installed process pack's multi-component evidence-gathering step, mechanism verified stack-agnostic. RED (measured on 54e0f36): `NO_HITS: component boundar\|entry and exit\|each boundary\|enters and\|exits` across the package while the Instrument step named only "targeted logs"; head carries exactly one list line with the anchor. Baseline is instruction presence, not a behavioral run. Entrypoint grew 4416→4943 body words: every addition is a decision point at its firing step; method detail and verified sources went to the diagnosis playbook reference (Localization Playbook, Probe Ordering, Sources) and the Phase B sanitization re-list was consolidated into a pointer to fund the headroom. |
526
+ | Probe order is decided by discriminating power, then cost, then risk — the cheapest, safest probe whose outcome rules out the most alternatives runs first, likely-and-cheap before exotic — and every system-changing active probe is recorded and reverted before the next observation | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#Probe order must be decided by discriminating power, then cost, then risk | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external primary (SRE troubleshooting chapter, test-and-treat design: mutually exclusive alternatives, decreasing likelihood weighed against risk, side effects of active tests). RED (measured on 54e0f36): `NO_HITS: likelihood\|cheap\|order of\|discriminat\|revert\|pre-test\|restore` across the package — the three-strike rule governed stopping and the falsification rule governed probe validity, nothing governed probe ORDER; head carries exactly one anchored list line. |
527
+ | The hypothesis log is kept inside the diagnosis loop — hypothesis, falsifier, probe cost and side effects, result — so a new hypothesis is checked against recorded observations before it costs a probe and the three-strike count reads from the log instead of memory | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#The hypothesis log must be kept inside the loop | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external primary (Debugging Book introduction, keep-a-log; SRE chapter, take clear notes of ideas, tests and results). RED (measured on 54e0f36): `NO_HITS: audit trail\|running log\|log of`; the only log surfaces were the closeout evidence template and the escalation handoff packet; head carries exactly one anchored list line and the template gained a hypothesis-log field. |
528
+ | A diagnosis licenses a fix only when it explains both causality (how the defect produced this failure on the failing path) and incorrectness (why the code, data, or config is wrong against its contract); a change that removes the failure without the second half is a symptom patch, and an unlinked genuine defect is a different bug that is recorded, never shipped as this cause | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#record it, never ship it as this cause | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external primary (Debugging Book introduction, Checking Diagnoses: fix if and only if the diagnosis shows causality and incorrectness). RED (measured on 54e0f36): `NO_HITS: incorrectness\|why the code is wrong\|why it is wrong\|why the code was wrong`; the reachability-is-not-causation rule covered only the unlinked-defect half; head carries exactly one anchored list line. |
529
+ | A test that passes alone and fails in the suite is bisected over the tests that run before it (halve the preceding set until the polluting test or shared state remains), and the same halving isolates a failing input, config, or dataset when no orderable commit range exists | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#must be bisected over the tests that run before it | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external primary (Debugging Book, reducing failure-inducing inputs — delta debugging) plus an independent installed process pack's polluter-finding script, mechanism verified stack-agnostic. RED (measured on 54e0f36): `NO_HITS: polluter\|order-dependen\|test order`; the package called passes-alone-fails-in-suite a symptom and named no localization move; head carries exactly one anchored list line and the playbook's Localization table carries the recipe. |
530
+ | A hypothesis about a runtime value or state is settled by observing it (breakpoint, print, assertion, trace attribute at the exact point), never by inferring from source what the value must be | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#must be settled by observing it | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: independent agentic-debugging research (agents rewrite conditioned on the error message; interactive tool access improves repair and is under-used) plus the SRE chapter's what/where/why observation discipline. RED (measured on 54e0f36): the 2 hits for `breakpoint` listed the debugger only as an instrumentation option and the code-reading warning fired only on the production telemetry path; head carries exactly one anchored list line covering the local-runtime path. |
531
+ | A wrong value is traced upstream to the first point where a correct input produced a wrong output, and the fix lands at that transition; validation added where the symptom surfaced is defence in depth, never the fix | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#must be traced upstream to the first point where it became wrong | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external primary (Debugging Book introduction, fault propagation from defect to failure) plus an independent installed process pack's backward-tracing reference, ≥2 independent sources. RED (measured on 54e0f36): `NO_HITS: correct.*faulty\|transition`; the Non-Negotiable rule forbade stopping at the wrong line but named no direction of travel; head carries exactly one anchored list line. |
532
+ | A production symptom that cannot be re-triggered in place is not blocked on reproduction: the failing run's own telemetry is the reproduction substitute, suggestive race/deadlock evidence is admitted at its grade, and the cause still owes a falsifying probe before any fix | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#is not blocked on reproduction | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external primary (SRE chapter: some tests are only suggestive; telemetry-first examination). RED (measured on 54e0f36): `NO_HITS: irreproducible\|not reproducible\|cannot be reproduced`; the observability-driven path existed under Instrument but the Reproduce step never routed to it, so an agent could stall at remediation-for-reproduction on a symptom that is diagnosable from telemetry; head carries exactly one anchored list line. |
533
+ | A commit bisection narrows its search by pathspec and by every known-good commit before the first checkout, and a half-finished search is handed off through `git bisect log` / `replay` rather than restarted | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#Do not bisect the whole history | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external primary (`git-scm.com/docs/git-bisect`: cutting down bisection with pathspec and multiple good commits; bisect log and replay). RED (measured on 54e0f36): `NO_HITS: pathspec\|bisect log\|replay`; the two `bisect` hits were the exit-code and `--first-parent` passages; head carries exactly one anchored list line. |
534
+ | In the AI-assisted diagnosis discipline the final cause verdict and the regression test belong to whoever ran the verification commands and read their output — the agent when the agent verified — and a cause proposed by any model, including the diagnosing agent's own analysis, stays a hypothesis until then | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#a cause proposed by any model, including your own analysis, stays a hypothesis until then | `updated` | Owner key `defect-diagnosis/SKILL.md`. W-sweep finding: the base text assigned the verdict to "the human", which for this skill's primary reader (an agent) licensed punting the verdict to the user and contradicted the same block's "YOU verify each candidate" sentence and the repository's autonomy goal. RED (measured on 54e0f36): `grep -c "the human still owns"` = 1 in the package; head = 0, and the replacement line is the anchor. No recorded incident; benchmark-derived. |
535
+ | The persisted-evidence sanitization rule carries its category list by pointer to the external-send rule (Phase A item (d)) plus the one category only it named (config values) instead of re-listing the seven categories inline | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#never pasted into shared diagnosis evidence | `updated` | Owner key `defect-diagnosis/SKILL.md`. Consolidation with a zero-loss obligation map: secrets/tokens, customer data / PII, credential-bearing values, internal hostnames / IPs / URLs / paths, raw SQL and query bodies, request/response bodies, env values, proprietary identifiers all survive verbatim in Phase A item (d); config values survive inline in the consolidated sentence; the incident-store, retention, and link-not-paste obligations of the same bullet are byte-identical. Reviewer to confirm no trigger/scope/routing/validation/acceptance change. |
536
+ | Boundary-walk instrumentation logs only allowlisted, redacted metadata (ids, sizes, status codes, field presence, config keys received) and never raw bodies, headers, secrets, PII, or env/config values; the single instrumented run applies only when the chain can be re-run safely with every boundary reachable, otherwise the telemetry path, partial boundary evidence, or layer narrowing is used with the visibility gap recorded; pathspec bisect narrowing applies only when evidence confines the cause to those paths and falls back to the full range when no reproducing commit is found; the hypothesis log separates the prediction only a cause produces from the falsifier that cannot occur if it is true | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#never raw bodies, headers, secrets, PII, or env/config values | `updated` | Owner key `defect-diagnosis/SKILL.md`. Review-round tightening of this round's own additions: independent review (round 1, codex) found that the first draft licensed raw entry/exit logging before copy-time sanitization, made the pathspec restriction unconditional, made the single instrumented run block the telemetry route, and labelled a discriminating prediction as the falsifier. RED (measured on the round-1 candidate 2317757d…): the boundary bullet contained no redaction predicate and the bisect bullet no fallback predicate; head carries exactly one anchored list line and the playbook table carries separate Prediction and Falsifier columns. Dispositions recorded as `fixed` in the round evidence directory. |
537
+ | The diagnosis playbook's localization table condenses the entrypoint and must never loosen a condition SKILL.md states — the boundary-walk recipe carries the safe-rerun condition, the redacted-metadata restriction, and the telemetry/partial-evidence fallback, the commit-bisection recipe carries the evidence-confined pathspec condition and the full-range rerun, and SKILL.md wins when the two disagree | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/references/diagnosis-playbook.md#a recipe here must never loosen a condition SKILL.md states | `updated` | Owner key `defect-diagnosis/SKILL.md`. Succession challenge (round 3, codex) found the reference table still prescribing raw entry/exit logging and unconditional pathspec narrowing after the entrypoint had been tightened — a mirror drift between entrypoint and reference within one round. RED (measured on candidate 753ad96d…): the two table rows carried none of the entrypoint's conditions; head carries the mirrored conditions in both rows plus one anchored drift-guard list line. Disposition recorded as `fixed` in the round evidence directory. |
538
+ | Probe order is one rule — safety is a filter (reject any probe outside the safety boundary first), then rank by alternatives ruled out per unit of cost, with likelihood and residual risk as tie-breakers; suite bisection keeps the original order and, when neither half fails alone, keeps both halves and reduces by smaller chunks toward a minimal polluting sequence; the upstream trace fixes the correct→faulty transition only when it is owned and changeable and otherwise records the upstream cause and enforces the contract at the nearest owned boundary | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#enforce the contract at the nearest owned boundary | `updated` | Owner key `defect-diagnosis/SKILL.md`. Second wrapper chain (review + challenge, codex) on the fixed candidate: three competing probe orderings in one bullet, a halving loop with no failing half for two-test pollution, and an upstream-trace rule that demanded a fix at an unowned producer. RED (measured on candidate f9da56b1…): the three defects were present verbatim; head carries the single ordering rule, the order-preserving reduction, and the owned-boundary qualification, each mirrored in the playbook table. Dispositions recorded in the round evidence directory; two review findings about the round's own evidence packaging are `accepted_tradeoff` against the repository's recorded evidence-is-caller-controlled and enforcement-inside-the-candidate boundaries. |
539
+ | Suite reduction is order-preserving delta debugging: halve the preceding tests and keep a failing half; when neither half fails alone, remove one chunk at a time and keep the reduced set whenever the failure persists without that chunk, then halve the chunk size and repeat until every remaining chunk is needed (a minimal ordered polluting subsequence) or the shared fixture/state is found — this supersedes the halving-only summary in the preceding round's tightening row, which did not guarantee reduction for a jointly-caused two-test pollution | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#remove one chunk at a time and keep the reduced set | `updated` | Owner key `defect-diagnosis/SKILL.md`. Human-authorized continuation lane (the maintainer authorized further external rounds in this session after the landing lane's final succession returned this finding). RED (measured on candidate cdd133da…): the recipe named halving and finer chunks but no chunk-removal step, so an ordered A+D pollution out of A–D could not reduce; head carries the complement step in the entrypoint and the playbook row. Word budget funded by three gloss trims (bisect script determinism, `exit 125` idiom, `--first-parent`, CI trigger-variant gloss, LLM hallucination sentence) with the conditions, consequences, and actions of each kept. |
540
+ | Suite reduction applies only to a failure that reproduces on every run under a fixed serial order; parallel or intermittent failures keep the failing schedule and validate each kept or dropped subset over repeated runs per the flaky rule, or route to concurrency diagnosis; the playbook's probe-ordering paragraph reproduces the entrypoint's single ordering rule; the consolidated persisted-evidence sanitization rule names credential-bearing values explicitly; the blameless-postmortem sentence cites the SRE chapter's own reason instead of an unsourced empirical claim | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#only for a failure that reproduces on every run under a fixed serial order | `updated` | Owner key `defect-diagnosis/SKILL.md`. Human-authorized continuation lane, wrapper chain (review + challenge, codex) on candidate aa34665b…: the reduction recipe assumed a serial deterministic suite, the playbook probe paragraph still carried the cheapest-and-safest ordering, the pointer consolidation narrowed credential-bearing values to variable values, and one empirical sentence had no primary source. RED (measured on aa34665b…): all four present verbatim; head carries the precondition (entrypoint and playbook row), the mirrored ordering, the explicit category, and the SRE-sourced sentence with its excerpt in the round's attribution evidence. |
541
+ | Fan-out is a walked five-item gate, not a cost note: a genuine constraint must exist, slices are cut by context boundary (never role splits over one feature, never shared state/files/contracts/sequencing), shared implicit decisions are pre-made and carried in every brief, every brief carries an effort budget with width starting small, and the task must be worth the 3–10× single-agent premium | `multi-agent-delegation` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/multi-agent-delegation/SKILL.md#never parallelize work that shares state, files, migrations, contracts, or sequencing | `updated` | Owner key `multi-agent-delegation/SKILL.md`. Benchmark round against public primaries (Anthropic multi-agent research 2025-06 and when-to-multi-agent 2026-01, Google agent-scaling study arXiv 2512.08296, Cognition 2025-06, Claude Code agent-teams docs). RED baseline at origin/dev f156079: `grep -rEil 'scale effort\|effort budget\|effort scal' skills/multi-agent-delegation` → NO_HITS and `grep -rEil 'implicit decision' skills/multi-agent-delegation` → NO_HITS, so a controller following the skill had no rule budgeting worker effort or pre-deciding shared choices, and the four cost sub-bullets duplicated the playbook verbatim (same-facet double write). With change: the gate replaces the duplicated sub-bullets; rationale, sources, and the 15×→3–10× baseline correction live once in the playbook; zero-loss map of the four retired obligations recorded in the round charter. |
542
+ | Delegation execution gains three verification-side rules: large or load-bearing worker outputs go to a durable artifact and the controller verifies from the artifact rather than the relayed summary; a failed return is classified by the multi-agent failure taxonomy (MAST mapping in harness-patterns §2) to pick the fix layer before re-dispatch; integration checks slices for divergent implicit decisions, not only merge conflicts; peer-messaging topology is reserved for workers that must exchange findings, default hub-and-spoke with the controller as validation point | `multi-agent-delegation` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/multi-agent-delegation/SKILL.md#verifies from the artifact, never from the relayed summary | `updated` | Owner key `multi-agent-delegation/SKILL.md`. Sources: Anthropic research-system appendix (subagent output to filesystem to avoid the telephone game; end-state evaluation), MAST arXiv 2503.13657 (14 modes, most failures from system design), Claude Code agent-teams doc (lead must not start implementing while teammates run). RED baseline at f156079: `grep -rEil 'game of telephone\|MAST' skills/multi-agent-delegation` → NO_HITS; the review step verified diffs but had no artifact-not-relay rule and no failure-classification step. observed-failure: no — benchmark-derived, no incident this round. |
543
+ | The sub-agent isolation checklist keeps its internal names but maps them to the literature taxonomy (MAST: three categories, fourteen modes) with a fix-layer column, and the external-source list records the four new primaries (Anthropic when-to-multi-agent 2026-01, Google agent-scaling 2025-12, MAST NeurIPS 2025, Cognition 2025-06) so a later round re-verifies against named sources instead of re-borrowing | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/harness-patterns-and-eval.md#必须先按 MAST 类别定位该修哪一层 | `updated` | Owner key `skill-extraction-workflow/references/harness-patterns-and-eval.md`. Reference-only borrow record; the executable landing is in `multi-agent-delegation` (review step classifies by MAST before re-dispatch). RED baseline at f156079: `grep -rEil 'MAST' skills/skill-extraction-workflow` → NO_HITS. Functional-equivalent check recorded per mode in the round's verdict table (every mode had a scattered counterpart; the missing piece was the classification step and fix-layer routing). |
544
+ | Prompt-cache design precedes miss attribution: static-first ordering with the provider's prefix hierarchy, byte-stable append-only prefix (no volatile tokens, deterministic serialization, no in-place rewrites, mode toggles counted as prefix changes), tool set fixed within a loop with masking over redefinition, provider cache contract respected (minimum length verified from usage fields, breakpoint cap, TTL ordering), cache-read share tracked per route, batch caching treated as best-effort; the tool-dispatch reference gains the turn-boundary rule for surfacing/evicting dynamic tools and the context-freshness reference gains the placement rule (objective and constraints near the end, stable block at the start, distractor-bearing long-context evals) | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/llm-client-gateway.md#never rewrite earlier turns in place | `updated` | Owner key `llm-inference-integration/SKILL.md`. Sources: Claude prompt-caching docs (prefix order tools→system→messages, invalidation table, minimum lengths, four breakpoints, TTL ordering, usage fields), Manus context-engineering (100:1 prefill ratio, append-only, mask-don't-remove), Chroma context-rot and arXiv 2307.03172 (placement). RED baseline at f156079: `grep -rEil 'prefix cach' skills/llm-inference-integration` → NO_HITS and `cache_control\|cache breakpoint` hit only the attribution section — the skill could diagnose a miss but had no rule preventing one. SKILL.md step 3 gains the pointer so the rule fires at gateway design time. |
545
+ | Capacity work uses the standard per-phase vocabulary (TTFT, TPOT/ITL, E2EL) with the averaging caveat, gates rollout and batch tuning on goodput (requests meeting every SLO) rather than raw throughput, carries a serving-lever table that names which metric each hosted-inference lever moves and its caveat (continuous batching, paged KV cache, prefix caching, speculative decoding, prefill/decode disaggregation, quantization, prefix-aware routing), routes latency-insensitive volume to provider batch endpoints under their distinct contract, ramps traffic to avoid acceleration limits, and pins in-flight agent runs to their started version during deploys | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/inference-capacity-operations.md#report tails (p95/p99) per phase, never one blended latency | `updated` | Owner key `llm-inference-integration/SKILL.md`. Sources: BentoML inference handbook (metric definitions, request- vs token-weighted averages), DistServe arXiv 2401.09670 and vLLM disaggregated-prefill doc (TTFT/ITL tuned separately, no throughput gain), vLLM speculative-decoding and prefix-caching docs, NVIDIA inference-optimization blog, K8s Gateway API Inference Extension and AIBrix (prefix/KV-aware routing), Claude batch and rate-limit docs, Anthropic research-system post (rainbow deploys). RED baseline at f156079: `grep -rEil 'goodput\|continuous batch\|speculative decod\|batch API\|Batches API' skills/llm-inference-integration` → NO_HITS; the load checklist used the non-standard phrase 'first-token latency and final-token latency' (W-type vocabulary drift, replaced). |
546
+ | Eval reliability names the judge-bias controls (position swap or randomization with order-consistent verdicts, length-penalizing rubric or normalization, cross-family judge or agreement when the candidate shares the judge's family, per-version human-agreement reporting, judge swap treated as suite migration), the statistical minimum (standard error or confidence interval beside every score, clustered errors for grouped questions, paired differences, power-sized eval sets), and the agent-eval choices (pass@k versus pass^k declared before measuring, end-state or checkpoint grading, saturation graduation, transcript reading before trusting a score) | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/model-prompt-evaluation.md#treat a judge model or prompt swap as an eval-suite migration | `updated` | Owner key `llm-inference-integration/SKILL.md`. Sources: MT-Bench arXiv 2306.05685 (position, verbosity, self-enhancement biases), self-preference bias arXiv 2410.21819, Anthropic statistical-approach-to-evals 2024-11 (SEM, clustered SE, paired differences, power), Anthropic demystifying-evals 2026-01 (graders, pass@k/pass^k, capability vs regression, saturation, transcripts), Anthropic research-system appendix (end-state evaluation). RED baseline at f156079: `grep -rEil 'position bias\|self.preference\|pass\^k\|pass@k\|clustered standard\|power analysis' skills/llm-inference-integration` → NO_HITS; the judge rule said only 'calibrate and watch for drift'. SKILL.md step 5 gains the pointer so the controls fire at eval design time. |
547
+ | Every agent design runs the lethal-trifecta test (private data + untrusted content + external channel) and, when it holds, must remove a capability or impose a named structural injection-defense pattern; each design review walks the OWASP LLM Top 10 (2025) against its owning rule, adding the system-prompt-leakage rule (no secrets, credentials, or authorization logic in the system prompt); SDK building blocks note the guardrail execution-mode choice (parallel guardrails can trip after tools ran) and the error-amplification reason to keep a validating orchestrator on the path to the user | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/retrieval-agent-safety.md#you must remove one capability or impose a structural pattern | `updated` | Owner key `llm-inference-integration/SKILL.md`. Sources: Willison lethal trifecta 2025-06, arXiv 2506.08837 design patterns, OWASP LLM Top 10 2025 list, OpenAI Agents SDK guardrails doc (parallel vs blocking execution), Google agent-scaling study (error amplification independent vs centralized). RED baseline at f156079: `grep -rEil 'trifecta\|OWASP\|excessive agency\|unbounded consumption' skills/llm-inference-integration` → NO_HITS; nine of the ten OWASP entries had owning rules but no enumeration walk reached them and system-prompt leakage had no rule. SKILL.md step 3 gains the trifecta pointer. |
548
+ | Review-round tightening of the benchmark landing: the append-only prefix rule yields to mandatory invalidation (compaction, privacy deletion, revoked authorization, safety or policy updates rewrite the prefix as a new cache generation with a baseline reset); a tool discovered mid-loop takes effect at the next model invocation of the same loop with its schema appended after the last cache breakpoint; the lethal-trifecta pattern must provably cut one edge with a negative test and context minimization counts only when the private data is absent at tool-selection and action time; order-inconsistent pairwise judgments stay in the denominator as ties or abstentions with the inconsistency rate reported; the OWASP supply-chain and poisoning mappings name enforceable checks (inventory, pin, verify, approve, roll back every model, adapter, prompt, tool, skill, and dependency; integrity validation and change monitoring of every authorized source) | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/llm-client-gateway.md#a stable prefix is never a reason to keep revoked | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-1 wrapper chain on candidate d5919e7: independent review (codex) raised four P1 and one P2 on the landed text — append-only conflicting with compaction and privacy deletion, an under-specified turn boundary for discovered tools, a trifecta rule satisfiable by an ineffective pattern, and silent exclusion of order-inconsistent judge pairs — plus one packet-evidence finding dispositioned accepted_tradeoff (gate outputs cannot bind the tree containing them); the same-candidate adversarial challenge raised one P1 on the OWASP mapping. RED baseline is the reviewed candidate itself (the pre-fix wording is in the round-1 and round-2 receipts under the round's evidence directory). All five applied in this commit; the succession challenge binds the post-fix candidate. |
549
+ | Succession-round tightening: the tool-set mutation boundary is stated once and identically in the gateway prompt-cache design and the tool-dispatch dynamic-tool rules — mutation is forbidden only during an in-flight invocation, a tool discovered mid-loop becomes callable at the next model invocation, and because tool definitions head the cached prefix that change is a new cache generation whose miss is accepted only when the tool is genuinely needed, with pre-declared schemas and masked availability as the prefix-stable alternative; the earlier suffix-only re-prefill claim is withdrawn | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/llm-client-gateway.md#Do not mutate the tool set during an in-flight invocation | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-1 succession challenge (codex) on candidate b21e253 found the round-1 tool-boundary fix contradicting the gateway rule (tools head the prefix, so a schema appended mid-loop cannot re-prefill only a suffix); recorded needs_human_decision in the lane-1 ledger, then fixed under the maintainer's continuation authorization (continuation-authorization-1.md). RED baseline is the contradicting wording in the lane-1 succession receipt. |
550
+ | Provider batch-endpoint routing is a per-provider checklist with cited answers (completion window and expiry, result ordering, cancel semantics, billed terminal states, own rate limits, spend-limit overshoot), never a universal contract copied from one provider; and this row supersedes the tool-boundary sentence of the lane-1 tightening row above (the sentence 'a tool discovered mid-loop takes effect at the next model invocation of the same loop with its schema appended after the last cache breakpoint' is withdrawn — tool definitions head the cached prefix, so that change is a new cache generation at the next invocation, as the succession-round tightening row states; ledger rows are append-only, so the withdrawal is recorded here by pointer rather than by editing the earlier row) | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/inference-capacity-operations.md#never assumed from another provider | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-2 (human-authorized continuation) review (codex) on candidate 564824a: the batch-endpoint bullet read as one generic contract (P2) and the lane-1 tightening row still carried the withdrawn suffix-only sentence (P1); the packet-evidence finding is accepted_tradeoff as before; the same-candidate challenge found the plan-then-execute wording fixing only tool choice (P1), so the plan now binds operation, destination, argument fields, and data flow, and the negative test injects destination and payload changes. RED baseline is the reviewed wording in the lane-2 round-1 receipt. |
551
+ | The entrypoint's prompt-cache pointer states the tool-set boundary exactly as the references do (mutated never during an in-flight invocation, between invocations only as a new cache generation) — this row supersedes the phrase 'tool set fixed within a loop' in the prompt-cache design row above, which is withdrawn by pointer because ledger rows are append-only; and the per-provider batch checklist adds create idempotency (client request key or server-side deduplication), lookup-based reconciliation of an ambiguous submission before any retry, and terminal usage reconciliation by item and batch id | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/inference-capacity-operations.md#so a retry never bills a duplicate batch | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-2 succession challenge (codex) on candidate f4ad9df found the entrypoint and the register row lagging the reconciled boundary (P1) and the batch checklist silent on create idempotency and post-ambiguity reconciliation (P1); fixed in lane 3 under the maintainer's continuation authorization (continuation-authorization-2.md). RED baseline is the reviewed wording in the lane-2 succession receipt. |
552
+ | Cache-usage arithmetic runs only on provider-adapter-normalized disjoint counters (cache-read, cache-creation, uncached remainder derived by subtraction where a provider's total is inclusive), with an absent cache field recorded as unknown rather than zero; and the batch checklist requires an explicit capability decision when a provider offers neither idempotent creation nor an authoritative lookup key (forbid automatic retry of an ambiguous submission with a persisted submission_unknown state, or decline the endpoint), stable item and submission identifiers persisted before sending | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/llm-client-gateway.md#recorded as unknown, never as zero or as not cached | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-3 (human-authorized continuation) review and same-candidate challenge (codex) on candidate 04d8916: the usage equation double-counted cache reads for providers whose input total is inclusive and misread absent fields as not cached (P1, both rounds); the batch checklist had no path when lookup-before-retry cannot run (P1); the packet-evidence finding is accepted_tradeoff as before. RED baseline is the reviewed wording in the lane-3 round-1 receipt. |
553
+ | History-preserving eviction yields to mandatory invalidation: the dynamic-tool rule never to evict a tool the loop's history cites holds only on capacity or recency grounds, while a revoked authorization or a privacy, safety, or policy update removes the tool's definition and dependent prompt material, resets the cache generation, and restarts the loop or fails closed if the retained history cannot stay valid without it | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/agent-tool-dispatch.md#reset the cache generation, and restart the loop or fail closed | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-3 succession challenge (codex) on candidate 3a44680 found the unqualified never-evict rule retaining a revoked tool against the gateway override (P1); fixed in lane 4 under the maintainer's standing instruction (continuation-authorization-3.md). RED baseline is the reviewed wording in the lane-3 succession receipt. |
554
+ | Every model invocation and tool call is bound to the authorization/tool generation it was issued under, and a completion's tool calls are re-authorized against the current generation before any side effect (stale-generation calls rejected) so a revocation during an in-flight invocation cannot execute through it | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/agent-tool-dispatch.md#a call issued under a stale generation is rejected, never executed | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-4 same-candidate challenge (codex) on candidate f0f3b71: an in-flight invocation could return a call to a tool revoked mid-invocation (P1); the review plan's register-row count was stale (P1, fixed in the caller-held plan's evidence rows; the acceptance sentence is scope-bound and superseded by the evidence row). RED baseline is the reviewed wording in the lane-4 round-1 receipt. |
555
+ | The delegation fan-out gate forbids parallel work that shares state, files, migrations, contracts, or sequencing through writes, while parallel read-only use of one artifact (independent investigations; review plus challenge over one diff) stays allowed when the outputs are independent | `multi-agent-delegation` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/multi-agent-delegation/SKILL.md#shared-write work runs sequentially or stays local | `updated` | Owner key `multi-agent-delegation/SKILL.md`. Lane-4 review (codex) on candidate f0f3b71: gate item 2's shared-files prohibition contradicted the read-only fan-out the playbook and execution flow permit (P1); fixed with zero-loss trims elsewhere so the entrypoint stays at 4988 body words. RED baseline is the reviewed wording in the lane-4 round-1 receipt. |
556
+ | A revocation, narrowing, or policy invalidation advances the authorization/tool generation atomically before any in-flight completion is accepted, and a completion's tool calls are re-authorized against current policy on the exact operation, destination, arguments, and data scope before any side effect, so the stale-generation rejection can always fire | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/agent-tool-dispatch.md#advances the authorization/tool generation atomically | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-4 succession challenge (codex) found the generation-binding clause never requiring the generation to advance on revocation, so an issued call's generation could equal the current one and the rejection never fire (P1); fixed in lane 5 under the maintainer's standing instruction (continuation-authorization-4.md). RED baseline is the reviewed wording in the lane-4 succession receipt. |
557
+ | Side-effect admission shares the revocation's serialization boundary: advancing the authorization/tool generation fences or cancels calls accepted but not yet started, and only a call admitted under an unchanged generation enters the irreversible handler; and pairwise judging runs both orders for every pair before an inconsistency rate is claimed, a single randomized order per pair permitting only an aggregate position-effect analysis | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/agent-tool-dispatch.md#only a call admitted under an unchanged generation enters the irreversible handler | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-5 review and same-candidate challenge (codex) on candidate 56342b9: the judge rule allowed a single randomized order yet demanded a per-pair inconsistency rate (P1), and re-authorization was not serialized with a concurrent revocation before the irreversible handler (P1); the packet-evidence finding is accepted_tradeoff as before. RED baseline is the reviewed wording in the lane-5 receipts. |
558
+ | Fan-out width is counted in workers that each own one bounded slice — tightly related items may sit inside that one slice — and every brief must carry an effort budget, so a width rule can never produce multi-task workers whose ownership, deadline, and failed-return classification cannot be attributed to one bounded unit | `multi-agent-delegation` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/multi-agent-delegation/SKILL.md#each owning one bounded slice | `updated` | Owner key `multi-agent-delegation/SKILL.md`. Lane-5 review (codex) on candidate 56342b9: gate item 4's 'several tasks each' contradicted execution step 3's one bounded task per agent (P1); fixed with zero-loss trims so the entrypoint stays within the 5000-word gate. RED baseline is the reviewed wording in the lane-5 round-1 receipt. |
559
+ | Revocation inside a tool-bearing loop is stated as one fencing invariant rather than accumulated ordering patches: the generation advance and the call's final generation check at the irreversible handler's commit boundary are serialized by the same lock or fence, a call rechecks immediately before crossing that boundary, and the lease and fencing-token mechanics already required for stale agents are reused rather than re-derived | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/agent-tool-dispatch.md#revocation is a fencing problem, not a prompt problem | `updated` | Owner key `llm-inference-integration/SKILL.md`. Four consecutive same-class findings (lanes 1, 3, 4, 5: append-only vs invalidation, eviction vs invalidation, in-flight revocation, admission ordering) showed the class was a concurrency protocol being specified one patch at a time; the lane-5 succession's 'started is ambiguous' finding (P1) is fixed by the invariant form and by routing the mechanics to the existing fencing-token rule, per the same-class-recurrence design rule. RED baseline is the reviewed wording in the lane-5 succession receipt. |
560
+ | The code-then-execute pattern cuts the trifecta edge only when the privileged code, its allowed sinks, and its permitted data flows are generated and frozen before any untrusted content is read, untrusted input entering afterwards only as non-instruction typed data with the negative test restoring that ordering; and the serving-lever table states prefill/decode disaggregation's throughput effect as engine- and workload-dependent to be measured on the target engine, not as a categorical claim | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/retrieval-agent-safety.md#generated and frozen before any untrusted content is read | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-6 review (codex) on candidate 09a040c: the code-then-execute option did not require code and sinks to be frozen before exposure (P1) and the disaggregation caveat was categorical (P2); the packet-evidence finding is accepted_tradeoff as before; the missing lane-5 authorization record and the authorization-chain wording were corrected in the evidence directory. RED baseline is the reviewed wording in the lane-6 round-1 receipt. |
561
+ | The merge gate binds an integration branch that accumulated several already-bound rounds and is promoted as one pull request by walking HEAD's first-parent chain down to the first commit already on the target: each round merge is rebound by the same gate in a detached checkout of its second parent against its first parent with that checkout's own controller and validator, a merge whose second parent is already on the target is a sync merge that owes nothing, every step must equal the automatic merge of its parents, a non-merge commit on the chain is refused, and the chain is consulted only after the single-ledger and manifest paths and only for the default path set | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py, its suite, references/dual-track-review-gate.md, references/extraction-quickstart.md, and a .github/workflows/ci.yml comment). Observed failure: a promotion of three stacked rounds could be bound neither by one ledger nor by a path partition, because two of the rounds appended to this register and no round ever froze the sum of both appends, so the release had to land as three separate pull requests each pointing at one round's merge commit. Every such round had already been bound at its own base when it merged; the gate simply discarded that evidence at promotion time. Chain cases were written first and observed failing on the previous gate (10 failing), then green; a mutation walk reds each load-bearing predicate exactly on its own cases; the real three-round promotion binds as three rounds. Merge-queue aggregation of unmerged pull requests remains unsolved and is now stated as such in the ci.yml comment. |
562
+ | A historical round on the first-parent chain is judged with the landing tree's controller and validator, never the round's own, and the detached checkout it is rebound in must be verifiably released (removal result checked, registry read back, no repository-wide prune) or the verdict is an error; the walk accepts a chain of exactly the bound's length and refuses one longer — superseding the previous row's clause that a round is rebound with its own checkout's tools | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py, its suite, and references/dual-track-review-gate.md). The round's own dual-track lane found all three: the adversarial challenge (P1) showed a round could install a hollow validator in its own branch, forge a ledger the real validator rejects, and stay bound after a later round restored the real validator, because history was being judged with history's tools; the independent review (P1) showed the checkout release discarded the removal result and ran a repository-wide prune that also drops registrations the run did not create; both lanes found the off-by-one that refused a chain of exactly the bound's length (P2). Fixes held until after the challenge; three cases were written first and observed failing on the pre-fix gate, then green. RED baseline is the pre-fix gate accepting the forged round, refusing the 64-step chain, and reporting ok past a failed release. |
563
+ | A detached checkout whose creation fails after git registered or populated it is released through the same verified path as a successful one, so a failed rebind leaves no repository worktree state behind and the error names any release problem alongside the creation failure | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). The round's succession challenge left this open at the exhausted lane budget: `git worktree add --detach` can return nonzero after creating its registration, and the failure branch only deleted the directory. Closed in an authorized continuation lane: the case (a git shim that runs the real add and then fails) was written first and observed failing on the pre-fix gate, then green. RED baseline is the pre-fix gate leaving the registration behind. |
564
+ | Releasing a detached checkout distinguishes a registration that survives removal (an error) from an unregistered directory this run created before git registered anything (deleted by the run itself, reported only if it survives), so a creation that fails before registering errors cleanly without a release complaint | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Found while pre-covering the previous row's fix before review: routing a failed creation through the verified release path made an add that never registered anything report a spurious release problem and leave its directory behind. The benign before-registration sibling case sits beside the registered-failure case as the precision row; the registered-failure case is the RED baseline for the class and this row's case pins that the benign shape neither complains nor leaks. |
565
+ | The worktree registry is read back NUL-terminated when a detached checkout is released, because line-oriented porcelain cannot carry every path the gate may compare against — a newline splits the line, and quoting of other unusual bytes depends on git version and core.quotePath — so a surviving registration under such a temporary directory would match neither spelling and read as gone | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). The continuation lane's adversarial challenge (P2) named a tab-bearing TMPDIR; on the git in use a tab is printed raw and did not reproduce, while a newline-bearing path did split the line, so the case uses a newline: an add that registers and returns nonzero plus a refused remove left the registration in place while the release reported clean. Fixed by reading `git worktree list --porcelain -z` and comparing exact NUL-delimited fields; the case was written first and observed failing on the pre-fix gate, then green. RED baseline is the pre-fix gate reporting no release problem with the registration surviving. |
566
+ | Retirement or relocation of an over-budget entrypoint or reference cites the host-local usage census (per-file session counts derived from the agent's own transcripts, counts only) instead of the author's opinion, and the census must report `unevaluated` rather than a zero table whenever it could not evaluate | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_reference_access_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: external primaries — per-bullet helpful/harmful usage counters in itemized-context evolution (ACE, arXiv 2510.04618 §3.1–3.2), root-vs-auxiliary placement by frequency (Trace2Skill, arXiv 2603.25158 §2.1), and the official skill-authoring guidance that a bundled file the agent never accesses is unnecessary or poorly signaled. Observed failure this round: the first census version printed `sessions_touching_package=0` on a 60-day window (the candidate list exceeded one exec batch and the batch error was swallowed) while a 14-day window on the same host counted touching sessions normally — an instrument that reads zero when it could not evaluate is the false-green shape ratchet invariant 2 forbids. Fixed by batching; the regression test's ARG_MAX leg (450 synthetic transcripts, 3 touching one reference) and its unevaluated-sentinel and privacy-contract legs are the RED-to-GREEN evidence. RED (measured on fa0a7de): `NO_HITS: never accessed\|ignored content\|access log\|usage count` across SKILL.md and references (ledger excluded). Supporting paths: scripts/reference-access-census.sh, scripts/test_check_ccl_regressions.sh (fast lane registration), references/attention-budget-ratchet.md §Retirement and relocation signal, references/extraction-quickstart.md tool inventory. Host census figures stay in the private charter. |
567
+ | Low-frequency entrypoint detail relocates verbatim into the reference the entrypoint already points at (UI/UX judgment obligations, correction-type routing for test/verifier/deferred-evidence corrections, read-modify-write example-code obligations), the entrypoint keeps a one-bullet summary carrying the load-bearing obligations and every check-ccl phrase pin, and the entrypoint's body must shrink, never grow, in a round that touches it while it is over budget | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#never close a judgment-delta row whose visual direction/tokens fields were not inspected | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: the entrypoint that enforces a 5,000-body-word cap on every other skill carried 16,748 body words at base (size gate `entrypoint_size_severe_debt`, 3.3× the cap; 90 top-level bullets, 12 over 300 words) — the placement rule it states was violated by its own file, and Codex consumers cannot read it in one pass. RED (measured on fa0a7de): `check-size-budget.sh` base_body_words=16748; GREEN at head: head_body_words=16426, `changed_entrypoint_word_delta … delta_body_words=-322`, `entrypoint_size_blocking_ok`. Zero-loss obligation map: six UI/UX bullets → references/uiux-judgment-extraction.md §Entrypoint obligations (byte-identical), DFE-CONT + tests-before-test-cases rules → references/correction-routing-map.md (byte-identical, new file linked from Reference Loading), five RMW obligations → references/validation-and-landing.md §Read-modify-write example code (byte-identical); the four `skill_extraction_test_case_first_gate` phrases stay inline in the summary bullet (the first draft dropped one and the gate went RED, then GREEN after restoring it); the five ledger `file:` anchors and eight script pins into SKILL.md were enumerated first and none sits in the moved text (`register_firing_path_resolution_ok`). Remaining debt after this batch: 16,426 body words; the census and cost row exist so the next batch is chosen from evidence. |
568
+ | Content placement sorts by firing frequency as well as by kind: a rule that fires only on a narrow source class or correction type is low-frequency detail even when it is non-negotiable and must live verbatim in the reference the entrypoint already points at, behind a one-bullet summary that keeps the load-bearing obligations inline; the census is advisory and never a gate | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#must live verbatim in the reference the entrypoint already points at | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: Trace2Skill §2.1 (root document stores broadly applicable procedural knowledge, auxiliary files provide lower-frequency details) and §2.4/App. B (joint consolidation over many traces beats order-dependent one-lesson-at-a-time editing — the regime this skill's per-round appends have been in); Anthropic skill-creator progressive disclosure and the best-practices navigation observations (overreliance → move into the entrypoint; ignored → remove); Goodhart for the never-a-gate clause (same anchor as the health roll-up). RED (measured on fa0a7de): `NO_HITS: lower-frequency\|low-frequency\|firing frequency` in SKILL.md and references (ledger excluded); the placement rule sorted by kind only, so every non-negotiable read as entrypoint material. Head carries exactly one anchored list line. |
569
+ | A skill-effect comparison defines its baseline arm by change type (no-skill for a new skill, a frozen pre-change snapshot for an existing one), starts both arms in the same turn, records tokens and duration per arm, and reads results per assertion (pass-both, fail-both, one-sided, high-variance) before trusting an aggregate pass rate | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/harness-patterns-and-eval.md#the with-skill and baseline arms must start in the same turn | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: Anthropic skill-creator SKILL.md (head 2026-04-20; Step 1 spawn with-skill and baseline runs in the same turn with the snapshot rule for existing skills, Step 3 capture `total_tokens`/`duration_ms`, Step 4 analyst pass) and agents/analyzer.md §Analyzing Benchmark Results step 2 (per-assertion patterns). RED (measured on fa0a7de): `NO_HITS: non-discriminating\|passes in both\|without-skill\|duration_ms\|total_tokens` in harness and validation references; §3.1 compared before/after arms only, had no cost column, and read only aggregate deltas. Supporting path: references/external-practice-controls.md gains the IFScale row (density degradation, primacy bias; evidence grade recorded), which is a table row and carries no anchor. Concept-adjacent coverage is not functional equivalence; that bar makes this row I→updated, not P. |
570
+ | Every non-wording closeout records a cost row (review and challenge rounds run, findings fixed, accepted, or deferred, wall-clock from charter to PR, and net body-word delta per touched entrypoint) and the next round must read it before choosing its batch shape | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/extraction-quickstart.md#must read this row before deciding its batch shape | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: Google SRE Postmortem Culture (regularly survey whether writing a postmortem entails too much toil and improve the process from the answers) and the skill-creator per-run cost capture. RED (measured on fa0a7de): `NO_HITS: toil\|wall-clock` in extraction-quickstart.md; the closeout template recorded final state and lessons only, so process cost (one recent round ran 18 review rounds across 7 lanes; another re-ran full verification 8 times) lived only in private notes and never fed the next round's batch decision. Head carries exactly one anchored list line. |
571
+ | The census CLI treats a flag without its value as a usage error (exit 2, stderr) rather than an unbound-variable abort, and its regression suite's large-window leg must exceed every common ARG_MAX so that the first version's single-exec expansion cannot pass it | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_reference_access_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Self-review under the terminal owner and the testing owner (controller-derived owners for the new CLI and its test) found two defects before external review: a flag without a value aborted under `set -u` with exit 1, and the ARG_MAX leg used 450 short paths (~54 KB), which no ARG_MAX rejects — a mutation never applied, so a hypothesis. Observed RED: with 11,000 transcripts of ~230-character paths (~2.5 MiB) the exact first-version mutation applied to a disposable copy failed the leg with `sessions_touching_package=0` (mutant rc=1); the restored suite is GREEN in 3.9 s. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh. |
572
+ | The census withholds its table on any input error (unreadable, vanished, or unexecutable transcript inputs: exit 2 with an `unevaluated` count-only message) instead of printing zeros or the ok token, and every sentinel and usage error is path-free — no log root, caller path, or host detail reaches stdout or stderr | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#withholds the table (exit 2 on input errors) | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Round-1 review (codex, candidate c2c4d377) findings 0 and 1 and the round-2 challenge finding 0: the empty-log sentinel interpolated the log-root path (an absolute private path) and the test asserted that leak; find/grep/xargs errors were collapsed by `\|\| true` into zero counts followed by the ok token. Fixed: path-free sentinels (test asserts the log root and any absolute path are absent), a stderr-collecting error log that withholds the table with exit 2 (new test leg: a chmod-000 transcript → exit 2, `1 input error(s)`, no table, no ok token, no path), and the ratchet bullet now states the contract. RED is the reviewed candidate; GREEN is `test_reference_access_census_ok` on the fixed tree. Correction to two earlier rows of this round: their `NO_HITS` evidence for the terms `passes in both` and `wall-clock` overstated the grep — the base has one unrelated hit each (`passes in both directions of the wrong answer` in rule-consolidation.md; `wall-clock deadline` in the harness MAST table); the mechanisms had no prior carrier, which is the consolidation question those rows answer; recorded here because rows are append-only. Review findings 2 (ledger over 500 lines) and 3 (two rows sharing a command locator) are `accepted_tradeoff` with evidence in the round's disposition files. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh, specs/111-extraction-skill-benchmark/evidence/primary-source-excerpts.md (finding 4, fixed). |
573
+ | The census treats an explicitly supplied log root that does not exist as an input error (an absent default root stays normal), never echoes a caller argument in a usage error, and guards every stat substitution so an input error always reaches the withholding path instead of aborting under set -e | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#must be treated as an input error | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Continuation lane (scope: the defects named in the preceding closeout). Observed failures: succession challenge (codex, candidate 282c6e14) P2 ×2 and the mis-packeted first succession attempt's P1 (explicit missing root skipped silently). Each new test leg went RED under its applied mutation on a disposable copy (explicit-root check removed → leg 8 red; argument echo restored → leg 7 red; stat guard removed → leg 9 red with the pipeline status) and GREEN on the fixed tree. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh. |
574
+ | From the second review round on, the packet's exclusion list must exclude every evidence JSON the round has already added — receipts, dispositions, closeout files — so that the reviewed candidate equals the candidate the merge-side binder computes, and bound evidence (base attestations, excerpt files) is committed before the round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/extraction-quickstart.md#must exclude every evidence JSON the round has already added | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed twice in consecutive rounds: the previous round's private notes recorded three receipts binding three candidates, and this round's first succession attempt excluded only the two receipts and bound candidate 0b91f263 while the binder computed 282c6e14 (`review_ledger_binding.py --print-candidate` before and after; the mis-packeted receipt is kept as history in the evidence directory). RED is that receipt; GREEN is the retry with the eight-file exclusion set whose receipt binds 282c6e14 and the validator's `extraction_review_state_ok`. |
575
+ | A stable-success row records its mechanism as work-as-done (the adjustment that matched the actual conditions) before any rule-fired claim, and a sustained practice names the condition under which it is re-tried against an alternative | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/incident-postmortem-extraction.md#Record the success mechanism as work-as-done | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: external primaries read this round to check learning-from-success theory — Hollnagel & Leonhardt 2013 (Safety-II white paper, PDF read: performance adjustment as the reason things go right), Levitt & March 1988 (competency trap, superstitious learning), Ellis & Davidi 2005 (successes+failures review beats failures-only, PubMed abstract), TC 25-20 (AAR sustain/improve lists, PDF read). Functional-equivalent check at fa0a7de: the classification rule's mechanism / non-luck / reuse-conditions / disconfirming-observation fields and the LARGE-session sustain axis exist (P for Ellis & Davidi, AAR, superstitious learning); `NO_HITS: work-as-done\|competency trap\|re-examination trigger` — the compliance-vs-adjustment distinction and the re-exploration trigger had no carrier. Head carries exactly one anchored list line; entrypoint gains a 20-word pointer while its round delta stays negative. |
576
+ | A description trigger evaluation grades each bank utterance at least three times per description version, keeps negatives as near-misses that share vocabulary with the skill, and selects an auto-rewritten description on a held-out split, never on the training split | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#each bank utterance must be graded at least three times per description version | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: Anthropic skill-creator Description Optimization (20 realistic queries, near-miss negatives, 3 runs per query, 60/40 train/held-out, best by test score). Functional-equivalent check at fa0a7de: the bank already carries `expected_skill: none` sentinels and `must_not_route_to` decoys and the runner has `--replicas` (P for negatives and repetition capability); `NO_HITS: held-out\|train/test\|overfit` — repetition was optional and no held-out discipline existed. Landed because an identified gap is a fix item, not a record. |
577
+ | The census keeps a NUL-delimited, pathname-deduplicated transcript inventory so overlapping or repeated log roots and newline-bearing names count a session once, tolerates an unset HOME (no default roots, never an unbound-variable abort), never echoes an invalid `--skill` value, and adds a shim-forced transcript-read error leg that does not depend on file permissions; the success-review reference requires a sustain row only when stable-success evidence meets the classification entry bar | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#overlapping or repeated log roots must count a transcript once | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Continuation lane (chain r3, codex): review finding 2 and challenge finding 0 (overlap double-count, newline-split names), review finding 3 (`--skill` echoed), challenge finding 1 (HOME unset abort; root-dependent unreadable leg), challenge finding 2 (the success-review pairing sentence universalized a sustain row beyond the entry bar — reworded to require `no-new-lesson`/`unstable` otherwise). Applied mutations on disposable copies: dedupe removed → leg 10 RED; `--skill` echo restored → leg 11 RED; HOME guard removed → leg 13 RED; restored suite GREEN in 3.6 s. The `seq` portability finding is refuted on this host (`/usr/bin/seq` present on macOS) and recorded as such in its disposition. Review findings 0/1 (ledger over 500 lines; command locators) repeat the earlier accepted tradeoffs; challenge finding 3 (packet evidence) repeats the packet-evidence class. |
578
+ | The census maps grep's no-match status to success inside each scan batch and treats any other batch failure — with or without a stderr line — as an input error that withholds the table, and its ARG_MAX regression fixture builds its padding without `seq` and asserts the inventory exceeds the largest common ARG_MAX before the mutation-sensitive check | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#a scan batch that fails without writing stderr must still withhold the table | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Second continuation lane (scope: the defects named in the preceding closeout). Observed failures: continuation-lane succession (codex, candidate c949ad35) findings 1 and 2. Applied mutation on a disposable copy: restoring the plain `xargs grep` pipeline makes the new silent-exit-2 shim leg (15) fail with a table and the ok token; the fixture-size assertion is checked at run time (inventory > 2 MiB). Restored suite GREEN in 6.5 s. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh. |
579
+ | The census batch wrapper survives an inherited errexit (a caller-exported SHELLOPTS=errexit no longer turns an all-no-match batch into an input error), and the ARG_MAX regression leg proves its fixture exceeds the running host's exec limit by executing the single-exec shape and requiring E2BIG instead of trusting a fixed byte figure | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#must not misreport an all-no-match batch under an inherited errexit | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Third continuation lane (scope: the defects named in the preceding closeout). Observed failures: second continuation lane's challenge (codex, candidate ef0010e3) P2 findings 1 and 2. Applied mutations on disposable copies: the errexit-unsafe wrapper (`grep; rc=$?`) fails leg 16 under `env SHELLOPTS=errexit`; restoring the single-exec `$(cat …)` scan makes the fixture leg fail with E2BIG on this host (rc 126). Restored suite GREEN in 7.8 s. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh. |
580
+ | The entrypoint's relocation summary bullets keep every load-bearing obligation of the moved text inline (mini-program owner split, execution-only disclosure, the pending/out-of-scope disposition for uninspected token fields, the DFE-CONT validation and rename-sync clauses); shared-tree evidence carries neutral scope facts only, never a conversation transcript or adjudication narrative; and a transcript listing that fails without stderr is an input error that withholds the table | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#do not claim design-judgment extraction | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Third continuation lane (chain r6, codex, candidate 420ab84e): review findings 0 (summary bullets dropped obligations), 3 and the challenge finding (three evidence notes and three ledger rows carried conversation-level content on a shared surface — removed and reworded to scope facts), 4 (silent find failure). Applied mutation on a disposable copy: ignoring find's status makes the new leg 17 fail with exit 0 and the no-transcript sentinel where an input error is required; restored suite GREEN in 5.4 s. The entrypoint grows by the retained obligations (+80 body words) while its round delta stays negative. |
581
+ | The ARG_MAX regression leg derives its fixture size from the running host's exec limit (getconf ARG_MAX) and skips its E2BIG probe with a printed note when the limit exceeds the fixture bound, never asserting a fixed byte figure; the entrypoint summary bullets carry the relocated observable-proxy checklist, breakpoint scope, the deferred-evidence trigger set, and the bare-mention qualifier inline; and shared-tree evidence and ledger rows state scope facts without host census figures or conversation-level narrative | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#a bare mention of runtime or external access is not enough | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failures: the CI fast lane on the Linux runner failed the previous E2BIG probe (2.5 MB inventory below that host's exec limit while it exceeds macOS's 1 MiB) — the host-derived-threshold class the prior succession named; the prior succession's summary-bullet and shared-surface findings. Applied mutation on a disposable copy: restoring the single-exec scan fails the leg on this host (ARG_MAX 1 MiB, E2BIG); restored suite GREEN in 5.6 s. Rows citing host figures or narrative were corrected at their origin commits (the branch is unmerged; rows are never edited in place). Supporting paths: scripts/test_reference_access_census.sh, specs/111-extraction-skill-benchmark/evidence/continuation-lanes.md, specs/111-extraction-skill-benchmark/evidence/primary-source-excerpts.md. |
582
+ | The census last-touched column is the newest touching transcript's date and the counting leg asserts that exact date (and the older date on the other file) from deterministic mtimes, so selecting the oldest timestamp or emitting an unparsed value fails the suite | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#the last-touched column must be the newest touching transcript's date | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Fourth continuation lane review (codex) P2: successful-row assertions accepted any non-empty date. Fixed with `touch -t` mtimes on two transcripts and exact `alpha.md \| 2 \| <newest> \| 66%` / `beta.md \| 1 \| <older> \| 33%` assertions. Applied mutation on a disposable copy: selecting the oldest date (`sort \| head -1`) fails the leg; restored suite GREEN in 5.9 s. Supporting paths: scripts/test_reference_access_census.sh. |
583
+ | A transcript older than the census window moves no count, share, or last-touched date, and the suite proves it with a stale fixture: removing the mtime filter turns the denominator assertion red | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_reference_access_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Registered gap from the previous round's terminal succession (no test leg for out-of-window transcripts), reproduced against the current script with a control leg: with the filter present the stale fixture leaves every existing assertion green; with the filter removed the suite fails at the denominator assertion; restored tree green. Supporting evidence: `skill-extraction-workflow/scripts/test_reference_access_census.sh`. |
584
+ | Relocating entrypoint detail to a reference keeps every obligation under exactly one carrier: the row set is derived by the governing-chain tool, each derived obligation has one grep carrier under a named chain, every ledger anchor and script-pinned phrase survives verbatim, and no over-cap reference grows | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#must cover three axes, not only the executor's canonical phrase | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Zero-loss obligation map: `specs/113-extraction-entry-slim/obligation-preservation.md` (row set from `governing-chain-diff.py`; every survivor phrase counted once across the package with this ledger excluded). Supporting evidence: `skill-extraction-workflow/references/coverage-exhaustion-traps.md`, `skill-extraction-workflow/references/description-authoring.md`, `skill-extraction-workflow/references/external-practice-controls.md`. Paired control: the pinned-phrase, sync-pointer, contract-anchor, register firing-path resolution, and size gates report the same ok status on the base tree and on the landing tree, with the entrypoint body-word delta negative. |
585
+ | A usage-census mention share is not an open count: before a census figure justifies relocating, splitting, promoting, or retiring a file, the read shape in the same window is counted and the whole-read count carries the read-side cost argument | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#A mention count is not an open count | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Baseline (recorded incident): the previous round's closeout registered a ledger split as follow-up work on the ledger's mention share alone. With the rule applied: the same window's read-shape count showed whole-file loads in a small minority of the mentioning transcripts and bounded reads or gate echo in the rest, so the split was withdrawn as a read-side measure; the counts stay in the private charter. Supporting evidence: `skill-extraction-workflow/references/attention-budget-ratchet.md`. |
586
+ | Shared-tree guidance states a usage-census conclusion qualitatively; every measured ratio or count from a host census stays in the private charter, and a clause carried by a reference has exactly one carrier in that file | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#the read shape must be counted in the same window | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: the first review round and the same-candidate challenge both found a measured ratio in the new read-shape bullet; the fix batch replaced it with the qualitative conclusion and reworded one cross-reference lead-in in the dual-track reference so the clean-only-oracle clause has a single carrier (obligation table row 74). Supporting evidence: `skill-extraction-workflow/references/attention-budget-ratchet.md`, `skill-extraction-workflow/references/dual-track-review-gate.md`, `specs/113-extraction-entry-slim/obligation-preservation.md`. |
587
+ | A suite case that deliberately leaves shared fixture state mutated must restore it inside the case and assert the restoration: a registry prune only drops registrations whose directory is already gone, so deleting that directory after the prune leaves a stale registration every following case inherits | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Registered gap carried open in a prior round's frozen closeout, treated as hypothesis and reproduced against the current suite with a control leg. RED baseline: with the case's own cleanup ordering unchanged, the added restoration assertion is the single failing case in the suite and every other case stays green, so the mutated registry was invisible to the existing assertions -- including the passing-state case that runs immediately after it. GREEN: deleting the directory before the prune turns the same assertion green with no other case changed. Sibling-form evidence: the failed-add case earlier in the same file already captures the pre-case registry and asserts equality, so the corrected case matches a form the suite already contains. Supporting evidence: `skill-extraction-workflow/scripts/test_review_ledger_binding.sh`. |
588
+ | The other finding class carried open in the same frozen closeout is recorded closed rather than fixed: a failed detached-checkout creation is released through the verified path and a suite case already asserts the registry is restored | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `unchanged` | Owner key `skill-extraction-workflow/SKILL.md`. The registration's predicate (a failed add only removes the directory and never unregisters or verifies) does not reproduce against the current baseline, so no code lands for it. Oracle sensitivity proven rather than assumed: replacing the failure branch's verified release with the predicate's own described behavior turns the owning failed-add case red, together with the newline-path case that also reads that release path, and no unrelated case fails; the tree was restored and re-verified green. Supporting evidence: `skill-extraction-workflow/scripts/review_ledger_binding.py`, `skill-extraction-workflow/scripts/test_review_ledger_binding.sh`. |
589
+ | A claim about a whole file's guidance form is a measurement, so it carries a runnable instrument that states its own definitions and is committed with its baseline reading before the edit it will judge; a figure recorded without its method supports no trend claim in either direction | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/rule-consolidation.md#A whole-file form claim carries a recorded ruler | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/rule-consolidation.md (form-by-failure wording rules) plus scripts/entrypoint_form_census.py and scripts/test_entrypoint_form_census.sh, registered in the lane every run exercises. Observed failure (recorded incident): an earlier round recorded this entrypoint's prohibitive-token count with no recorded counting method; recounting the same file this round by an independently written method produced a different figure over a different scope, so no trend could be claimed from the pair and the earlier number could not be used as a baseline. RED baseline (applied mutation, differential): making a blank line close a rule turns the continuation case red and no other case, restored green. The first draft of that case anchored on the rule count, which does not move under the defect because a dropped continuation is still not a new top-level bullet -- the mutation ran green against it, and the case was re-anchored on the token total, which does move. The instrument encodes no threshold. Supporting evidence: `skill-extraction-workflow/scripts/entrypoint_form_census.py`, `skill-extraction-workflow/scripts/test_entrypoint_form_census.sh`. |
590
+ | The density of imperative-negative vocabulary in a rule set is not a measure of its guidance form: a required-slot rule, a conditional keyed to an observable predicate, and a positive recipe all carry that vocabulary inside them, so the count names a set to classify one rule at a time and never a set to rewrite | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/rule-consolidation.md#The density of imperative-negative vocabulary is not itself a measure of form | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; merged into the form-by-failure wording rule that already owns the ruler requirement rather than appended as a second bullet. Observed failure (recorded incident): an earlier round inferred from this entrypoint's prohibitive-token-to-rationale ratio that the entrypoint did not apply its own form table, and registered a rewrite programme on that inference. Walked evidence: every rule the census flags was classified against the form-by-failure rows in `specs/114-entry-form-and-routing-baselines/form-classification.md`; the flagged set is overwhelmingly already in a form the table endorses, one row is mixed, none is in a form the table calls wrong, so the inference is withdrawn and no rewrite lands. Residual gap recorded rather than closed by writing: the discipline-slip rows need rationalization-vs-reality pairs quoted verbatim from baseline or pressure runs, a form this package prescribes in two places and realizes in none, with no capture channel operated -- the missing input is evidence, not authoring effort. Supporting evidence: `skill-extraction-workflow/references/rule-consolidation.md`, `specs/114-entry-form-and-routing-baselines/form-classification.md`. |
591
+ | A delivery that adds or removes a routing-bank row rebuilds the baseline in the same round or records that every prior baseline is now orphaned: the runner treats a differing bank fingerprint as a different ruler and suppresses the diff, so a bank edit does not age a baseline, it makes it permanently incomparable while it still reads as a baseline on disk | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#改 bank 的那一轮必须同轮重建基线 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure (recorded incident): the only full-bank report on disk was taken against a 136-row bank at one replica; the bank has since grown to 158 rows and the routing surface moved by six changed descriptions and one added skill, and no run in between could have detected it. RED baseline, produced by the runner itself rather than asserted: passing that report as `--baseline` to a fresh full-bank run emitted `baseline not compared — different ruler: bank content differs`, with both `newly_failed` and `newly_passed` empty because the comparison never ran. The new report carries `replicas: 3` and its `bank_sha256`, so a later round using the same bank and replica count is comparable to it; that is the property the rule exists to preserve. Second measured result from the same run, recorded because it is what the three-replica discipline buys: 13 of 14 failures carry `ownership_split` (the replicas disagreed and conservative consensus reports FAIL) and exactly one fails consistently across all three gradings, a distinction a single-replica report cannot express. Routing dispositions are deliberately not taken here -- editing a description while holding a routing measurement in the same round is the co-change shape the runner warns about. Supporting evidence: `eval/evidence/routing-baseline-replicas3-2026-09-03/`. |
592
+
593
+ Supersede note (round 114, ledger correction with no rule change): the row above beginning "Shared-tree guidance states a usage-census conclusion qualitatively" carries, in its evidence cell, an account of which review round and which challenge produced which finding. That is conversation-level process narrative on a shared surface, and the register is append-only, so the row stays byte-identical and is corrected here by pointer rather than edited. The obligation it records, restated at artifact level: `attention-budget-ratchet.md` carries the read-shape conclusion in qualitative form with no ratio or count; the measured census figures live only in the private charter; the clean-only-oracle clause has exactly one carrier in `dual-track-review-gate.md`, recorded as row 74 of `specs/113-extraction-entry-slim/obligation-preservation.md`. The superseded row's own behavioral-evidence declaration and firing-path anchor are unaffected. This is a note rather than a table row because it changes no rule and therefore has no owner-scoped anchor of its own to declare.
594
+ | An instrument whose stated definitions ARE its contract owes a case per definition, and the counting rule it documents must be the rule its figures were produced by: a regular-expression alternation matches leftmost and non-overlapping, so a listed phrase absorbs the words inside it and a comment promising independent counts describes a different ruler than the one that ran | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_entrypoint_form_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: two independent reviewer lenses on one candidate both read the census script's comment as promising that a phrase and the words inside it are counted separately, and no fixture exercised the overlap, so the documented rule and the reported totals were different rulers while the suite stayed green. The figures are unchanged by the correction because the behaviour was always leftmost non-overlapping and only the comment was wrong: the same run reports the same totals before and after. Two cases added, each with an applied mutation: a phrase-absorption case pins `must not` at one token, and an empty-Core-Rules case pins the false-empty guard -- disabling that guard turns exactly that case red with no other case failing, restored green. The classification artifact gains a stable per-rule identifier so each verdict maps to a rule without the generated JSON. Supporting evidence: `skill-extraction-workflow/scripts/entrypoint_form_census.py`, `skill-extraction-workflow/scripts/test_entrypoint_form_census.sh`, `specs/114-entry-form-and-routing-baselines/form-classification.md`. |
@@ -177,3 +177,14 @@ Use this when validating that an extraction skill or design/client skill actuall
177
177
  3. Record what changed in the extraction method before the test, then run the source evidence through the required chain: observation, judgment, rule, acceptance.
178
178
  4. Compare the result against current skills. If the source reveals a gap, update the smallest owning skill/reference. If it confirms existing guidance, record that no new rule was needed.
179
179
  5. Mark the coverage honestly: targeted pressure test, file-level refresh, node/artifact inventory, or full workflow extraction. Never upgrade a targeted pressure test into a full-source claim.
180
+
181
+ ## Entrypoint obligations
182
+
183
+ The six rules below are the entrypoint-level obligations for UI/UX, Figma, frontend, app, miniapp, and client sources. They were relocated verbatim from `SKILL.md`'s `What to extract` Core Rules group (low-frequency detail per the entrypoint's content-placement rule); the entrypoint keeps a one-bullet summary that points here, and Step 6's UI/UX validation rows resolve against this section. Wording changes here go through the same shared-skill gates as an entrypoint edit.
184
+
185
+ - Design/client extraction must cover the judgment layer, not only the engineering layer. For UI/UX, extract aesthetic logic, interaction logic, behavioral logic, and user psychology from source evidence before landing rules about layout, components, breakpoints, or tests.
186
+ - UI/UX judgment extraction must use observable proxies, not adjectives. Read state families, navigation/entry/return paths, disabled reasons, recovery controls, timing/feedback, accessibility, responsive/device variants, and code state machines before claiming behavioral or psychology rules. Use `references/uiux-judgment-extraction.md` for the required method.
187
+ - UI/UX lessons usually route to multiple owners. Before editing, map each candidate to design, web, app, miniapp, testing, product workflow, or this extraction workflow using `references/uiux-routing-map.md`; do not land only the design rule when implementation or scenario testing is required. For mini-program lessons, `testing-strategy` owns layer/scenario selection, while `miniapp-product-dev` owns host-platform implementation, developer-tool or real-device evidence, review/release mechanics, and miniapp runtime constraints.
188
+ - Judgment-layer extraction must name what changed. For UI/UX/client sources, record whether each judgment layer produced a new rule, confirmed an existing rule, narrowed an existing rule, or found no new evidence. If the pass only improves execution/validation, say so instead of implying new aesthetic, behavioral, psychology, or interaction knowledge.
189
+ - For UI/UX/client extraction, the judgment-dimension axis enumeration lives in `references/uiux-judgment-extraction.md`. When the adjacency-scan rule fires on a UI/UX source, walk that enumeration — do not re-derive the axis list from memory.
190
+ - A UI/UX judgment-delta row is not complete with labels such as `confirmed`, `narrowed`, or `no new evidence` alone. Each visual direction/tokens row must satisfy the field list in `references/uiux-judgment-extraction.md`; if those fields were not inspected, mark the row `pending` or `out of scope` and do not claim design-judgment extraction.
@@ -54,6 +54,17 @@ When the skill is installed outside the source repo, resolve the installed `skil
54
54
 
55
55
  If the validator reports `missing_required_command`, keep the failure visible and complete the static validation bullets manually.
56
56
 
57
+ ## Read-modify-write example code
58
+
59
+ Relocated verbatim from `SKILL.md`'s `Validation & the dual-track gate` Core Rules group (the entrypoint keeps a one-bullet summary that points here); it applies to every reference file or skill example that ships runnable code against external mutable state, and wording changes here go through the same shared-skill gates as an entrypoint edit.
60
+
61
+ - Reference example code that performs read-modify-write on external mutable state (Bitable records, database rows, file contents, API state) must:
62
+ - Read failure: raise explicitly; never return `{}`, `""`, `None`, or any empty-success value that silently drops the prior state.
63
+ - Write failure: raise or skip, never silently continue.
64
+ - Uniqueness invariant: when the example declares, assumes, or depends on one, detect violations such as duplicate unique keys and raise before propagating bad state.
65
+ - Lost-update control: use optimistic concurrency controls (ETag, version field, CAS, transaction, or compare-and-swap), append-only API semantics, or an explicitly declared single-writer precondition — RMW examples that silently assume no concurrent writers will produce lost-update bugs under normal conditions.
66
+ - Data-loss anti-pattern: log-and-continue after a read failure on an append-only field.
67
+
57
68
  ## Behavioral Validation
58
69
 
59
70
  - For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology — **and the firing point is BEFORE drafting the body, not only before finalizing**. Eval-first authoring for a NEW skill (or a new hard-rule section): (1) write the evaluation scenarios first — **at least three** for a new skill (both the vendor's published authoring guide and the high-star practice pack converge on three-plus scenarios before body text; a single-rule edit may scope down to that rule's own scenario); (2) run them WITHOUT the skill and record the observed failures verbatim — and for a discipline-slip failure (the agent knows the rule and skips it under pressure), capture the agent's rationalizations word-for-word: each verbatim excuse is the raw material for one rationalization-vs-reality row and one red-flag line in the skill text (the discipline-slip form in `rule-consolidation.md`'s form-by-failure table); an invented hypothetical excuse does not qualify — counter only what a run actually said, and don't add rows for excuses no run produced; **a no-skill control that does not exhibit the failure is a stop signal — do not author guidance for a failure you cannot observe** (record the null finding instead; this is the pre-draft face of "Evidence must come before new rules"); (3) draft the **minimal** content that addresses the observed failures, then re-run the same scenarios WITH the skill; (4) when a later run, review round, or live miss surfaces a NEW rationalization for an existing discipline gate, add its explicit counter row to that gate's table and re-run the tempting scenario — counter tables accrete from observed excuses across rounds, never from imagination. The code-level RED-GREEN-REFACTOR method (write the failing case first, watch a fresh agent violate the rule WITHOUT the skill, then add the skill and watch it comply) is owned by `superpowers:writing-skills` + `superpowers:test-driven-development` — **if installed, route there; otherwise apply the RED-baseline rule inline** (manually record the without-change failure and the with-change compliance). This is the skill-authoring face of **eval-driven development** (for a behavior/routing change, run the scenario before you finalize; never special-case the scenario just to make it pass) — borrow the *principle*, not a claim of production-grade eval rigor.
@@ -0,0 +1,169 @@
1
+ #!/usr/bin/env python3
2
+ """Count the guidance FORM of an entrypoint's rules, so form claims carry a ruler.
3
+
4
+ `references/rule-consolidation.md`'s form-by-failure table picks the guidance form
5
+ from the baseline failure a rule answers. Judging whether an entrypoint follows
6
+ its own table needs a number, and the number is worthless unless the next round
7
+ can recompute it: an earlier round recorded a prohibitive-token count with no
8
+ recorded method, and a later round counting the same file by a different method
9
+ got a different figure, so no trend could be claimed in either direction. That is
10
+ the failure this script exists to prevent -- not the counting itself, which is
11
+ easy, but the counting being reproducible.
12
+
13
+ What is counted, stated here because the definition IS the instrument:
14
+
15
+ - A `rule` is one top-level `- ` bullet inside `## Core Rules`, together with
16
+ every continuation and sub-bullet line up to the next top-level bullet or the
17
+ next heading. Sub-bullets are not separate rules; they are part of the rule
18
+ whose form is being judged.
19
+ - A `prohibitive token` is a match of PROHIBITIVE_RE: the imperative-negative
20
+ vocabulary the form table calls the right form for a discipline slip and the
21
+ wrong form for every other baseline failure. Case is significant only where
22
+ the capitalised spelling is itself the emphasis (MUST / NEVER / ALWAYS).
23
+ - A `named baseline failure` is a match of FAILURE_SHAPE_RE: the rule states the
24
+ observed failure it answers, rather than only the prohibition. The form table
25
+ needs the baseline failure to pick a form, so a prohibition with no named
26
+ failure is a rule whose form was never derived from anything.
27
+
28
+ The reported diagnostic is `unanchored_prohibition_rules`: rules carrying at
29
+ least one prohibitive token and no named baseline failure. That is the set the
30
+ form table has something to say about; it is not a defect count, because a
31
+ discipline-slip rule is legitimately a prohibition -- it is the set a form pass
32
+ must classify one by one.
33
+
34
+ Exit status is 0 whenever the file parses; this is an instrument, not a gate.
35
+ Nothing here decides whether a form is right, and no threshold is encoded: a
36
+ threshold would make the ruler an argument for its own reading.
37
+ """
38
+
39
+ from __future__ import annotations
40
+
41
+ import argparse
42
+ import json
43
+ import re
44
+ import statistics
45
+ import sys
46
+ from pathlib import Path
47
+
48
+ # The imperative-negative vocabulary. Alternation is leftmost and matches do not
49
+ # overlap, so the LONGEST listed spelling wins at a position and the words inside
50
+ # it are not counted again: "must not" is one token, not "must not" plus "must".
51
+ # The phrase alternatives are therefore listed before the words they contain, and
52
+ # that ordering is load-bearing rather than cosmetic. The figure counts token
53
+ # occurrences, never distinct rules.
54
+ PROHIBITIVE_RE = re.compile(
55
+ r"\bMUST NOT\b|\bMUST\b|\bNEVER\b|\bALWAYS\b"
56
+ r"|\bmust not\b|\bmust\b|\bnever\b|\bcannot\b|\bcan not\b"
57
+ r"|\bdo not\b|\bdon't\b|\bdoes not\b|\bmay not\b|\bshall not\b"
58
+ r"|\bforbid(?:s|den)?\b|\bprohibit(?:s|ed)?\b|\bno[tn]-negotiable\b"
59
+ )
60
+
61
+ # The rule states the failure it answers. These are the phrasings this package
62
+ # already uses for that job; a rule that names its baseline failure some other
63
+ # way reads as unanchored here, which biases the diagnostic toward over-reporting
64
+ # rather than under-reporting -- the safe direction for a set meant to be walked.
65
+ FAILURE_SHAPE_RE = re.compile(
66
+ r"failure shape|failure-shape|failure mode|the failure it prevents"
67
+ r"|recurring shape|observed failure|the exact .{0,40}failure"
68
+ r"|recurrence signal|the tell that|the defect this prevents"
69
+ r"|failure it prevents|the dodge this prevents",
70
+ re.IGNORECASE,
71
+ )
72
+
73
+ BOLD_RE = re.compile(r"\*\*[^*]+\*\*")
74
+ CORE_RULES_HEADING = "## Core Rules"
75
+
76
+
77
+ def parse_rules(text: str) -> list[dict]:
78
+ """Return one record per top-level bullet inside `## Core Rules`."""
79
+ lines = text.splitlines()
80
+ try:
81
+ start = next(i for i, l in enumerate(lines) if l.strip() == CORE_RULES_HEADING)
82
+ except StopIteration:
83
+ raise SystemExit(f"entrypoint_form_census_error: no {CORE_RULES_HEADING!r} heading")
84
+ # The section ends at the next same-level heading.
85
+ end = len(lines)
86
+ for i in range(start + 1, len(lines)):
87
+ if lines[i].startswith("## "):
88
+ end = i
89
+ break
90
+
91
+ rules: list[dict] = []
92
+ current: dict | None = None
93
+ group = None
94
+ for line in lines[start + 1 : end]:
95
+ if line.startswith("### "):
96
+ group = line[4:].strip()
97
+ current = None
98
+ continue
99
+ if line.startswith("- "):
100
+ current = {"group": group, "lines": [line]}
101
+ rules.append(current)
102
+ continue
103
+ if current is not None:
104
+ # A blank line does not close a rule: sub-bullets and continuations
105
+ # are separated by blanks in this file, and treating a blank as a
106
+ # terminator would split rules and inflate the rule count.
107
+ current["lines"].append(line)
108
+ return rules
109
+
110
+
111
+ def measure(rule: dict) -> dict:
112
+ body = "\n".join(rule["lines"])
113
+ prohibitions = PROHIBITIVE_RE.findall(body)
114
+ named_failure = bool(FAILURE_SHAPE_RE.search(body))
115
+ return {
116
+ "group": rule["group"],
117
+ "head": rule["lines"][0][2:][:80],
118
+ "words": len(body.split()),
119
+ "lines": len(rule["lines"]),
120
+ "prohibitive_tokens": len(prohibitions),
121
+ "bold_spans": len(BOLD_RE.findall(body)),
122
+ "names_baseline_failure": named_failure,
123
+ "unanchored_prohibition": bool(prohibitions) and not named_failure,
124
+ }
125
+
126
+
127
+ def main() -> int:
128
+ ap = argparse.ArgumentParser(description=__doc__)
129
+ ap.add_argument(
130
+ "path",
131
+ nargs="?",
132
+ default=str(Path(__file__).resolve().parent.parent / "SKILL.md"),
133
+ help="entrypoint to measure (default: this package's own SKILL.md)",
134
+ )
135
+ ap.add_argument("--json", dest="json_path", help="write the full per-rule table here")
136
+ args = ap.parse_args()
137
+
138
+ text = Path(args.path).read_text(encoding="utf-8")
139
+ records = [measure(r) for r in parse_rules(text)]
140
+ if not records:
141
+ raise SystemExit("entrypoint_form_census_error: no rules parsed")
142
+
143
+ words = [r["words"] for r in records]
144
+ unanchored = [r for r in records if r["unanchored_prohibition"]]
145
+ summary = {
146
+ "path": args.path,
147
+ "rules": len(records),
148
+ "rule_words_median": int(statistics.median(words)),
149
+ "rule_words_max": max(words),
150
+ "rules_over_300_words": sum(1 for w in words if w > 300),
151
+ "prohibitive_tokens": sum(r["prohibitive_tokens"] for r in records),
152
+ "bold_spans": sum(r["bold_spans"] for r in records),
153
+ "rules_naming_baseline_failure": sum(1 for r in records if r["names_baseline_failure"]),
154
+ "unanchored_prohibition_rules": len(unanchored),
155
+ }
156
+ for key, value in summary.items():
157
+ print(f"{key}={value}")
158
+ print("entrypoint_form_census_ok")
159
+
160
+ if args.json_path:
161
+ Path(args.json_path).write_text(
162
+ json.dumps({"summary": summary, "rules": records}, ensure_ascii=False, indent=2) + "\n",
163
+ encoding="utf-8",
164
+ )
165
+ return 0
166
+
167
+
168
+ if __name__ == "__main__":
169
+ sys.exit(main())