@ccoalm/ccl-skills 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +10 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +14 -41
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/correction-routing-map.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/coverage-exhaustion-traps.md +7 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/incident-postmortem-extraction.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +73 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/entrypoint_form_census.py +169 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +114 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/reference-access-census.sh +157 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +483 -109
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +188 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_form_census.sh +174 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_reference_access_census.sh +209 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +394 -5
- package/dist/assets/release.json +77 -47
- package/package.json +1 -1
|
@@ -24,6 +24,12 @@
|
|
|
24
24
|
- **OpenSSF Scorecard** (github.com/ossf/scorecard) — 仓库安全/健康度评分:**每 check 0–10 → 按风险加权聚合 0–10**(Critical=10 / High=7.5 / Medium=5 / Low=2.5),weekly 扫 + 公开 BigQuery 数据集追历史。直接印证"加权 0–10 综合分 + 趋势"是业内做法(§3.4 的外部锚)
|
|
25
25
|
- **Goodhart's law**(Goodhart 1975,英格兰货币政策;Strathern 1997 普及化:"When a measure becomes a target, it ceases to be a good measure")— 度量一旦成为目标就被博弈优化、失去原效。§3.4 综合分**只做 advisory 不当 gate**、历史文件 git-ignore 的依据
|
|
26
26
|
- **Eval-driven development**(evaldriven.org + Braintrust 等多个独立 practitioner 源)— 改动前先写 eval/场景、每次改动跑同一套、**绝不为过测试而特判 eval**;把"改了才发现"提前成"改前就量"。§3.2 的 authoring-time RED-baseline + F4 防作弊(冻结 task、改 skill 顺手改测试告警)即此实践;映射的"watch it fail first"= TDD red-green(`superpowers` owns)
|
|
27
|
+
- **Anthropic `skill-creator`**(github.com/anthropics/skills, `skills/skill-creator/SKILL.md` + `agents/analyzer.md`, head 2026-04-20)— 官方技能创建/改进/度量 recipe:with-skill 与 baseline 两臂同轮起跑、每臂记 token+耗时、逐断言 pattern 读数(两臂恒过/恒挂/单侧过/高方差)、description 触发 eval(20 条 should/should-not、近似负例、每条跑 3 次、60/40 held-out 选 best)。§3.1 的两臂定义、成本列与逐断言读数三条即借此
|
|
28
|
+
- **Trace2Skill**(arXiv 2603.25158, 2026)— 根 SKILL.md 放广适用程序、auxiliary 放**低频**细节(§2.1);并行分析 + 分层归并优于顺序依赖的逐条编辑(§2.4、App. B);因果讲不通的失败不进 patch pool(§2.3)。`attention-budget-ratchet.md` 的"按触发频率放置"与普查工具的依据之一;059/060 轮已借其归并骨架
|
|
29
|
+
- **ACE — Agentic Context Engineering**(arXiv 2510.04618, 2025-10)— context 以带 id 与 helpful/harmful 计数的条目化 bullet 表示、增量 delta 更新、grow-and-refine 去重;命名两种失效:brevity bias(优化把 context 压成短而泛的口号)与 context collapse(整体重写把细节压没)。前者是我们的"过压缩"警戒,后者是零损失义务表存在的理由;计数器对应本仓的引用访问普查
|
|
30
|
+
- **IFScale**(arXiv 2507.11538, 2025-07)— 指令密度上升时遵循率下降、偏向靠前指令;证据档与用法见 `external-practice-controls.md` §Instruction-following mechanisms 表
|
|
31
|
+
- **从成功中学习(人因/组织学习侧)**:Hollnagel & Leonhardt, *From Safety-I to Safety-II* (EUROCONTROL 白皮书, 2013) — 事情做对是因为人把工作调整到匹配条件,不是因为照规则做;Ellis & Davidi, *J. Applied Psychology* 90(5) 2005 — 成功+失败一起复盘比只复盘失败提升更大(准实地实验);美军 TC 25-20 *A Leader's Guide to After-Action Reviews* (1993) — 四问 + sustain/improve 双清单;Levitt & March, *Organizational Learning* (Annual Review of Sociology 14, 1988) — 能力陷阱与迷信学习。落点:`incident-postmortem-extraction.md` §Success reviews
|
|
32
|
+
- **GAO-01-1015R / GAO-02-195**(2001–2002,NASA lessons-learned 流程调查)— 教训库"收了不用"的经典证据:管理者不常识别/提交/使用教训、系统耗时、不熟悉他中心的教训、缺激励。这是"教训要推到触发点而不是存进库"(firing point 而非 ledger)这条本仓设计的外部反例锚
|
|
27
33
|
|
|
28
34
|
---
|
|
29
35
|
|
|
@@ -115,6 +121,8 @@ Result inflation 没有 MAST 对应——它是 context / 成本问题,不是
|
|
|
115
121
|
- reviewer 或 challenger 核准 task 抽样
|
|
116
122
|
- 成功标准 改前定义 (specific assertions on output / behavior)
|
|
117
123
|
- 比对 **transcript + outcome**,不只 final prose(per Anthropic 2026 evals)
|
|
124
|
+
- **两臂的定义按改动类型定**:新技能 → 对照臂是 *无技能*;改既有技能 → 对照臂是改前快照(先 `cp -r` 冻结再改,别拿改后的树当基线);两臂**同一轮并发起跑**(the with-skill and baseline arms must start in the same turn),别先跑实验臂再补对照臂。每臂记录 **token 用量 + 耗时**(宿主的子 agent 完成通知里有,过时不候),效果与成本同列——一条只提 pass rate 不提成本的对照没法判"值不值"(Anthropic skill-creator 2026 的 benchmark 形态)
|
|
125
|
+
- **逐断言读数,不只看聚合通过率**:两臂都恒过 = 该断言不区分技能价值;两臂都恒挂 = 断言坏了或超出能力;有技能过、无技能挂 = 技能在此生效;有技能挂、无技能过 = 技能在此帮倒忙;高方差 = 断言 flaky 或行为非确定。任何一类都先记录再定性,聚合分会把这些抵消掉(同上源 analyzer 判据)
|
|
118
126
|
- **两臂无差异时不得在这一步定性为该文本无用**:成因至少两种——既有规则已覆盖该 failure-class(那是合并决策,见上),或该行为在 host 基线上本就存在。要分清得另取证据,**本节不规定取法**:低 N、臂配置混杂、以及"只给对照臂换夹具或换权限就多一个变量"这三样让这类归因很容易做错,宿主自身系统层也始终在场。取不到可靠证据就把成因判为**未确定**,别猜一个填上——本节判"改动有没有效"的对照始终是 before/after 两臂
|
|
119
127
|
|
|
120
128
|
实测踩过的那次里,零差分的成因是既有规则已覆盖,不是这句话本身无效。
|
|
@@ -177,6 +177,14 @@ When the extraction is incident-driven, the main `extraction-quickstart.md` flow
|
|
|
177
177
|
- **Stopping at the first controllable layer** — landing only the test or review rule when the contract / schema / capability owner is the real earliest practical prevention point. Continue the chain; list later layers as secondary controls.
|
|
178
178
|
- **Re-extracting the same incident twice** — happens when the first extraction landed only the procedural patch. Treat the second pass as a failure-class extraction, supersede the procedural rule, and record why the first pass was insufficient.
|
|
179
179
|
|
|
180
|
+
## Success reviews — the sustain half of learning
|
|
181
|
+
|
|
182
|
+
An incident is one source class; ordinary work that went right is the larger one, and the entrypoint's result-classification rule already requires a `stable success` to carry mechanism, non-luck evidence, reuse conditions, firing point, and owner. Two obligations that rule leaves implicit, each with a primary source:
|
|
183
|
+
|
|
184
|
+
- **Record the success mechanism as work-as-done, not as rule compliance.** Things go right mostly because the agent adjusted its work to the actual conditions, not because a rule was followed to the letter; a `stable success` row must name the adjustment that produced the outcome (which condition was read, what was varied, what was checked) and only then whether an existing rule fired. A row that says "the rule was applied" names compliance, not the mechanism, and cannot transfer when conditions differ. (Hollnagel & Leonhardt, *From Safety-I to Safety-II*, EUROCONTROL 2013: "the reason that things go right is not people behave as they are told to, but that people can adjust their work so that it matches the conditions"; "we cannot make sure things go right just by preventing them from going wrong".)
|
|
185
|
+
- **A sustained practice carries its own re-examination trigger.** Repeated success with a procedure accumulates experience with it and starves the alternative of a fair trial, so a sustain row must name the condition under which the practice is re-tried against an alternative (a changed host capability, a cheaper tool, a second consecutive workaround, a cost row that stops improving) — reuse conditions say where it still holds; the trigger says when to stop assuming it does. (Levitt & March, "Organizational Learning", *Annual Review of Sociology* 14, 1988: competency trap and superstitious learning — the disconfirming observation the classification rule already requires is the guard against the second.)
|
|
186
|
+
- Every retro asks the sustain question next to the improve question, but a sustain row must be written only when stable-success evidence meets the entry bar (mechanism + non-luck evidence + reuse conditions + firing point + owner); otherwise the retro records an explicit `no-new-lesson` or `unstable/insufficient evidence` disposition on the sustain side, never a fabricated success. The pairing has evidence in its own settings: reviewing successes and failures together improved subsequent performance more than reviewing failures alone in a quasi-field experiment on navigation training (Ellis & Davidi, *Journal of Applied Psychology* 90(5), 2005), and the Army's after-action review ends with two lists, sustain and improve (TC 25-20, 1993). The entrypoint's LARGE-session sustain axis and the DO-CONFIRM card's sustain row are the firing points.
|
|
187
|
+
|
|
180
188
|
## What gets committed
|
|
181
189
|
|
|
182
190
|
A successful incident extraction usually produces, in one commit or a tight series:
|
|
@@ -37,6 +37,7 @@ Two wording rules for whichever form wins:
|
|
|
37
37
|
|
|
38
38
|
- **No nuance clauses.** "Don't X unless it matters" reopens the negotiation — appending a single nuance clause to a winning recipe degraded it from consistent to noisy in the same tests. Write a real exception as its own conditional on an observable predicate.
|
|
39
39
|
- **Exemption clauses don't scope.** "This limit doesn't apply to code blocks" still suppresses code blocks; if part of the output must be exempt, restructure the rule so it cannot reach that part.
|
|
40
|
+
- **A whole-file form claim carries a recorded ruler, committed before the edit it will judge.** A claim about a file's form as a whole ("this entrypoint over-uses prohibitions", "the form table is not applied to its own bullets") is a measurement, and a measurement whose method lives only in the round that took it cannot be recomputed: a figure recorded bare is not comparable to the next round's figure, so neither a trend nor an improvement may be claimed from the pair, in either direction. Record the method as a runnable instrument that states its own definitions — what counts as one rule, as a prohibition, as a named baseline failure — and commit it, with its baseline reading, before the edit whose effect it will report. `scripts/entrypoint_form_census.py` is this package's instrument; it encodes no threshold, because a ruler that also scores makes itself the argument for its own reading. Failure shape: a round measured this entrypoint's prohibitive tokens and recorded the count alone; the round that came to act on it counted the same file by its own method, got a different figure, and could not tell a real change from a change of ruler. **The density of imperative-negative vocabulary is not itself a measure of form.** A required-slot rule, a conditional keyed to an observable predicate, and a positive recipe all legitimately carry that vocabulary inside them, so the count names a set to classify one rule at a time, never a set to rewrite — read that way it is the same category error as reading a mention count as an open count. Failure shape: a round inferred from a high prohibitive-token-to-rationale ratio that this entrypoint did not apply its own form table; walking the resulting set rule by rule found the great majority already in a form the table endorses, and the inference was withdrawn rather than acted on.
|
|
40
41
|
|
|
41
42
|
Boundary vs the owning Core Rule's salience mechanism: that rule governs how JOINT requirements are structured (walked enumeration at the firing point; merge-over-append); this table governs the form of a SINGLE rule's text once its landing spot is chosen.
|
|
42
43
|
|
|
@@ -483,6 +483,8 @@ Round 073-receipt-bundling rows (new table so the entry renders as a table row a
|
|
|
483
483
|
|
|
484
484
|
| A before/after routing comparison may vary only ONE routing variable (one description, or one skill's indivisible routing face) for its delta to be attributable | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#一次改前/改后对照只准动**一个路由变量** | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source-side paired single-description A/B is the observed working mechanism (borrow round; sanitized provenance in the private alias archive); no mis-attribution incident observed in this repository yet, so the clause lands as protocol item 5 with the existing four items unchanged as the paired control. |
|
|
485
485
|
|
|
486
|
+
| A reviewer-isolation gate judges only what can invoke something: the exact tool set, the tool_use scan, an empty MCP list and the pinned permission mode. Skill, command and plugin lists are vocabulary the host and its plugins own; recording them is fine, judging them by name, shape or origin is not, because nothing listed there is invocable past the pinned tools and every such predicate turns a routine CLI release or an older installed plugin into a reviewer-lane outage that proves nothing. An owner-skill binding the wrapper cannot establish costs the binding receipt, never the review | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_init_policy_matrix.sh | `updated` | Owner key `code-review/SKILL.md` (contract paragraphs on the Claude wrapper, plugin binding and the binding receipt rewritten); lands in `code-review/scripts/parse_probe_result.py` (vocabulary fields become `KNOWN_VOCABULARY_INIT_FIELDS`, never judged; `REQUIRED_EMPTY_INIT_FIELDS` keeps only `mcp_servers`; the built-in name snapshots, the baseline loader, the whole-value and bare-identifier gates and the unclassifiable class are deleted), `code-review/scripts/claude_review.sh` (no acknowledgement-only baseline invocation, no help-prose probe of `--safe-mode`, `--disable-slash-commands` optional; an unverifiable installed registry or an unloadable plugin degrades to `native_skill_binding=unavailable` instead of refusing), `code-review/scripts/review_gate.py` (accepts the `unavailable` receipt as a review with `controller-profile` skill evidence and no natively reviewed skills; a wrapper attesting neither still fails closed), the oracle `init_policy_matrix.py` (policy G restated: vocabulary is data; the real 2.1.261 owner-aware init is a tolerated row beside every breach class), `test_init_policy_matrix.sh`, `test_parse_probe_result.sh`, `test_claude_review_probe.sh`, and the contract references `client-routing.md`, `manual-invocation-and-prompts.md`, `staged-review-contract.md`, `scripts/runtime-surface-verification-design.md`, `scripts/AGENTS.md`; plan in `specs/116-reviewer-vocabulary-is-data/plan.md`. Observed failure, first-hand: Claude Code 2.1.261 added the built-in skill `workflow-authoring`; the owner-aware lane classified it `unclassifiable_host_vocabulary`, reported `capability_missing`, and every review in that period fell through to another client — the third release-driven outage of the same class after `/import` and `auto-mode-setup`, and the baseline mechanism could not help because baseline skills were by design not authority. Reproduced differentially against the base parser with a success stream synthesized from the real 2.1.261 init: rejected before, accepted after, while a synthesized inherited MCP server is still refused. RED-baseline (applied mutations, each in a disposable copy, control green at 270 cases / 0 mismatches): dropping the MCP emptiness requirement flips 40 rows; reading a populated vocabulary list as a breach flips 76; reading it as schema drift flips 72; the four pre-existing mutants stay detected. Wrapper-level evidence through the fake CLI: a plugin manifest declaring hooks, an older registry whose package hash mismatches the profile, and a missing registry each produce a challenge verdict with `unavailable` and no `--plugin-dir`; every vocabulary fixture that used to refuse (empty enumeration, omitted owner, host built-ins, foreign entries, duplicates) now attests `established`; a CLI without `--disable-slash-commands` and a CLI whose `--safe-mode` prose contradicts itself both still run; a CLI without `--safe-mode` still refuses. Supersedes the four vocabulary rows of the 015 round above: their mechanism is removed rather than extended, per same-class convergence by deletion, and their transferable lesson survives in inverted form here — the predicate was never on a property the control owns, so the class was discharged by deleting the predicate. |
|
|
487
|
+
|
|
486
488
|
| Cross-skill / cross-reference routing pointers in body text must carry the routing quadruple (trigger / scope / output / return point); a bare "refer to X if useful" pointer is never a landing shape | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/description-authoring.md#routing pointer in body text must carry the routing quadruple | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Borrowed from an adopting skill pack where unbounded pointers were the dominant dead-routing shape (two adopters verified in source; sanitized provenance in the private alias archive); landed in the routing-surface authoring reference because the entrypoint is size-ratcheted level — the reference is the required pre-edit reading for routing-surface work, and eval-routing.md's silent-skip row points back at it; the description-side Skip-when idiom already satisfies the quadruple and is named as the unchanged control. |
|
|
487
489
|
| Review-finding fixes are held un-applied until the full review+challenge chain has run on the frozen candidate: under the 1+1 budget apply-now is never fundable after the review round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (enumeration item 1 + cadence Round 2). Observed failure: a prior round applied its review fixes before the challenge; the tracked chain broke (challenge binds to the round-1 candidate), the round lost its double-receipt terminal, and closure required a user-granted continuation chain. The prior wording stated the rule only as a fundability conditional whose arithmetic the agent under pressure never ran; the operative unconditional form (hold all fixes; challenge on the frozen candidate; land the batch after the chain) is now explicit at both firing points. RED baseline: the recorded chain-break incident is the without-change failure; the with-change compliance surface is the explicit hold rule at the enumeration walked when a round returns findings. |
|
|
488
490
|
| A frozen eval case is sacred: deleting or re-scoping a bank task or golden trace requires a same-round `case-retired:`/`case-rescoped:` register adjudication row, and average improvement never offsets a frozen-case loss | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#平均改善不得抵消单条冻结案例的失守 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/eval-routing.md (冻结案例神圣 bullet), scripts/test_frozen_case_sanctity.sh (fast lane, registered), docs/f4-skill-effectiveness-harness.md (pointer line). Rule semantics: a previously-passing frozen case that degrades — including to unsure/INCONCLUSIVE — is a regression, and its sacredness attaches per case, so no aggregate improvement offsets it; the mechanism's source-verification record lives in the round's private archive, and transferred evidence does not exempt the behavioral row. RED baseline (replayed, throwaway clone at the round base): deleting the non-pinned bank case ctrl-unit-test passed the pre-change surface silently (test_routing_bank_integrity.sh exit 0) and reddens the new gate (exit 1 naming the id and the required adjudication row); re-scope and golden-trace-deletion mutants red for the right reason; adjudicated-deletion and untouched-tree control legs green; no-base and unresolvable-base legs print the explicit skip token. |
|
|
@@ -558,3 +560,74 @@ Round 073-receipt-bundling rows (new table so the entry renders as a table row a
|
|
|
558
560
|
| Fan-out width is counted in workers that each own one bounded slice — tightly related items may sit inside that one slice — and every brief must carry an effort budget, so a width rule can never produce multi-task workers whose ownership, deadline, and failed-return classification cannot be attributed to one bounded unit | `multi-agent-delegation` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/multi-agent-delegation/SKILL.md#each owning one bounded slice | `updated` | Owner key `multi-agent-delegation/SKILL.md`. Lane-5 review (codex) on candidate 56342b9: gate item 4's 'several tasks each' contradicted execution step 3's one bounded task per agent (P1); fixed with zero-loss trims so the entrypoint stays within the 5000-word gate. RED baseline is the reviewed wording in the lane-5 round-1 receipt. |
|
|
559
561
|
| Revocation inside a tool-bearing loop is stated as one fencing invariant rather than accumulated ordering patches: the generation advance and the call's final generation check at the irreversible handler's commit boundary are serialized by the same lock or fence, a call rechecks immediately before crossing that boundary, and the lease and fencing-token mechanics already required for stale agents are reused rather than re-derived | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/agent-tool-dispatch.md#revocation is a fencing problem, not a prompt problem | `updated` | Owner key `llm-inference-integration/SKILL.md`. Four consecutive same-class findings (lanes 1, 3, 4, 5: append-only vs invalidation, eviction vs invalidation, in-flight revocation, admission ordering) showed the class was a concurrency protocol being specified one patch at a time; the lane-5 succession's 'started is ambiguous' finding (P1) is fixed by the invariant form and by routing the mechanics to the existing fencing-token rule, per the same-class-recurrence design rule. RED baseline is the reviewed wording in the lane-5 succession receipt. |
|
|
560
562
|
| The code-then-execute pattern cuts the trifecta edge only when the privileged code, its allowed sinks, and its permitted data flows are generated and frozen before any untrusted content is read, untrusted input entering afterwards only as non-instruction typed data with the negative test restoring that ordering; and the serving-lever table states prefill/decode disaggregation's throughput effect as engine- and workload-dependent to be measured on the target engine, not as a categorical claim | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/llm-inference-integration/references/retrieval-agent-safety.md#generated and frozen before any untrusted content is read | `updated` | Owner key `llm-inference-integration/SKILL.md`. Lane-6 review (codex) on candidate 09a040c: the code-then-execute option did not require code and sinks to be frozen before exposure (P1) and the disaggregation caveat was categorical (P2); the packet-evidence finding is accepted_tradeoff as before; the missing lane-5 authorization record and the authorization-chain wording were corrected in the evidence directory. RED baseline is the reviewed wording in the lane-6 round-1 receipt. |
|
|
563
|
+
| The merge gate binds an integration branch that accumulated several already-bound rounds and is promoted as one pull request by walking HEAD's first-parent chain down to the first commit already on the target: each round merge is rebound by the same gate in a detached checkout of its second parent against its first parent with that checkout's own controller and validator, a merge whose second parent is already on the target is a sync merge that owes nothing, every step must equal the automatic merge of its parents, a non-merge commit on the chain is refused, and the chain is consulted only after the single-ledger and manifest paths and only for the default path set | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py, its suite, references/dual-track-review-gate.md, references/extraction-quickstart.md, and a .github/workflows/ci.yml comment). Observed failure: a promotion of three stacked rounds could be bound neither by one ledger nor by a path partition, because two of the rounds appended to this register and no round ever froze the sum of both appends, so the release had to land as three separate pull requests each pointing at one round's merge commit. Every such round had already been bound at its own base when it merged; the gate simply discarded that evidence at promotion time. Chain cases were written first and observed failing on the previous gate (10 failing), then green; a mutation walk reds each load-bearing predicate exactly on its own cases; the real three-round promotion binds as three rounds. Merge-queue aggregation of unmerged pull requests remains unsolved and is now stated as such in the ci.yml comment. |
|
|
564
|
+
| A historical round on the first-parent chain is judged with the landing tree's controller and validator, never the round's own, and the detached checkout it is rebound in must be verifiably released (removal result checked, registry read back, no repository-wide prune) or the verdict is an error; the walk accepts a chain of exactly the bound's length and refuses one longer — superseding the previous row's clause that a round is rebound with its own checkout's tools | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py, its suite, and references/dual-track-review-gate.md). The round's own dual-track lane found all three: the adversarial challenge (P1) showed a round could install a hollow validator in its own branch, forge a ledger the real validator rejects, and stay bound after a later round restored the real validator, because history was being judged with history's tools; the independent review (P1) showed the checkout release discarded the removal result and ran a repository-wide prune that also drops registrations the run did not create; both lanes found the off-by-one that refused a chain of exactly the bound's length (P2). Fixes held until after the challenge; three cases were written first and observed failing on the pre-fix gate, then green. RED baseline is the pre-fix gate accepting the forged round, refusing the 64-step chain, and reporting ok past a failed release. |
|
|
565
|
+
| A detached checkout whose creation fails after git registered or populated it is released through the same verified path as a successful one, so a failed rebind leaves no repository worktree state behind and the error names any release problem alongside the creation failure | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). The round's succession challenge left this open at the exhausted lane budget: `git worktree add --detach` can return nonzero after creating its registration, and the failure branch only deleted the directory. Closed in an authorized continuation lane: the case (a git shim that runs the real add and then fails) was written first and observed failing on the pre-fix gate, then green. RED baseline is the pre-fix gate leaving the registration behind. |
|
|
566
|
+
| Releasing a detached checkout distinguishes a registration that survives removal (an error) from an unregistered directory this run created before git registered anything (deleted by the run itself, reported only if it survives), so a creation that fails before registering errors cleanly without a release complaint | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Found while pre-covering the previous row's fix before review: routing a failed creation through the verified release path made an add that never registered anything report a spurious release problem and leave its directory behind. The benign before-registration sibling case sits beside the registered-failure case as the precision row; the registered-failure case is the RED baseline for the class and this row's case pins that the benign shape neither complains nor leaks. |
|
|
567
|
+
| The worktree registry is read back NUL-terminated when a detached checkout is released, because line-oriented porcelain cannot carry every path the gate may compare against — a newline splits the line, and quoting of other unusual bytes depends on git version and core.quotePath — so a surviving registration under such a temporary directory would match neither spelling and read as gone | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). The continuation lane's adversarial challenge (P2) named a tab-bearing TMPDIR; on the git in use a tab is printed raw and did not reproduce, while a newline-bearing path did split the line, so the case uses a newline: an add that registers and returns nonzero plus a refused remove left the registration in place while the release reported clean. Fixed by reading `git worktree list --porcelain -z` and comparing exact NUL-delimited fields; the case was written first and observed failing on the pre-fix gate, then green. RED baseline is the pre-fix gate reporting no release problem with the registration surviving. |
|
|
568
|
+
| Retirement or relocation of an over-budget entrypoint or reference cites the host-local usage census (per-file session counts derived from the agent's own transcripts, counts only) instead of the author's opinion, and the census must report `unevaluated` rather than a zero table whenever it could not evaluate | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_reference_access_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: external primaries — per-bullet helpful/harmful usage counters in itemized-context evolution (ACE, arXiv 2510.04618 §3.1–3.2), root-vs-auxiliary placement by frequency (Trace2Skill, arXiv 2603.25158 §2.1), and the official skill-authoring guidance that a bundled file the agent never accesses is unnecessary or poorly signaled. Observed failure this round: the first census version printed `sessions_touching_package=0` on a 60-day window (the candidate list exceeded one exec batch and the batch error was swallowed) while a 14-day window on the same host counted touching sessions normally — an instrument that reads zero when it could not evaluate is the false-green shape ratchet invariant 2 forbids. Fixed by batching; the regression test's ARG_MAX leg (450 synthetic transcripts, 3 touching one reference) and its unevaluated-sentinel and privacy-contract legs are the RED-to-GREEN evidence. RED (measured on fa0a7de): `NO_HITS: never accessed\|ignored content\|access log\|usage count` across SKILL.md and references (ledger excluded). Supporting paths: scripts/reference-access-census.sh, scripts/test_check_ccl_regressions.sh (fast lane registration), references/attention-budget-ratchet.md §Retirement and relocation signal, references/extraction-quickstart.md tool inventory. Host census figures stay in the private charter. |
|
|
569
|
+
| Low-frequency entrypoint detail relocates verbatim into the reference the entrypoint already points at (UI/UX judgment obligations, correction-type routing for test/verifier/deferred-evidence corrections, read-modify-write example-code obligations), the entrypoint keeps a one-bullet summary carrying the load-bearing obligations and every check-ccl phrase pin, and the entrypoint's body must shrink, never grow, in a round that touches it while it is over budget | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#never close a judgment-delta row whose visual direction/tokens fields were not inspected | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: the entrypoint that enforces a 5,000-body-word cap on every other skill carried 16,748 body words at base (size gate `entrypoint_size_severe_debt`, 3.3× the cap; 90 top-level bullets, 12 over 300 words) — the placement rule it states was violated by its own file, and Codex consumers cannot read it in one pass. RED (measured on fa0a7de): `check-size-budget.sh` base_body_words=16748; GREEN at head: head_body_words=16426, `changed_entrypoint_word_delta … delta_body_words=-322`, `entrypoint_size_blocking_ok`. Zero-loss obligation map: six UI/UX bullets → references/uiux-judgment-extraction.md §Entrypoint obligations (byte-identical), DFE-CONT + tests-before-test-cases rules → references/correction-routing-map.md (byte-identical, new file linked from Reference Loading), five RMW obligations → references/validation-and-landing.md §Read-modify-write example code (byte-identical); the four `skill_extraction_test_case_first_gate` phrases stay inline in the summary bullet (the first draft dropped one and the gate went RED, then GREEN after restoring it); the five ledger `file:` anchors and eight script pins into SKILL.md were enumerated first and none sits in the moved text (`register_firing_path_resolution_ok`). Remaining debt after this batch: 16,426 body words; the census and cost row exist so the next batch is chosen from evidence. |
|
|
570
|
+
| Content placement sorts by firing frequency as well as by kind: a rule that fires only on a narrow source class or correction type is low-frequency detail even when it is non-negotiable and must live verbatim in the reference the entrypoint already points at, behind a one-bullet summary that keeps the load-bearing obligations inline; the census is advisory and never a gate | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#must live verbatim in the reference the entrypoint already points at | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: Trace2Skill §2.1 (root document stores broadly applicable procedural knowledge, auxiliary files provide lower-frequency details) and §2.4/App. B (joint consolidation over many traces beats order-dependent one-lesson-at-a-time editing — the regime this skill's per-round appends have been in); Anthropic skill-creator progressive disclosure and the best-practices navigation observations (overreliance → move into the entrypoint; ignored → remove); Goodhart for the never-a-gate clause (same anchor as the health roll-up). RED (measured on fa0a7de): `NO_HITS: lower-frequency\|low-frequency\|firing frequency` in SKILL.md and references (ledger excluded); the placement rule sorted by kind only, so every non-negotiable read as entrypoint material. Head carries exactly one anchored list line. |
|
|
571
|
+
| A skill-effect comparison defines its baseline arm by change type (no-skill for a new skill, a frozen pre-change snapshot for an existing one), starts both arms in the same turn, records tokens and duration per arm, and reads results per assertion (pass-both, fail-both, one-sided, high-variance) before trusting an aggregate pass rate | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/harness-patterns-and-eval.md#the with-skill and baseline arms must start in the same turn | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: Anthropic skill-creator SKILL.md (head 2026-04-20; Step 1 spawn with-skill and baseline runs in the same turn with the snapshot rule for existing skills, Step 3 capture `total_tokens`/`duration_ms`, Step 4 analyst pass) and agents/analyzer.md §Analyzing Benchmark Results step 2 (per-assertion patterns). RED (measured on fa0a7de): `NO_HITS: non-discriminating\|passes in both\|without-skill\|duration_ms\|total_tokens` in harness and validation references; §3.1 compared before/after arms only, had no cost column, and read only aggregate deltas. Supporting path: references/external-practice-controls.md gains the IFScale row (density degradation, primacy bias; evidence grade recorded), which is a table row and carries no anchor. Concept-adjacent coverage is not functional equivalence; that bar makes this row I→updated, not P. |
|
|
572
|
+
| Every non-wording closeout records a cost row (review and challenge rounds run, findings fixed, accepted, or deferred, wall-clock from charter to PR, and net body-word delta per touched entrypoint) and the next round must read it before choosing its batch shape | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/extraction-quickstart.md#must read this row before deciding its batch shape | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: Google SRE Postmortem Culture (regularly survey whether writing a postmortem entails too much toil and improve the process from the answers) and the skill-creator per-run cost capture. RED (measured on fa0a7de): `NO_HITS: toil\|wall-clock` in extraction-quickstart.md; the closeout template recorded final state and lessons only, so process cost (one recent round ran 18 review rounds across 7 lanes; another re-ran full verification 8 times) lived only in private notes and never fed the next round's batch decision. Head carries exactly one anchored list line. |
|
|
573
|
+
| The census CLI treats a flag without its value as a usage error (exit 2, stderr) rather than an unbound-variable abort, and its regression suite's large-window leg must exceed every common ARG_MAX so that the first version's single-exec expansion cannot pass it | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_reference_access_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Self-review under the terminal owner and the testing owner (controller-derived owners for the new CLI and its test) found two defects before external review: a flag without a value aborted under `set -u` with exit 1, and the ARG_MAX leg used 450 short paths (~54 KB), which no ARG_MAX rejects — a mutation never applied, so a hypothesis. Observed RED: with 11,000 transcripts of ~230-character paths (~2.5 MiB) the exact first-version mutation applied to a disposable copy failed the leg with `sessions_touching_package=0` (mutant rc=1); the restored suite is GREEN in 3.9 s. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh. |
|
|
574
|
+
| The census withholds its table on any input error (unreadable, vanished, or unexecutable transcript inputs: exit 2 with an `unevaluated` count-only message) instead of printing zeros or the ok token, and every sentinel and usage error is path-free — no log root, caller path, or host detail reaches stdout or stderr | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#withholds the table (exit 2 on input errors) | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Round-1 review (codex, candidate c2c4d377) findings 0 and 1 and the round-2 challenge finding 0: the empty-log sentinel interpolated the log-root path (an absolute private path) and the test asserted that leak; find/grep/xargs errors were collapsed by `\|\| true` into zero counts followed by the ok token. Fixed: path-free sentinels (test asserts the log root and any absolute path are absent), a stderr-collecting error log that withholds the table with exit 2 (new test leg: a chmod-000 transcript → exit 2, `1 input error(s)`, no table, no ok token, no path), and the ratchet bullet now states the contract. RED is the reviewed candidate; GREEN is `test_reference_access_census_ok` on the fixed tree. Correction to two earlier rows of this round: their `NO_HITS` evidence for the terms `passes in both` and `wall-clock` overstated the grep — the base has one unrelated hit each (`passes in both directions of the wrong answer` in rule-consolidation.md; `wall-clock deadline` in the harness MAST table); the mechanisms had no prior carrier, which is the consolidation question those rows answer; recorded here because rows are append-only. Review findings 2 (ledger over 500 lines) and 3 (two rows sharing a command locator) are `accepted_tradeoff` with evidence in the round's disposition files. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh, specs/111-extraction-skill-benchmark/evidence/primary-source-excerpts.md (finding 4, fixed). |
|
|
575
|
+
| The census treats an explicitly supplied log root that does not exist as an input error (an absent default root stays normal), never echoes a caller argument in a usage error, and guards every stat substitution so an input error always reaches the withholding path instead of aborting under set -e | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#must be treated as an input error | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Continuation lane (scope: the defects named in the preceding closeout). Observed failures: succession challenge (codex, candidate 282c6e14) P2 ×2 and the mis-packeted first succession attempt's P1 (explicit missing root skipped silently). Each new test leg went RED under its applied mutation on a disposable copy (explicit-root check removed → leg 8 red; argument echo restored → leg 7 red; stat guard removed → leg 9 red with the pipeline status) and GREEN on the fixed tree. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh. |
|
|
576
|
+
| From the second review round on, the packet's exclusion list must exclude every evidence JSON the round has already added — receipts, dispositions, closeout files — so that the reviewed candidate equals the candidate the merge-side binder computes, and bound evidence (base attestations, excerpt files) is committed before the round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/extraction-quickstart.md#must exclude every evidence JSON the round has already added | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed twice in consecutive rounds: the previous round's private notes recorded three receipts binding three candidates, and this round's first succession attempt excluded only the two receipts and bound candidate 0b91f263 while the binder computed 282c6e14 (`review_ledger_binding.py --print-candidate` before and after; the mis-packeted receipt is kept as history in the evidence directory). RED is that receipt; GREEN is the retry with the eight-file exclusion set whose receipt binds 282c6e14 and the validator's `extraction_review_state_ok`. |
|
|
577
|
+
| A stable-success row records its mechanism as work-as-done (the adjustment that matched the actual conditions) before any rule-fired claim, and a sustained practice names the condition under which it is re-tried against an alternative | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/incident-postmortem-extraction.md#Record the success mechanism as work-as-done | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: external primaries read this round to check learning-from-success theory — Hollnagel & Leonhardt 2013 (Safety-II white paper, PDF read: performance adjustment as the reason things go right), Levitt & March 1988 (competency trap, superstitious learning), Ellis & Davidi 2005 (successes+failures review beats failures-only, PubMed abstract), TC 25-20 (AAR sustain/improve lists, PDF read). Functional-equivalent check at fa0a7de: the classification rule's mechanism / non-luck / reuse-conditions / disconfirming-observation fields and the LARGE-session sustain axis exist (P for Ellis & Davidi, AAR, superstitious learning); `NO_HITS: work-as-done\|competency trap\|re-examination trigger` — the compliance-vs-adjustment distinction and the re-exploration trigger had no carrier. Head carries exactly one anchored list line; entrypoint gains a 20-word pointer while its round delta stays negative. |
|
|
578
|
+
| A description trigger evaluation grades each bank utterance at least three times per description version, keeps negatives as near-misses that share vocabulary with the skill, and selects an auto-rewritten description on a held-out split, never on the training split | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#each bank utterance must be graded at least three times per description version | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source: Anthropic skill-creator Description Optimization (20 realistic queries, near-miss negatives, 3 runs per query, 60/40 train/held-out, best by test score). Functional-equivalent check at fa0a7de: the bank already carries `expected_skill: none` sentinels and `must_not_route_to` decoys and the runner has `--replicas` (P for negatives and repetition capability); `NO_HITS: held-out\|train/test\|overfit` — repetition was optional and no held-out discipline existed. Landed because an identified gap is a fix item, not a record. |
|
|
579
|
+
| The census keeps a NUL-delimited, pathname-deduplicated transcript inventory so overlapping or repeated log roots and newline-bearing names count a session once, tolerates an unset HOME (no default roots, never an unbound-variable abort), never echoes an invalid `--skill` value, and adds a shim-forced transcript-read error leg that does not depend on file permissions; the success-review reference requires a sustain row only when stable-success evidence meets the classification entry bar | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#overlapping or repeated log roots must count a transcript once | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Continuation lane (chain r3, codex): review finding 2 and challenge finding 0 (overlap double-count, newline-split names), review finding 3 (`--skill` echoed), challenge finding 1 (HOME unset abort; root-dependent unreadable leg), challenge finding 2 (the success-review pairing sentence universalized a sustain row beyond the entry bar — reworded to require `no-new-lesson`/`unstable` otherwise). Applied mutations on disposable copies: dedupe removed → leg 10 RED; `--skill` echo restored → leg 11 RED; HOME guard removed → leg 13 RED; restored suite GREEN in 3.6 s. The `seq` portability finding is refuted on this host (`/usr/bin/seq` present on macOS) and recorded as such in its disposition. Review findings 0/1 (ledger over 500 lines; command locators) repeat the earlier accepted tradeoffs; challenge finding 3 (packet evidence) repeats the packet-evidence class. |
|
|
580
|
+
| The census maps grep's no-match status to success inside each scan batch and treats any other batch failure — with or without a stderr line — as an input error that withholds the table, and its ARG_MAX regression fixture builds its padding without `seq` and asserts the inventory exceeds the largest common ARG_MAX before the mutation-sensitive check | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#a scan batch that fails without writing stderr must still withhold the table | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Second continuation lane (scope: the defects named in the preceding closeout). Observed failures: continuation-lane succession (codex, candidate c949ad35) findings 1 and 2. Applied mutation on a disposable copy: restoring the plain `xargs grep` pipeline makes the new silent-exit-2 shim leg (15) fail with a table and the ok token; the fixture-size assertion is checked at run time (inventory > 2 MiB). Restored suite GREEN in 6.5 s. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh. |
|
|
581
|
+
| The census batch wrapper survives an inherited errexit (a caller-exported SHELLOPTS=errexit no longer turns an all-no-match batch into an input error), and the ARG_MAX regression leg proves its fixture exceeds the running host's exec limit by executing the single-exec shape and requiring E2BIG instead of trusting a fixed byte figure | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#must not misreport an all-no-match batch under an inherited errexit | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Third continuation lane (scope: the defects named in the preceding closeout). Observed failures: second continuation lane's challenge (codex, candidate ef0010e3) P2 findings 1 and 2. Applied mutations on disposable copies: the errexit-unsafe wrapper (`grep; rc=$?`) fails leg 16 under `env SHELLOPTS=errexit`; restoring the single-exec `$(cat …)` scan makes the fixture leg fail with E2BIG on this host (rc 126). Restored suite GREEN in 7.8 s. Supporting paths: scripts/reference-access-census.sh, scripts/test_reference_access_census.sh. |
|
|
582
|
+
| The entrypoint's relocation summary bullets keep every load-bearing obligation of the moved text inline (mini-program owner split, execution-only disclosure, the pending/out-of-scope disposition for uninspected token fields, the DFE-CONT validation and rename-sync clauses); shared-tree evidence carries neutral scope facts only, never a conversation transcript or adjudication narrative; and a transcript listing that fails without stderr is an input error that withholds the table | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#do not claim design-judgment extraction | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Third continuation lane (chain r6, codex, candidate 420ab84e): review findings 0 (summary bullets dropped obligations), 3 and the challenge finding (three evidence notes and three ledger rows carried conversation-level content on a shared surface — removed and reworded to scope facts), 4 (silent find failure). Applied mutation on a disposable copy: ignoring find's status makes the new leg 17 fail with exit 0 and the no-transcript sentinel where an input error is required; restored suite GREEN in 5.4 s. The entrypoint grows by the retained obligations (+80 body words) while its round delta stays negative. |
|
|
583
|
+
| The ARG_MAX regression leg derives its fixture size from the running host's exec limit (getconf ARG_MAX) and skips its E2BIG probe with a printed note when the limit exceeds the fixture bound, never asserting a fixed byte figure; the entrypoint summary bullets carry the relocated observable-proxy checklist, breakpoint scope, the deferred-evidence trigger set, and the bare-mention qualifier inline; and shared-tree evidence and ledger rows state scope facts without host census figures or conversation-level narrative | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#a bare mention of runtime or external access is not enough | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failures: the CI fast lane on the Linux runner failed the previous E2BIG probe (2.5 MB inventory below that host's exec limit while it exceeds macOS's 1 MiB) — the host-derived-threshold class the prior succession named; the prior succession's summary-bullet and shared-surface findings. Applied mutation on a disposable copy: restoring the single-exec scan fails the leg on this host (ARG_MAX 1 MiB, E2BIG); restored suite GREEN in 5.6 s. Rows citing host figures or narrative were corrected at their origin commits (the branch is unmerged; rows are never edited in place). Supporting paths: scripts/test_reference_access_census.sh, specs/111-extraction-skill-benchmark/evidence/continuation-lanes.md, specs/111-extraction-skill-benchmark/evidence/primary-source-excerpts.md. |
|
|
584
|
+
| The census last-touched column is the newest touching transcript's date and the counting leg asserts that exact date (and the older date on the other file) from deterministic mtimes, so selecting the oldest timestamp or emitting an unparsed value fails the suite | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#the last-touched column must be the newest touching transcript's date | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Fourth continuation lane review (codex) P2: successful-row assertions accepted any non-empty date. Fixed with `touch -t` mtimes on two transcripts and exact `alpha.md \| 2 \| <newest> \| 66%` / `beta.md \| 1 \| <older> \| 33%` assertions. Applied mutation on a disposable copy: selecting the oldest date (`sort \| head -1`) fails the leg; restored suite GREEN in 5.9 s. Supporting paths: scripts/test_reference_access_census.sh. |
|
|
585
|
+
| A transcript older than the census window moves no count, share, or last-touched date, and the suite proves it with a stale fixture: removing the mtime filter turns the denominator assertion red | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_reference_access_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Registered gap from the previous round's terminal succession (no test leg for out-of-window transcripts), reproduced against the current script with a control leg: with the filter present the stale fixture leaves every existing assertion green; with the filter removed the suite fails at the denominator assertion; restored tree green. Supporting evidence: `skill-extraction-workflow/scripts/test_reference_access_census.sh`. |
|
|
586
|
+
| Relocating entrypoint detail to a reference keeps every obligation under exactly one carrier: the row set is derived by the governing-chain tool, each derived obligation has one grep carrier under a named chain, every ledger anchor and script-pinned phrase survives verbatim, and no over-cap reference grows | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#must cover three axes, not only the executor's canonical phrase | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Zero-loss obligation map: `specs/113-extraction-entry-slim/obligation-preservation.md` (row set from `governing-chain-diff.py`; every survivor phrase counted once across the package with this ledger excluded). Supporting evidence: `skill-extraction-workflow/references/coverage-exhaustion-traps.md`, `skill-extraction-workflow/references/description-authoring.md`, `skill-extraction-workflow/references/external-practice-controls.md`. Paired control: the pinned-phrase, sync-pointer, contract-anchor, register firing-path resolution, and size gates report the same ok status on the base tree and on the landing tree, with the entrypoint body-word delta negative. |
|
|
587
|
+
| A usage-census mention share is not an open count: before a census figure justifies relocating, splitting, promoting, or retiring a file, the read shape in the same window is counted and the whole-read count carries the read-side cost argument | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#A mention count is not an open count | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Baseline (recorded incident): the previous round's closeout registered a ledger split as follow-up work on the ledger's mention share alone. With the rule applied: the same window's read-shape count showed whole-file loads in a small minority of the mentioning transcripts and bounded reads or gate echo in the rest, so the split was withdrawn as a read-side measure; the counts stay in the private charter. Supporting evidence: `skill-extraction-workflow/references/attention-budget-ratchet.md`. |
|
|
588
|
+
| Shared-tree guidance states a usage-census conclusion qualitatively; every measured ratio or count from a host census stays in the private charter, and a clause carried by a reference has exactly one carrier in that file | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/attention-budget-ratchet.md#the read shape must be counted in the same window | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: the first review round and the same-candidate challenge both found a measured ratio in the new read-shape bullet; the fix batch replaced it with the qualitative conclusion and reworded one cross-reference lead-in in the dual-track reference so the clean-only-oracle clause has a single carrier (obligation table row 74). Supporting evidence: `skill-extraction-workflow/references/attention-budget-ratchet.md`, `skill-extraction-workflow/references/dual-track-review-gate.md`, `specs/113-extraction-entry-slim/obligation-preservation.md`. |
|
|
589
|
+
| A suite case that deliberately leaves shared fixture state mutated must restore it inside the case and assert the restoration: a registry prune only drops registrations whose directory is already gone, so deleting that directory after the prune leaves a stale registration every following case inherits | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Registered gap carried open in a prior round's frozen closeout, treated as hypothesis and reproduced against the current suite with a control leg. RED baseline: with the case's own cleanup ordering unchanged, the added restoration assertion is the single failing case in the suite and every other case stays green, so the mutated registry was invisible to the existing assertions -- including the passing-state case that runs immediately after it. GREEN: deleting the directory before the prune turns the same assertion green with no other case changed. Sibling-form evidence: the failed-add case earlier in the same file already captures the pre-case registry and asserts equality, so the corrected case matches a form the suite already contains. Supporting evidence: `skill-extraction-workflow/scripts/test_review_ledger_binding.sh`. |
|
|
590
|
+
| The other finding class carried open in the same frozen closeout is recorded closed rather than fixed: a failed detached-checkout creation is released through the verified path and a suite case already asserts the registry is restored | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `unchanged` | Owner key `skill-extraction-workflow/SKILL.md`. The registration's predicate (a failed add only removes the directory and never unregisters or verifies) does not reproduce against the current baseline, so no code lands for it. Oracle sensitivity proven rather than assumed: replacing the failure branch's verified release with the predicate's own described behavior turns the owning failed-add case red, together with the newline-path case that also reads that release path, and no unrelated case fails; the tree was restored and re-verified green. Supporting evidence: `skill-extraction-workflow/scripts/review_ledger_binding.py`, `skill-extraction-workflow/scripts/test_review_ledger_binding.sh`. |
|
|
591
|
+
| A claim about a whole file's guidance form is a measurement, so it carries a runnable instrument that states its own definitions and is committed with its baseline reading before the edit it will judge; a figure recorded without its method supports no trend claim in either direction | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/rule-consolidation.md#A whole-file form claim carries a recorded ruler | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/rule-consolidation.md (form-by-failure wording rules) plus scripts/entrypoint_form_census.py and scripts/test_entrypoint_form_census.sh, registered in the lane every run exercises. Observed failure (recorded incident): an earlier round recorded this entrypoint's prohibitive-token count with no recorded counting method; recounting the same file this round by an independently written method produced a different figure over a different scope, so no trend could be claimed from the pair and the earlier number could not be used as a baseline. RED baseline (applied mutation, differential): making a blank line close a rule turns the continuation case red and no other case, restored green. The first draft of that case anchored on the rule count, which does not move under the defect because a dropped continuation is still not a new top-level bullet -- the mutation ran green against it, and the case was re-anchored on the token total, which does move. The instrument encodes no threshold. Supporting evidence: `skill-extraction-workflow/scripts/entrypoint_form_census.py`, `skill-extraction-workflow/scripts/test_entrypoint_form_census.sh`. |
|
|
592
|
+
| The density of imperative-negative vocabulary in a rule set is not a measure of its guidance form: a required-slot rule, a conditional keyed to an observable predicate, and a positive recipe all carry that vocabulary inside them, so the count names a set to classify one rule at a time and never a set to rewrite | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/rule-consolidation.md#The density of imperative-negative vocabulary is not itself a measure of form | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; merged into the form-by-failure wording rule that already owns the ruler requirement rather than appended as a second bullet. Observed failure (recorded incident): an earlier round inferred from this entrypoint's prohibitive-token-to-rationale ratio that the entrypoint did not apply its own form table, and registered a rewrite programme on that inference. Walked evidence: every rule the census flags was classified against the form-by-failure rows in `specs/114-entry-form-and-routing-baselines/form-classification.md`; the flagged set is overwhelmingly already in a form the table endorses, one row is mixed, none is in a form the table calls wrong, so the inference is withdrawn and no rewrite lands. Residual gap recorded rather than closed by writing: the discipline-slip rows need rationalization-vs-reality pairs quoted verbatim from baseline or pressure runs, a form this package prescribes in two places and realizes in none, with no capture channel operated -- the missing input is evidence, not authoring effort. Supporting evidence: `skill-extraction-workflow/references/rule-consolidation.md`, `specs/114-entry-form-and-routing-baselines/form-classification.md`. |
|
|
593
|
+
| A delivery that adds or removes a routing-bank row rebuilds the baseline in the same round or records that every prior baseline is now orphaned: the runner treats a differing bank fingerprint as a different ruler and suppresses the diff, so a bank edit does not age a baseline, it makes it permanently incomparable while it still reads as a baseline on disk | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#改 bank 的那一轮必须同轮重建基线 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure (recorded incident): the only full-bank report on disk was taken against a 136-row bank at one replica; the bank has since grown to 158 rows and the routing surface moved by six changed descriptions and one added skill, and no run in between could have detected it. RED baseline, produced by the runner itself rather than asserted: passing that report as `--baseline` to a fresh full-bank run emitted `baseline not compared — different ruler: bank content differs`, with both `newly_failed` and `newly_passed` empty because the comparison never ran. The new report carries `replicas: 3` and its `bank_sha256`, so a later round using the same bank and replica count is comparable to it; that is the property the rule exists to preserve. Second measured result from the same run, recorded because it is what the three-replica discipline buys: 13 of 14 failures carry `ownership_split` (the replicas disagreed and conservative consensus reports FAIL) and exactly one fails consistently across all three gradings, a distinction a single-replica report cannot express. Routing dispositions are deliberately not taken here -- editing a description while holding a routing measurement in the same round is the co-change shape the runner warns about. Supporting evidence: `eval/evidence/routing-baseline-replicas3-2026-09-03/`. |
|
|
594
|
+
|
|
595
|
+
Supersede note (round 114, ledger correction with no rule change): the row above beginning "Shared-tree guidance states a usage-census conclusion qualitatively" carries, in its evidence cell, an account of which review round and which challenge produced which finding. That is conversation-level process narrative on a shared surface, and the register is append-only, so the row stays byte-identical and is corrected here by pointer rather than edited. The obligation it records, restated at artifact level: `attention-budget-ratchet.md` carries the read-shape conclusion in qualitative form with no ratio or count; the measured census figures live only in the private charter; the clean-only-oracle clause has exactly one carrier in `dual-track-review-gate.md`, recorded as row 74 of `specs/113-extraction-entry-slim/obligation-preservation.md`. The superseded row's own behavioral-evidence declaration and firing-path anchor are unaffected. This is a note rather than a table row because it changes no rule and therefore has no owner-scoped anchor of its own to declare.
|
|
596
|
+
| An instrument whose stated definitions ARE its contract owes a case per definition, and the counting rule it documents must be the rule its figures were produced by: a regular-expression alternation matches leftmost and non-overlapping, so a listed phrase absorbs the words inside it and a comment promising independent counts describes a different ruler than the one that ran | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_entrypoint_form_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: two independent reviewer lenses on one candidate both read the census script's comment as promising that a phrase and the words inside it are counted separately, and no fixture exercised the overlap, so the documented rule and the reported totals were different rulers while the suite stayed green. The figures are unchanged by the correction because the behaviour was always leftmost non-overlapping and only the comment was wrong: the same run reports the same totals before and after. Two cases added, each with an applied mutation: a phrase-absorption case pins `must not` at one token, and an empty-Core-Rules case pins the false-empty guard -- disabling that guard turns exactly that case red with no other case failing, restored green. The classification artifact gains a stable per-rule identifier so each verdict maps to a rule without the generated JSON. Supporting evidence: `skill-extraction-workflow/scripts/entrypoint_form_census.py`, `skill-extraction-workflow/scripts/test_entrypoint_form_census.sh`, `specs/114-entry-form-and-routing-baselines/form-classification.md`. |
|
|
597
|
+
| A real-checkout regression accepts every documented terminal result of the candidate printer: a bounded hash, no change, or the packet-ceiling refusal, so a large merge-parent diff does not make the self-test reject correct gate behavior | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. RED baseline: a clean merge checkout whose parent diff exceeded the one-packet ceiling returned the documented `review packet exceeds 200000 bytes` refusal while the real-checkout assertion accepted only hash or no-change and failed. The same topology passes after the assertion accepts that exact refusal. Gate behavior and the packet ceiling are unchanged. Supporting evidence: `skill-extraction-workflow/scripts/test_review_ledger_binding.sh`. |
|
|
598
|
+
| A routing surface that omits a territory the owner's own body, its reference, or a sibling's Skip leg already assigns to it is a hole rather than a collision: the graders answer `none` or fall to the nearest-looking rival, and no amount of disambiguation between claimants fixes an utterance class nobody claims | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#you must use the R&D standards checklist in Workflow; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#product-rd-workflow: p3-resume-refactor moved 0/10 to 10/10| `updated` | Owner key `product-rd-workflow/SKILL.md`. Two frozen cases, each measured on paired trees differing only in the routing surface. The resume/continue utterance class was absent from a trigger list that already carried the whole redo family, and nine of ten gradings answered `none`. Separately, a stack with no architecture sibling had its service-boundary and data-ownership decisions assigned here by the dev skill's body and by this workflow's own reference, while the routing surface never said so and the two stacks that DO have architecture siblings claimed those words -- so the nearest-looking rival won 13/20 of the time. This entrypoint is at severe size debt with a zero-growth byte budget, so the two triggers were paid for rather than appended: sixteen separators compressed, one duplicate implement-phase phrasing dropped, and two sentences deleted from an unrelated standards bullet whose obligations the owning checklist reference already carries verbatim -- authority statement plus sync-gate, the testing-standard child doc with its layer and CI-gate coverage, and the execution-layer-versus-governing-Spec ruling. Net file size fell by 305 bytes. Both new triggers carry an English handle mirroring the redo family's, because dropping it would have left English-phrased requests of that class with no handle at all. The deletions are recorded as measured on the owner's whole case set rather than argued: every frozen case expecting this owner was re-run on paired trees because the edit reformatted the entire trigger list, not one token. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
599
|
+
| An owner whose territory is claimed only in the language the utterances are not written in is unreachable by keyword match, and scattered rivals at one hit each identify the defect as the owner's rather than a competitor's | `platform-observability` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-observability/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#platform-observability: p3-log-plus-test moved 2/10 to 10/10| `updated` | Owner key `platform-observability/SKILL.md`. Eight of seventeen gradings refused the compound logging/trace utterance outright and the remainder split across four rivals at one hit each. A second case had a sibling routing the production-state question here on both of that sibling's surfaces while this owner claimed it on neither of its own. Both moved to 10/10 on paired arms. The description was at 799 of 800 characters, so the addition required trimming a boilerplate sentence whose full obligation the body already carries verbatim. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
600
|
+
| A diagnosis owner that enumerates failure vocabulary but not degradation vocabulary reaches a slow-endpoint request only half the time, and the fix must be anchored to the degradation rather than to optimization, which is a delivery | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/defect-diagnosis/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#defect-diagnosis: p3-perf-plus-regression moved 5/10 to 10/10 | `updated` | Owner key `defect-diagnosis/SKILL.md`. The trigger list carried bug, 报错, test 挂了 and 线上问题 and nothing for an endpoint that got slower, so a request pairing that with a regression test reached the owner in five of ten gradings. The anchored form was chosen over a bare performance-optimization token deliberately: the bare token would have absorbed performance work that belongs to a multi-stage delivery owner. Paired arms, 10 replicas each, moved it to 10/10. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
601
|
+
| A client-stack owner that lists feature surfaces without the host toolchain refuses the toolchain question confidently, which is worse than refusing it with a clarify flag because nothing downstream signals that a route was missed | `miniapp-product-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/miniapp-product-dev/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#miniapp-product-dev: skip-miniapp-build moved 6/10 to 10/10 | `updated` | Owner key `miniapp-product-dev/SKILL.md`. The description enumerated pages, state, auth, sharing and review but not the host developer tool's build and on-device debug path, so four of ten gradings refused the build question and one of those refusals carried high confidence with no clarify flag. Paired arms moved it to 10/10. The pinned localized-refactor literal in the contract-anchor suite survives the edit unchanged. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
602
|
+
| A bare trigger token that names an activity also names its materials, and when five Skip legs point outward and none points back, the owner absorbs requests for the materials its own body says a sibling produces | `grill-me` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/grill-me/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#grill-me: ab-c5 moved 8/10 to 10/10 | `updated` | Owner key `grill-me/SKILL.md`. Four surfaces already assigned the question-pool deliverable to the sibling that produces it, including that sibling's own body table row naming the exact two fields the utterance asks for, and the always-on entry-routing layer. Only this description claimed the bare activity token. Anchoring it to the one-question-at-a-time form and adding the reciprocal Skip leg moved the materials case from 8/10 to 10/10 while the case that genuinely belongs here stayed 10/10 on both arms -- the narrowing was measured against the case it could have cost, not only against the case it was meant to fix. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
603
|
+
| A deliverable field named twice in a body but absent from the trigger list is not owned as far as routing is concerned, and the utterance that asks for both halves splits | `requirement-scope` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/requirement-scope/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#requirement-scope: ab-d6 moved 8/10 to 10/10| `updated` | Owner key `requirement-scope/SKILL.md`. The body names the compatibility and rollback boundary as a deliverable field in both its field table and its procedure, while the description carried only the version-slice half of the same utterance. At three replicas this case had been read as a collision with a release owner; at ten that rival appeared zero times and the only deviation was the lifecycle coordinator, which is why the fix is an addition here rather than a reciprocal Skip leg there. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
604
|
+
| A Skip list that enumerates six sibling destinations and omits the lifecycle coordinator leaves every request that continues past this owner's deliverable stranded on it | `requirement-doc-writer` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/requirement-doc-writer/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#requirement-doc-writer: p3-spec-then-tc claim added, case unmoved at 7/10| `updated` | Owner key `requirement-doc-writer/SKILL.md`. The body already routes multi-stage delivery and implementation or release planning to the coordinator; the description's Skip legs named six destinations and not that one. The alternative fix -- teaching the coordinator to claim sequential compound requests -- was rejected because two other frozen cases expect the leading-deliverable owner rather than the coordinator, so it would have bought one case with two. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
605
|
+
| A territory claimed inside a deliverable clause rather than in the trigger list is claimed where routing weight does not reach it; moving the same words changes the route without claiming anything new | `requirement-baseline` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/requirement-baseline/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#requirement-baseline: ab-b5 moved 14/20 to 20/20 | `updated` | Owner key `requirement-baseline/SKILL.md`. The code-evidence claim sat in the deliverable clause and the utterance asking for exactly that reached the owner in 14 of 20 gradings, with every deviation a refusal rather than a rival. This case is also the round's caution about its own decision rule: one ten-replica reading put it at 50 percent and the next twenty put it at 85 percent on the same candidate, so the threshold that separates a stable failure from a marginal case must be read off pooled observations. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
606
|
+
| A carve-out that pushes an utterance class away from one owner is only half a route: when the owner it points to never claims that class, the class has no home and the carve-out silently becomes a hole | `platform-release-engineering` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-release-engineering/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#platform-release-engineering: release-watch moved 18/20 to 20/20 | `updated` | Owner key `platform-release-engineering/SKILL.md`. Across the whole catalog only one description mentioned the on-call vocabulary, and it did so with an explicit carve-out excluding release watch and rollback; this owner, the destination of that carve-out, claimed none of it. Paired arms at twenty replicas moved the case from 18/20 to 20/20, and the one control-arm deviation had been a must-not-route-to violation rather than an ordinary miss. The pinned production-release literal in the contract-anchor suite survives the edit unchanged. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
607
|
+
| A measurement artifact produced below the resolution its own action rule requires reads as a findings list, and its consumers act on it; the floor has to be enforced by the instrument that writes the artifact, not by the prose that describes it | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The measurement protocol already required ten valid observations before a description edit, while the full-bank baseline its readers open is produced at three replicas -- a resolution the same protocol forbids acting on. In one round that ruler produced a wrong case-level reading in both directions: five cases it called failing are perfect at ten replicas, six drafted edits rested on them, one deviation supported an argument about a stale frozen expectation that had to be withdrawn, one neighbour was scored as a fresh regression that the paired control shows on the unedited tree, and three full-bank runs on near-identical candidates returned almost disjoint newly-failed sets. The runner now writes an explicit resolution field and prints a screening banner, the reference states the same floor plus the pooled-observation and candidate-list rules, and the new suite pins the doc and the executable to one number with three applied mutations each turning it red on its own assertion. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
608
|
+
| A resolution floor stated over VALID observations but computed from the requested sample size is not the floor it claims: a run that asked for ten replicas reports itself actionable while a timeout or an unparsable answer leaves a case short, so the gate passes exactly the evidence it exists to refuse | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Independent review raised this against the first cut, and the round that landed it had already hit the shape it misses: a fourteen-task run at ten replicas returned six grader errors and left two tasks at seven valid observations, and the re-run was owed by hand rather than signalled by the report. The field now derives from the weakest task's valid-observation count -- the weakest governs because a per-case edit is licensed per case -- and the report exposes that count so a consumer is not left reading the request. Reverting the computation to the requested replica count turns the new case red on its own assertion and restoring it returns green. The regression fixture drives a grader that fails one utterance persistently, because the runner already retries a single unparsable answer as a sampling accident and a fixture that fails once is silently repaired -- a detail found by watching a ten-replica fixture spend eleven calls. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`. |
|
|
609
|
+
|
|
610
|
+
Supersede note (round 115, ledger correction with no rule change): the eleven rows appended by this round each name one owner and the case that owner's description was edited for, in a form that reads as though the measured move were attributed to that description. Independent review held the record to the single-variable rule this same round lands — a comparison that moved several descriptions at once cannot be attributed to any one of them — and the rows stay byte-identical because the register is append-only, so the correction is by pointer. Restated at the level the evidence supports: the control is the branch base and the treatment is the landing candidate carrying all eleven edits, so every per-owner move in those rows is a PACKAGE result with a per-case interpretation attached. The interpretation is supported, not established, by two facts recorded in `eval/evidence/routing-115-underclaim-fix-2026-09-03/paired-measurements.tsv`: each case's expected owner has exactly one changed description, and no other changed owner appears anywhere in that case's observed verdict distribution on either arm. Exactly one of the eleven carries the isolation the protocol asks for: `ab-c5` measured 6/9 on a four-edit tree (catalog 31c9a686…) and 10/10 on that tree plus only the grill-me description (catalog 71a6793d…). The other ten were not split into per-description A/B trees, and that is this round's residual rather than a claim it makes. The rows' behavioral-evidence declarations, firing paths and bank-evidence locators are unaffected.
|
|
611
|
+
|
|
612
|
+
Supersede note (round 115, second ledger correction with no rule change): the `product-rd-workflow` row appended by this round records that the two sentences deleted from its standards bullet had their obligations carried by the owning checklist reference. Adversarial review checked that mapping clause by clause and it was incomplete in two places: the reference required an authority statement for a multi-doc family but not that a cross-stack product Spec live in exactly one authority surface with execution slices linking back rather than redefining its goals, and it required a testing standard child doc but not that stack documents may not replace the shared test-layer and CI-gate policy that standard owns. Both obligations are now stated in `product-rd-workflow/references/rd-standards-doc-family-checklist.md` items 5 and 6, where the placement rule puts detail, rather than restored to a severe-debt entrypoint under a zero-growth byte budget. The row stays byte-identical because the register is append-only. The lesson the round records against itself: a zero-loss map asserted at paragraph granularity passed, and the same map checked at clause granularity did not.
|
|
613
|
+
| A trigger removed on a synonym argument is removed on the author's reading, not on evidence: when no frozen case exercises the removed token, the bank cannot refute the argument and the narrowing lands unmeasured | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md#must not** redefine the product goals it owns; bank-evidence: downscoped:R115-TRIGGER-RESTORE-NO-FROZEN-CASE | `updated` | Owner key `product-rd-workflow/SKILL.md`. Paying this entrypoint's zero-growth byte budget, the round dropped one trigger as a synonym of a surviving one. Adversarial review supplied the counterexample the bank could not -- a request that names the transition without the approval wording -- and the trigger is restored inside the same budget. The same review checked the round's zero-loss map for two sentences deleted from a standards bullet and found it incomplete at clause granularity: the one-authority-surface rule and the prohibition on stack documents replacing shared test-layer and CI-gate policy are restored to `product-rd-workflow/references/rd-standards-doc-family-checklist.md` items 5 and 6, where the placement rule puts detail. Supporting evidence: `product-rd-workflow/references/rd-standards-doc-family-checklist.md`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
614
|
+
| A resolution verdict published once per report cannot license anything per case: a subset run that clears the floor for the cases it graded reads, to a consumer holding only the report-level field, as licence for an edit to a case the run never measured | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Adversarial review found the field this round had just landed to be report-wide while the obligation it enforces is per case. Each result now carries its own `actionable` and `valid_observations`, the report-level field is the conjunction over the cases actually graded, and a scope note says so in the report itself. Reverting to the report-wide computation turns the new case red on its own assertion and restoring it returns green. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`. |
|
|
615
|
+
| A fixture that cannot separate the property under test from the defect it guards against turns an applied mutation into theatre: the suite goes red on an unrelated assertion, the author records a proof, and the property was never pinned | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The resolution suite carried one case, and with one case a per-case verdict and a report-wide conjunction produce identical output, so the mutation this round ran to prove the per-case field went red on a different assertion and proved nothing. The fixture now carries two cases and degrades only one; substituting the conjunction for the per-case field fails on the assertion that the healthy case stays actionable, which is the discrimination the earlier fixture could not make. The same review found the round's own must-not claim read from one probe while a second run on the same tree carried the single forbidden verdict; the record now states the pooled counts on both arms. Supporting evidence: `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
616
|
+
| A property a reviewer cannot reach from the bounded packet is unverified even when the code is correct: the degenerate shapes that would expose an absent verdict list live in a construction site the packet excludes, so the check has to be moved into the suite rather than argued in the response | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Review could not establish from its packet that every result carries a verdict list, because the loop that builds one is outside the bounded diff; reading the source shows a single append site that always sets it, which answers the question for the author and for nobody else. The suite now exercises the two shapes that would expose an absent list -- a single-replica run, and a task whose every replica fails -- asserting neither crashes and neither reports itself actionable. Capturing the wholly-failed run's documented exit 3 required taking the status in the same command that produces it: under `set -e` the suite dies before the assignment, which is how the first version of this case reported nothing at all. Supporting evidence: `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
617
|
+
| A token that names a lifecycle phase does not only claim requests about that phase; it recolours the owner's whole description for a model reading the catalog, so the same token that catches one request class can push an unrelated frozen case toward a rival — the trade has to be measured on both, not argued from what the token says | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#description; bank-evidence: downscoped:R115-TRIGGER-RESTORE-NO-FROZEN-CASE | `updated` | Owner key `product-rd-workflow/SKILL.md`. Restoring a bare implementation-phase token after review moved a frozen spec-writing case from 10/10 on the prior candidate to 13/20, with the requirement-document owner as the rival; the two trees differed by that token alone, which is the single-variable A/B the protocol asks for and the round otherwise lacked. Merging the token with its approval-wording neighbour recovers the case to 15/20 — level with its 14/20 control — while the review counterexample still routes 20/20 and the approval-wording case stays 20/20. A further attempt to lift the case by sharpening the rival's Skip leg reached 20/20 there but moved a different frozen case from 19/20 to 15/20, and was withdrawn under the rule that average improvement may not offset a single frozen case; it also rested on the rate rather than on a surface asymmetry. The downscope declaration is restated for the merged form in the round's spec. Supporting evidence: `specs/115-routing-underclaim-fix/downscope.md`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
618
|
+
| A zero-loss map that names the artifact but not its owner has not preserved the obligation: routing a template to a skill is not the same as that skill owning the standard and the policy it carries, and a reader who holds the template can approve the policy without ever satisfying the original ownership clause | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md#owned by `testing-strategy`** — not by whichever document holds the template | `updated` | Owner key `product-rd-workflow/SKILL.md`. Adversarial review checked the round's second restoration of the deleted standards sentences and found a third gap: the reference said the testing-standard template routes to the testing owner, while the deleted entrypoint sentence had said the standard is OWNED by it. The reference now states the ownership of the standard and of the shared test-layer and CI-gate policy explicitly. The same review corrected four record errors the round had introduced while consolidating its evidence -- settled-tree values doubled by intermediate-tree rows in one table, a causal conclusion drawn about a must-not observation that the evidence cannot support either way, a probe utterance absent from the packet, and no per-run selection manifest against which a dropped case would show. Each is fixed in the record; none changes a routing surface. Supporting evidence: `product-rd-workflow/references/rd-standards-doc-family-checklist.md`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
619
|
+
|
|
620
|
+
Supersede note (round 115, third ledger correction with no rule change): the first supersede note above restates this round's per-owner rows as a package result supported by two facts, one of which the committed tables contradict — it says no other changed owner appears in any claimed case's observed distribution, while `p3-spec-then-tc` selects the edited `requirement-doc-writer` on both arms, `p3-log-plus-test` selects the edited `product-rd-workflow` on the control arm, and `ab-c5`'s only rival is the edited `grill-me`. The support that survives is the weaker fact alone: each claimed case's expected owner has exactly one changed description. The single isolated comparison (`ab-c5`, 6/9 → 10/10 across two trees differing only by the grill-me description) stands. Rows and the earlier note stay byte-identical because the register is append-only.
|
|
621
|
+
| A verdict that parses is not yet an observation: an answer naming a skill the catalog does not carry says nothing about the route, and a floor that counts it as usable can be satisfied by ten well-formed non-answers | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Adversarial review could not see from its packet whether the resolution counter validated a parsed selection, and reading the source showed it did not: any status other than ERROR counted, so a parseable verdict naming a non-catalog skill was a FAIL that still fed the floor. Such a verdict is now an ERROR -- absence of evidence, not evidence against the route -- and a fixture whose grader returns a well-formed answer naming no catalog skill must leave every case at zero valid observations and non-actionable; removing the validation turns that case red on its own assertion. All 4,905 non-error verdicts in the round's archived reports name a catalog skill or none, so no landed number moves. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
622
|
+
| Two lists that must agree and are built by two transformations will drift; the set of names a verdict may select has to be the same filtered list the prompt was built from, or a skill the grader was never shown becomes a valid answer | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Review noted that the selectable-name allow-list added earlier in this round was built from every directory carrying a SKILL.md while the prompt catalog was built by a separate transformation that drops entries without a usable description, so the two could disagree on exactly the entries the prompt omits. Both now derive from one filtered list. The fixture adds a skill whose description is empty and a grader that selects it; the verdict must be an ERROR, and rebuilding the allow-list from the unfiltered directory turns that case red on its own assertion. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
623
|
+
| A rewrite that keeps behaviour but raises the language floor of a shared tool is a change the tool's other hosts pay for, and the round that makes it owes either the floor or the restoration | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/eval-routing-bank.rb | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Review noted that the catalog build this round rewrote had moved from map plus compact to filter_map, which needs Ruby 2.7; the runner carried no such call at the branch base, the repository states no Ruby floor, and the round had not measured which hosts run it. The touched lines are restored to map plus compact so the round raises no floor the base did not already have; the gate script's own pre-existing uses are outside this round and unchanged. The resolution suite exercises the rebuilt catalog path on every run. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
624
|
+
|
|
625
|
+
Supersede note (round 115, fourth ledger correction with no rule change): the third note above names as the surviving support that each claimed case's expected owner has exactly one changed description. Adversarial review refuted it with `ab-c5`, whose expected owner `requirement-intent` is untouched and whose edit is on the rival `grill-me`. The support that survives is: for each claimed case exactly one description was edited for it, the expected owner's in ten cases and the rival's in one. The evidence directory now carries the frozen bank records for every measured case so expected owner, acceptable alternatives and must-not guards are checkable against the untouched bank rather than against this record. Rows and earlier notes stay byte-identical.
|
|
626
|
+
|
|
627
|
+
Supersede note (round 115, fifth ledger correction with no rule change): the routing-hole row above says the nearest-looking rival won 13/20 of the time for the Node architecture case. That misreads the base run: 13/20 is the count of replicas that selected the expected owner; the rival `go-microservice-architecture` took one replica and `none` took six (`paired-measurements.tsv`, run ctrl-J; the 10-replica base run split 6 expected / 4 none). The dominant failure was refusal, not the rival — which is the row's own thesis, so the lesson stands and only the distribution is corrected.
|
|
628
|
+
|
|
629
|
+
Supersede note (round 115, sixth ledger correction with no rule change, two rows): (a) the language-reachability row above says the logging/trace utterance's non-refusing gradings split across four rivals at one hit each. The pre-A and pre-A2 base runs total seventeen gradings: eight refusals, six selections of the expected owner `platform-observability`, and three rivals at one hit each (`product-rd-workflow`, `python-service-dev`, `nodejs-service-dev`). Scattered single-hit rivals remain the row's point; the count is corrected. (b) The release-watch row says the one control-arm deviation had been a must-not-route-to violation. Two base-tree control runs exist: ctrl-J (18/20, the run the row's move is measured from) deviated twice to `release-coordination`, an ordinary miss; ctrl-H (19/20) deviated once to `platform-observability`, the forbidden owner. The violation is real but belongs to the other control run.
|
|
630
|
+
|
|
631
|
+
Supersede note (round 115, seventh ledger correction with no rule change): the observation-validity row above says all 4,905 non-error verdicts in the round's archived reports name a catalog skill or `none`. That total is counted over the archived runner reports in the maintainer's scratch directory, which are not committed and cannot be recounted from this repository. What the repository does carry is `replica-verdicts.tsv`, whose 2430 non-error verdicts all resolve to a catalog skill or `none`; the archive-wide figure stands as the author's count, not as committed evidence.
|
|
632
|
+
| A round partition that depends on which ref is judged moves verdicts after they land: a branch judged per ledger commit as a pull request collapsed to one round once merged, so a row valid at pull-request time (a routing-surface `#description` anchor in a commit that changed only the description) was refused on every post-merge evaluation with nothing about it changed; a merge git rebuilds from its two parents must be expanded into the branch's own rounds so the same history partitions identically before and after it lands, and a merge git cannot rebuild keeps one boundary so content from neither parent is never left in no round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/impact-chain-gate.rb, its round-scoping fixtures, the verdict differential's named divergences, references/external-practice-controls.md and the .github/workflows/ci.yml checkout comments). Observed failure: the integration branch's promotion pull request reported `impact_chain_firing_path_missing` for a row that was green on its own pull request (15 rounds on the branch head, one after the merge), and the integration branch's push build had been red since that merge. RED baseline: round scoping 8 now runs the branch view and the merged view on one fixture and asserts them equal; on the previous gate it fails with `expected rc=1 got rc=0`, and the real promotion shape (integration head against the target) goes from rc=1 to rc=0 with no other change. Verdict differential: 64 integration points, six newly refused, each named by sha with `impact_chain_gate_missing` — all merges from before CI checked out the branch head, whose branches carry owner work outside the round that declares it; none newly accepted. Round scoping 13 pins the observed shape (body round then description-only round, merged) green and equal to its branch view; round scoping 14 pins that a merge whose tree is not the automatic merge keeps a single boundary. Refines the row above beginning "确定性闸对历史形态有前提": the checkout ref binding stays, and the partition no longer depends on it. |
|
|
633
|
+
| A landing chain that looks for a round's review evidence only inside that round's own checkout can never bind a round that merged without its ledger, however honestly the same bytes are reviewed later: evidence is a validator-accepted closeout whose candidate hash equals the round's packet, so the chain reads the landing tree's committed evidence, and a later review of exactly those bytes, landed as a round of its own, binds the earlier round — a closeout for any other digest binds nothing, wherever it sits | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Observed failure: the same promotion pull request's chain refused at a two-file round that had merged with the binder red (`no accepted review evidence binds the landing candidate`), and no forward path existed because the rebind enumerated evidence from the round's detached checkout. RED baseline: the chain case "a validator-accepted closeout for the round's own candidate, committed on the integration branch after the merge, binds it through the chain" fails on the previous binder and passes on this one; the companion case with a closeout for a different digest is refused on both. The round's packet is still frozen from its own checkout at its own base with the landing tree's controller, and its excludes still come from its own added receipts, so a later ledger's absence in the round checkout leaves the round's hash unchanged. The retrospective review of that round's bytes is committed in this round's evidence directory (its retro-round folder) and binds its candidate hash. |
|
|
@@ -177,3 +177,14 @@ Use this when validating that an extraction skill or design/client skill actuall
|
|
|
177
177
|
3. Record what changed in the extraction method before the test, then run the source evidence through the required chain: observation, judgment, rule, acceptance.
|
|
178
178
|
4. Compare the result against current skills. If the source reveals a gap, update the smallest owning skill/reference. If it confirms existing guidance, record that no new rule was needed.
|
|
179
179
|
5. Mark the coverage honestly: targeted pressure test, file-level refresh, node/artifact inventory, or full workflow extraction. Never upgrade a targeted pressure test into a full-source claim.
|
|
180
|
+
|
|
181
|
+
## Entrypoint obligations
|
|
182
|
+
|
|
183
|
+
The six rules below are the entrypoint-level obligations for UI/UX, Figma, frontend, app, miniapp, and client sources. They were relocated verbatim from `SKILL.md`'s `What to extract` Core Rules group (low-frequency detail per the entrypoint's content-placement rule); the entrypoint keeps a one-bullet summary that points here, and Step 6's UI/UX validation rows resolve against this section. Wording changes here go through the same shared-skill gates as an entrypoint edit.
|
|
184
|
+
|
|
185
|
+
- Design/client extraction must cover the judgment layer, not only the engineering layer. For UI/UX, extract aesthetic logic, interaction logic, behavioral logic, and user psychology from source evidence before landing rules about layout, components, breakpoints, or tests.
|
|
186
|
+
- UI/UX judgment extraction must use observable proxies, not adjectives. Read state families, navigation/entry/return paths, disabled reasons, recovery controls, timing/feedback, accessibility, responsive/device variants, and code state machines before claiming behavioral or psychology rules. Use `references/uiux-judgment-extraction.md` for the required method.
|
|
187
|
+
- UI/UX lessons usually route to multiple owners. Before editing, map each candidate to design, web, app, miniapp, testing, product workflow, or this extraction workflow using `references/uiux-routing-map.md`; do not land only the design rule when implementation or scenario testing is required. For mini-program lessons, `testing-strategy` owns layer/scenario selection, while `miniapp-product-dev` owns host-platform implementation, developer-tool or real-device evidence, review/release mechanics, and miniapp runtime constraints.
|
|
188
|
+
- Judgment-layer extraction must name what changed. For UI/UX/client sources, record whether each judgment layer produced a new rule, confirmed an existing rule, narrowed an existing rule, or found no new evidence. If the pass only improves execution/validation, say so instead of implying new aesthetic, behavioral, psychology, or interaction knowledge.
|
|
189
|
+
- For UI/UX/client extraction, the judgment-dimension axis enumeration lives in `references/uiux-judgment-extraction.md`. When the adjacency-scan rule fires on a UI/UX source, walk that enumeration — do not re-derive the axis list from memory.
|
|
190
|
+
- A UI/UX judgment-delta row is not complete with labels such as `confirmed`, `narrowed`, or `no new evidence` alone. Each visual direction/tokens row must satisfy the field list in `references/uiux-judgment-extraction.md`; if those fields were not inspected, mark the row `pending` or `out of scope` and do not claim design-judgment extraction.
|
|
@@ -54,6 +54,17 @@ When the skill is installed outside the source repo, resolve the installed `skil
|
|
|
54
54
|
|
|
55
55
|
If the validator reports `missing_required_command`, keep the failure visible and complete the static validation bullets manually.
|
|
56
56
|
|
|
57
|
+
## Read-modify-write example code
|
|
58
|
+
|
|
59
|
+
Relocated verbatim from `SKILL.md`'s `Validation & the dual-track gate` Core Rules group (the entrypoint keeps a one-bullet summary that points here); it applies to every reference file or skill example that ships runnable code against external mutable state, and wording changes here go through the same shared-skill gates as an entrypoint edit.
|
|
60
|
+
|
|
61
|
+
- Reference example code that performs read-modify-write on external mutable state (Bitable records, database rows, file contents, API state) must:
|
|
62
|
+
- Read failure: raise explicitly; never return `{}`, `""`, `None`, or any empty-success value that silently drops the prior state.
|
|
63
|
+
- Write failure: raise or skip, never silently continue.
|
|
64
|
+
- Uniqueness invariant: when the example declares, assumes, or depends on one, detect violations such as duplicate unique keys and raise before propagating bad state.
|
|
65
|
+
- Lost-update control: use optimistic concurrency controls (ETag, version field, CAS, transaction, or compare-and-swap), append-only API semantics, or an explicitly declared single-writer precondition — RMW examples that silently assume no concurrent writers will produce lost-update bugs under normal conditions.
|
|
66
|
+
- Data-loss anti-pattern: log-and-continue after a read failure on an append-only field.
|
|
67
|
+
|
|
57
68
|
## Behavioral Validation
|
|
58
69
|
|
|
59
70
|
- For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology — **and the firing point is BEFORE drafting the body, not only before finalizing**. Eval-first authoring for a NEW skill (or a new hard-rule section): (1) write the evaluation scenarios first — **at least three** for a new skill (both the vendor's published authoring guide and the high-star practice pack converge on three-plus scenarios before body text; a single-rule edit may scope down to that rule's own scenario); (2) run them WITHOUT the skill and record the observed failures verbatim — and for a discipline-slip failure (the agent knows the rule and skips it under pressure), capture the agent's rationalizations word-for-word: each verbatim excuse is the raw material for one rationalization-vs-reality row and one red-flag line in the skill text (the discipline-slip form in `rule-consolidation.md`'s form-by-failure table); an invented hypothetical excuse does not qualify — counter only what a run actually said, and don't add rows for excuses no run produced; **a no-skill control that does not exhibit the failure is a stop signal — do not author guidance for a failure you cannot observe** (record the null finding instead; this is the pre-draft face of "Evidence must come before new rules"); (3) draft the **minimal** content that addresses the observed failures, then re-run the same scenarios WITH the skill; (4) when a later run, review round, or live miss surfaces a NEW rationalization for an existing discipline gate, add its explicit counter row to that gate's table and re-run the tempting scenario — counter tables accrete from observed excuses across rounds, never from imagination. The code-level RED-GREEN-REFACTOR method (write the failing case first, watch a fresh agent violate the rule WITHOUT the skill, then add the skill and watch it comply) is owned by `superpowers:writing-skills` + `superpowers:test-driven-development` — **if installed, route there; otherwise apply the RED-baseline rule inline** (manually record the without-change failure and the with-change compliance). This is the skill-authoring face of **eval-driven development** (for a behavior/routing change, run the scenario before you finalize; never special-case the scenario just to make it pass) — borrow the *principle*, not a claim of production-grade eval rigor.
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Count the guidance FORM of an entrypoint's rules, so form claims carry a ruler.
|
|
3
|
+
|
|
4
|
+
`references/rule-consolidation.md`'s form-by-failure table picks the guidance form
|
|
5
|
+
from the baseline failure a rule answers. Judging whether an entrypoint follows
|
|
6
|
+
its own table needs a number, and the number is worthless unless the next round
|
|
7
|
+
can recompute it: an earlier round recorded a prohibitive-token count with no
|
|
8
|
+
recorded method, and a later round counting the same file by a different method
|
|
9
|
+
got a different figure, so no trend could be claimed in either direction. That is
|
|
10
|
+
the failure this script exists to prevent -- not the counting itself, which is
|
|
11
|
+
easy, but the counting being reproducible.
|
|
12
|
+
|
|
13
|
+
What is counted, stated here because the definition IS the instrument:
|
|
14
|
+
|
|
15
|
+
- A `rule` is one top-level `- ` bullet inside `## Core Rules`, together with
|
|
16
|
+
every continuation and sub-bullet line up to the next top-level bullet or the
|
|
17
|
+
next heading. Sub-bullets are not separate rules; they are part of the rule
|
|
18
|
+
whose form is being judged.
|
|
19
|
+
- A `prohibitive token` is a match of PROHIBITIVE_RE: the imperative-negative
|
|
20
|
+
vocabulary the form table calls the right form for a discipline slip and the
|
|
21
|
+
wrong form for every other baseline failure. Case is significant only where
|
|
22
|
+
the capitalised spelling is itself the emphasis (MUST / NEVER / ALWAYS).
|
|
23
|
+
- A `named baseline failure` is a match of FAILURE_SHAPE_RE: the rule states the
|
|
24
|
+
observed failure it answers, rather than only the prohibition. The form table
|
|
25
|
+
needs the baseline failure to pick a form, so a prohibition with no named
|
|
26
|
+
failure is a rule whose form was never derived from anything.
|
|
27
|
+
|
|
28
|
+
The reported diagnostic is `unanchored_prohibition_rules`: rules carrying at
|
|
29
|
+
least one prohibitive token and no named baseline failure. That is the set the
|
|
30
|
+
form table has something to say about; it is not a defect count, because a
|
|
31
|
+
discipline-slip rule is legitimately a prohibition -- it is the set a form pass
|
|
32
|
+
must classify one by one.
|
|
33
|
+
|
|
34
|
+
Exit status is 0 whenever the file parses; this is an instrument, not a gate.
|
|
35
|
+
Nothing here decides whether a form is right, and no threshold is encoded: a
|
|
36
|
+
threshold would make the ruler an argument for its own reading.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
from __future__ import annotations
|
|
40
|
+
|
|
41
|
+
import argparse
|
|
42
|
+
import json
|
|
43
|
+
import re
|
|
44
|
+
import statistics
|
|
45
|
+
import sys
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
|
|
48
|
+
# The imperative-negative vocabulary. Alternation is leftmost and matches do not
|
|
49
|
+
# overlap, so the LONGEST listed spelling wins at a position and the words inside
|
|
50
|
+
# it are not counted again: "must not" is one token, not "must not" plus "must".
|
|
51
|
+
# The phrase alternatives are therefore listed before the words they contain, and
|
|
52
|
+
# that ordering is load-bearing rather than cosmetic. The figure counts token
|
|
53
|
+
# occurrences, never distinct rules.
|
|
54
|
+
PROHIBITIVE_RE = re.compile(
|
|
55
|
+
r"\bMUST NOT\b|\bMUST\b|\bNEVER\b|\bALWAYS\b"
|
|
56
|
+
r"|\bmust not\b|\bmust\b|\bnever\b|\bcannot\b|\bcan not\b"
|
|
57
|
+
r"|\bdo not\b|\bdon't\b|\bdoes not\b|\bmay not\b|\bshall not\b"
|
|
58
|
+
r"|\bforbid(?:s|den)?\b|\bprohibit(?:s|ed)?\b|\bno[tn]-negotiable\b"
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# The rule states the failure it answers. These are the phrasings this package
|
|
62
|
+
# already uses for that job; a rule that names its baseline failure some other
|
|
63
|
+
# way reads as unanchored here, which biases the diagnostic toward over-reporting
|
|
64
|
+
# rather than under-reporting -- the safe direction for a set meant to be walked.
|
|
65
|
+
FAILURE_SHAPE_RE = re.compile(
|
|
66
|
+
r"failure shape|failure-shape|failure mode|the failure it prevents"
|
|
67
|
+
r"|recurring shape|observed failure|the exact .{0,40}failure"
|
|
68
|
+
r"|recurrence signal|the tell that|the defect this prevents"
|
|
69
|
+
r"|failure it prevents|the dodge this prevents",
|
|
70
|
+
re.IGNORECASE,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
BOLD_RE = re.compile(r"\*\*[^*]+\*\*")
|
|
74
|
+
CORE_RULES_HEADING = "## Core Rules"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def parse_rules(text: str) -> list[dict]:
|
|
78
|
+
"""Return one record per top-level bullet inside `## Core Rules`."""
|
|
79
|
+
lines = text.splitlines()
|
|
80
|
+
try:
|
|
81
|
+
start = next(i for i, l in enumerate(lines) if l.strip() == CORE_RULES_HEADING)
|
|
82
|
+
except StopIteration:
|
|
83
|
+
raise SystemExit(f"entrypoint_form_census_error: no {CORE_RULES_HEADING!r} heading")
|
|
84
|
+
# The section ends at the next same-level heading.
|
|
85
|
+
end = len(lines)
|
|
86
|
+
for i in range(start + 1, len(lines)):
|
|
87
|
+
if lines[i].startswith("## "):
|
|
88
|
+
end = i
|
|
89
|
+
break
|
|
90
|
+
|
|
91
|
+
rules: list[dict] = []
|
|
92
|
+
current: dict | None = None
|
|
93
|
+
group = None
|
|
94
|
+
for line in lines[start + 1 : end]:
|
|
95
|
+
if line.startswith("### "):
|
|
96
|
+
group = line[4:].strip()
|
|
97
|
+
current = None
|
|
98
|
+
continue
|
|
99
|
+
if line.startswith("- "):
|
|
100
|
+
current = {"group": group, "lines": [line]}
|
|
101
|
+
rules.append(current)
|
|
102
|
+
continue
|
|
103
|
+
if current is not None:
|
|
104
|
+
# A blank line does not close a rule: sub-bullets and continuations
|
|
105
|
+
# are separated by blanks in this file, and treating a blank as a
|
|
106
|
+
# terminator would split rules and inflate the rule count.
|
|
107
|
+
current["lines"].append(line)
|
|
108
|
+
return rules
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def measure(rule: dict) -> dict:
|
|
112
|
+
body = "\n".join(rule["lines"])
|
|
113
|
+
prohibitions = PROHIBITIVE_RE.findall(body)
|
|
114
|
+
named_failure = bool(FAILURE_SHAPE_RE.search(body))
|
|
115
|
+
return {
|
|
116
|
+
"group": rule["group"],
|
|
117
|
+
"head": rule["lines"][0][2:][:80],
|
|
118
|
+
"words": len(body.split()),
|
|
119
|
+
"lines": len(rule["lines"]),
|
|
120
|
+
"prohibitive_tokens": len(prohibitions),
|
|
121
|
+
"bold_spans": len(BOLD_RE.findall(body)),
|
|
122
|
+
"names_baseline_failure": named_failure,
|
|
123
|
+
"unanchored_prohibition": bool(prohibitions) and not named_failure,
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def main() -> int:
|
|
128
|
+
ap = argparse.ArgumentParser(description=__doc__)
|
|
129
|
+
ap.add_argument(
|
|
130
|
+
"path",
|
|
131
|
+
nargs="?",
|
|
132
|
+
default=str(Path(__file__).resolve().parent.parent / "SKILL.md"),
|
|
133
|
+
help="entrypoint to measure (default: this package's own SKILL.md)",
|
|
134
|
+
)
|
|
135
|
+
ap.add_argument("--json", dest="json_path", help="write the full per-rule table here")
|
|
136
|
+
args = ap.parse_args()
|
|
137
|
+
|
|
138
|
+
text = Path(args.path).read_text(encoding="utf-8")
|
|
139
|
+
records = [measure(r) for r in parse_rules(text)]
|
|
140
|
+
if not records:
|
|
141
|
+
raise SystemExit("entrypoint_form_census_error: no rules parsed")
|
|
142
|
+
|
|
143
|
+
words = [r["words"] for r in records]
|
|
144
|
+
unanchored = [r for r in records if r["unanchored_prohibition"]]
|
|
145
|
+
summary = {
|
|
146
|
+
"path": args.path,
|
|
147
|
+
"rules": len(records),
|
|
148
|
+
"rule_words_median": int(statistics.median(words)),
|
|
149
|
+
"rule_words_max": max(words),
|
|
150
|
+
"rules_over_300_words": sum(1 for w in words if w > 300),
|
|
151
|
+
"prohibitive_tokens": sum(r["prohibitive_tokens"] for r in records),
|
|
152
|
+
"bold_spans": sum(r["bold_spans"] for r in records),
|
|
153
|
+
"rules_naming_baseline_failure": sum(1 for r in records if r["names_baseline_failure"]),
|
|
154
|
+
"unanchored_prohibition_rules": len(unanchored),
|
|
155
|
+
}
|
|
156
|
+
for key, value in summary.items():
|
|
157
|
+
print(f"{key}={value}")
|
|
158
|
+
print("entrypoint_form_census_ok")
|
|
159
|
+
|
|
160
|
+
if args.json_path:
|
|
161
|
+
Path(args.json_path).write_text(
|
|
162
|
+
json.dumps({"summary": summary, "rules": records}, ensure_ascii=False, indent=2) + "\n",
|
|
163
|
+
encoding="utf-8",
|
|
164
|
+
)
|
|
165
|
+
return 0
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
if __name__ == "__main__":
|
|
169
|
+
sys.exit(main())
|