@ccoalm/ccl-skills 0.8.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +5 -0
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +6 -0
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +5 -0
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -0
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/feature-risk-router/SKILL.md +3 -1
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/architecture-playbook.md +1 -1
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/multi-tenant-isolation.md +1 -1
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/state-machine-task-patterns.md +2 -0
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/inference-capacity-operations.md +24 -0
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/llm-client-gateway.md +1 -1
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +4 -1
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/contracts-and-state.md +5 -0
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +1 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -2
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/metrics-conventions.md +8 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/sli-slo-design.md +2 -2
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/canary-and-rollout-strategy.md +16 -2
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/promotion-gate-and-review.md +9 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +12 -12
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/code-review-checklist.md +4 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +1 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +1 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +2 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +2 -2
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +1 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +8 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/multi-tenant-isolation.md +1 -1
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/state-machine-task-patterns.md +2 -0
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/SKILL.md +1 -1
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +12 -15
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +37 -0
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +13 -0
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +37 -30
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +24 -3
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +5 -5
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -1
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +81 -0
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +12 -0
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +1 -1
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +30 -0
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-contract-anchors.sh +126 -0
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +197 -1
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/contract-anchors.tsv +15 -0
  44. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +210 -36
  45. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +3 -3
  46. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/gate_receipt.py +576 -0
  47. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_antipattern_grep_panel.sh +80 -0
  48. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +99 -0
  49. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +25 -0
  50. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +251 -0
  51. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh +196 -0
  52. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +222 -0
  53. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +16 -10
  54. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity.sh +178 -0
  55. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_frozen_case_sanctity_selfproof.sh +108 -0
  56. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_gate_receipt.sh +431 -0
  57. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh +151 -0
  58. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +86 -5
  59. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +27 -21
  60. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +25 -15
  61. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/classical-test-design-techniques.md +1 -1
  62. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc-review-and-prioritization.md +1 -1
  63. package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/update-lifecycle.md +2 -0
  64. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +9 -9
  65. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +5 -1
  66. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/e2e-real-flow-testing.md +2 -2
  67. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/integration-contract-testing.md +10 -0
  68. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-code-authoring-patterns.md +2 -2
  69. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-topology-and-commands.md +1 -1
  70. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +2 -1
  71. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/annotation-driven-revision.md +9 -0
  72. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/figure-and-table-craft.md +8 -2
  73. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -0
  74. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/react-architecture.md +3 -0
  75. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-quality-release.md +37 -4
  76. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-ui-quality.md +10 -1
  77. package/dist/assets/release.json +127 -67
  78. package/package.json +1 -1
@@ -41,10 +41,11 @@
41
41
 
42
42
  ## Tier-2:路由 task-bank + 廉价 grader(已落地,advisory)
43
43
 
44
- `scripts/eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--desc-budget-chars N]`。`--desc-budget-chars N` 把每条 description 截到前 N 字符再评(消费端截断臂——如 Codex 在 ~2% 上下文预算/未知窗口 8,000 字符下压缩技能清单);plain 与 budgeted 各跑一遍即可定位"只活在描述尾部"的路由触发词。`make eval-routing-bank`。
44
+ `scripts/eval-routing-bank.rb <repo-root> [--bank p] [--model m] [--limit N] [--dry-run] [--json p] [--baseline p] [--desc-budget-chars N] [--replicas N]`。`--desc-budget-chars N` 把每条 description 截到前 N 字符再评(消费端截断臂——如 Codex 在 ~2% 上下文预算/未知窗口 8,000 字符下压缩技能清单);plain 与 budgeted 各跑一遍即可定位"只活在描述尾部"的路由触发词。`--replicas N` 每 task 评 N 次:task 判定取保守共识(任一有效副本 FAIL 即 FAIL),并报告副本 top1 一致率——不同 (bank, replicas) 配置是不同尺子,不得互相 diff 当回归。`make eval-routing-bank`。
45
45
 
46
- - 冻结 task-bank `eval/routing-tasks.jsonl`:每行 `{id, utterance, expected_skill, must_not_route_to?, source, why_expected, frozen_at_sha}`,种子取自 source-register 历史 miss + bootstrap 路由规则。
47
- - grader = 每 task 一次本机 `claude --print --tools "" --model <haiku>`,喂 utterance + 全部 skill description(agent 真正路由的那份面),要 `{selected_skill, confidence, rationale_short}`。
46
+ - 冻结 task-bank `eval/routing-tasks.jsonl`:每行 `{id, utterance, expected_skill, acceptable?, must_not_route_to?, source, why_expected, frozen_at_sha}`,种子取自 source-register 历史 miss + bootstrap 路由规则。**已修复的路由 miss 必须把其 utterance 冻结成 bank task 落在同一交付里**(修复不冻结=下次同类漂移无回归面)。`expected_skill: "none"` 是否定对照/覆盖空洞哨兵:正确结果是没有技能认领;`acceptable` 列出可辩护替代结果(如空洞探针上 coordinator 接管与拒绝都对);`must_not_route_to` 点名吸入诱饵邻居。结构由 `test_routing_bank_integrity.sh` 确定性把关(sentinel 只准出现在 expected/acceptable,不准进 must_not)。
47
+ - **冻结案例神圣(regressions-are-sacred)**:已冻结案例的删除或判定面改写(bank task 的 expected/acceptable/must_not,golden trace 的 assert 块)是一次回归裁决事项,与邻居回归同权——平均改善不得抵消单条冻结案例的失守,且「曾 yes 现非 yes」**含降级为 unsure/INCONCLUSIVE** 都算回归。删除/改判的每条案例必须在同一轮的 register 追加行里写 `case-retired: <id>` 或 `case-rescoped: <id>` 并给理由,交独立评审裁决;确定性半边由 `test_frozen_case_sanctity.sh` 按 `CCL_SKILL_BASE_REF` 把关——无裁决行即红,无 base ref 时打印显式 skip token(skipped ≠ passed),它只保证交易可见,不裁决交易正当性。
48
+ - grader = 每 task(×replicas)一次本机 `claude --print --tools "" --model <haiku>`,喂 utterance + 全部 skill description(agent 真正路由的那份面),要 `{selected_skill|none, clarify, confidence, rationale_short}`。**clarify 率、低置信率(<0.5)、副本一致率是一等报告字段**,不是旁注——路由质量的残余风险常在"高置信直选却选错、无自纠路径"这类 pass/fail 看不见的分布里。
48
49
  - **路由兼容性信号,不是真值预言机** —— grader 自己可能错;只衡量"当前 description 能否让廉价模型把固定 utterance 路由到 expected"。
49
50
  - **advisory**:不接 `check-ccl-skills.sh`,不挡 merge。退出码:`0` = 跑完;`2` = 用法;`3` = grader 整体不可用(claude CLI 缺失则打印 skipped 后 `0`)。路由 miss 永不非 0。
50
51
  - **防作弊**:runner 校验每 task 的 `frozen_at_sha` 是 HEAD 祖先(非祖先 = drift,排除出回归判定);同一改动若同时动 task-bank 和 SKILL.md description 会显式告警(防"改 skill 顺手改测试让它过")。
@@ -58,6 +59,18 @@
58
59
  2. 改后通过数必须在**最终措辞**上重测:中间稿的通过数在措辞再变的那一刻作废,不得挪用到最终候选的证据里。
59
60
  3. 受影响邻居用例集默认改前/改后各 **≥3 轮**,集合须含期望 owner 自己的兄弟用例与高词面重叠的他 owner 用例;邻居回归作为独立 finding 交由本轮实际门禁处置——**该 finding 须以 blocking 记入本轮 dual-track 评审记录,且只能由独立评审方豁免,不能由实现者自行判定「本轮没有门禁采用这组证据」而放行**。降级的是「F4 自己充当合并门禁」这一声称,不是「回归必须被人裁决」这一义务;后者若也随之消失,这一条就只剩被裁决方自审。
60
61
  4. 每轮判决必须连同 **runner 调用、grader 模型身份、候选身份**(commit 或描述内容指纹)与**原始逐轮工件的持久定位符**一并记入轮记录;没有定位符的通过数只能标注为 operator-reported,不得据以宣称修复轮已 concluded。
62
+ 5. **单变量归因**:一次改前/改后对照只准动**一个路由变量**(一条 description,或同一 skill 不可分割的一组路由面)。同时动多条 description 的批量改动,其对照差值不可归因到任何一条,只能按整包回归读——要归因就拆成逐条 A/B。(源侧实测形态:仅替换一条 description 的成对子集对照,把命中从约 2/3 提到 95%,且提升可归因到那一条改动——多条同动时这句话说不出口。)
63
+
64
+ ## 路由失效形态词表(跨层;报告与修复讨论用这套名字)
65
+
66
+ pass/fail 之外,路由失败有可命名的形态;每个形态有不同的检测器与不同的修法,混称"路由不准"会修错面:
67
+
68
+ | 形态 | 判据 | 检测器 | 典型修法 |
69
+ |---|---|---|---|
70
+ | **吸入 (absorbed)** | 应被拒绝(expected none)或应远离诱饵邻居(must_not)的 utterance 被某技能认领——覆盖空洞不被承认而被最近邻低置信/clarify 拉走 | bank 否定对照+空洞探针,runner `absorbed` 标签 | 认领空洞(补 owner)或在诱饵邻居 description 加排除句;不要靠 grader 自觉 |
71
+ | **归属分裂 (ownership_split)** | 同一 utterance 多副本给出不同 top1——所有权不稳定,谁都像 owner | `--replicas ≥2` 一致率 + `ownership_split` 标签 | Skip-when 互相消歧或触发词改具体(description-authoring 的 80% 阈值) |
72
+ | **静默跳过 (silent skip)** | 路由"成功"但被选技能的正文从未被读/其硬规则从未被应用——选择层过了,质量层空转。判据阶梯:mounted → invoked → 文件真被读 → 下游行为改变;mounted-only 不证明任何生效 | 非 Tier-1/2 可见;B 面 body-compliance 探针 + Tier-3 真 agent 事件流 | 修正文 firing point/read routing(body 指针补齐四元组,见 description-authoring.md「Body routing pointers」),不是修 description |
73
+ | **高置信错选** | 错选但 confidence 高、无 clarify——事后无自纠路径,比低置信错选更危险 | 报告里 FAIL ∩ 高置信 ∩ clarify=false 的行 | 触发词消歧;必要时在正文加入口自检;残余风险如实记录 |
61
74
 
62
75
  ## Tier-3:hub golden trace 真 agent 回放(已落地,advisory,人工判定)
63
76
 
@@ -71,6 +84,14 @@
71
84
  - **随机性**:agent 非确定;判定先人工、nightly 起步,有稳定史前不自动 gate。防作弊同 T2(`frozen_at_sha` 祖先校验)。
72
85
  - **双用途**:除回归外,Tier-3 还可当**改技能前的 RED-baseline**(改前手动跑触发场景看真 agent 是否真路由错,改后看 compliance)——可选;只有真观察到 miss 才算 RED(PASS/INCONCLUSIVE 不算),小 N + 非确定有噪声,手动跑两次自己留两份报告。落地 + 防作弊注意见 [validation-and-landing.md](validation-and-landing.md) "Optional real-agent RED-baseline"。
73
86
 
87
+ ## B 面:正文合规探针 body-compliance(已落地,advisory)
88
+
89
+ `eval/body-compliance-eval.rb <repo-root> [--arm L] [--json p] [--model m] [--timeout s] [--ids a,b]`。`make eval-body-compliance`。路由三层测「选没选对技能」;B 面测**已激活技能的正文硬规则是否真被应用**——正文即 prompt,逐探针 required/forbidden marker 契约判分,覆盖是 NAMED SUBSET(见 runner 头部声明)。
90
+
91
+ - 含 product-rd-workflow 停机谓词的**成对分类探针**(prd-stop-*/prd-continue-*):每对场景恒定、只变谓词判别特征,按闸自身的字面 `continuing:`/`blocked:` marker 判分——确定性锚钉的是这些谓词的**措辞存在性**,只有这些探针检验**案例被分到哪边**。
92
+ - 触发纪律:改动触及某技能正文硬规则或停机谓词时,must run the affected `--ids` probe subset on this machine before landing,结果按该改动既有门禁处置;advisory 契约与升级路径(提炼确定性不变量,永不阻断 LLM 判定)见 `eval/AGENTS.md` 与 [f4 手册](../../../docs/f4-skill-effectiveness-harness.md)「双轨承载面与运行分层」。
93
+ - 实测的双轨互补边界(锚管措辞、探针管行为漂移、不是埋句绊线)与小样本告诫,单一落点在 f4 手册「双轨承载面与运行分层」,此处不复述。
94
+
74
95
  ## Health roll-up:描述性仪表盘(已落地,advisory)
75
96
 
76
97
  把上面各信号卷成**一个加权 0–10 显示值 + 同尺子变化**,用于定位值得继续检查的维度。它借用 OpenSSF Scorecard 的呈现形态,但不把不同性质的 F4 信号变成“仓库整体变好/变差”的总判决。映射与限制见 [harness-patterns-and-eval.md](harness-patterns-and-eval.md) §3.4。
@@ -20,7 +20,7 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
20
20
  ├─ d. Sanitization pass with checklist (cheap, seconds)
21
21
  ├─ e. Owner review gate per mandatory table (deep, minutes)
22
22
  │ ├─ Strict wording-only → one independent code-review pass
23
- │ └─ Non-wording → extraction_review_gate review + at most two challenges
23
+ │ └─ Non-wording → extraction_review_gate review + challenge (wrapper-fixed budget)
24
24
  ├─ f. Apply fixes, re-sanitize
25
25
  ├─ g. Commit per batch on a feature branch → MR pending review (never push to main)
26
26
  └─ h. Update charter completion log
@@ -91,10 +91,10 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
91
91
 
92
92
  - When required: see `references/dual-track-review-gate.md` table.
93
93
  - Choose the review tier from that table, not from intuition. Do not restate the rows locally; record the exact `dual-track-review-gate.md` table row used. Record `challenge: not-required` only when that row classifies the actual diff as challenge-not-required (for shared skills, this means strict wording-only with deterministic scope proof + independent review confirmation). Non-wording shared-skill changes cannot skip challenge.
94
- - Run deterministic checks and implementer self-review first, and record what each proves before invoking review/challenge (this self-review-before-review ordering applies to every non-wording shared-skill change the dual-track table requires review for, not only the rows that look high-risk): `git diff --check` proves whitespace/conflict-marker hygiene only; validators prove schema/link/routing invariants; leakage/sanitization scans prove only their configured patterns; scope checks must name the changed files or expected file set; the self-review row is conclusive only when each required field is non-empty (acceptance criteria, changed-file scope, edge/failure paths, known residual risks) and the changed-file scope equals the candidate diff's changed-file set, or explicitly explains any excluded generated/irrelevant file. Persist it before the review/challenge run in a fresh, non-overwritten task-evidence path outside the candidate diff, pass that exact file as the gate's review plan, and retain the gate result that binds its profile hash; do not edit the candidate merely to record self-review or review outcome, because that creates self-referential candidate churn. A candidate-local row is appropriate only when the row itself is a substantive deliverable under review. A plain in-place-editable MR description or scratch log is not ordering proof unless its edit history is retrievable and checked; a backfilled row is invalid and forces a rerun. If the candidate diff changes after the row is saved — a file added/removed OR the content of any listed file materially changed — refresh the row and rerun review/challenge against the new candidate. Changing only the external self-review record refreshes the profile binding; it does not by itself invalidate implementation tests or the candidate packet. A missing field, "ok" placeholder, mismatched scope, or unprovable ordering makes the row inconclusive. Do not spend LLM review rounds on issues a script or implementer-side checklist can decide. If the independent pass is the first place basic scope, contract, privacy, or test issues surface, apply those findings to the diff, close the self-review gap, and rerun the deterministic gates before rerunning review/challenge; the process-defect repair is in addition to resolving the findings, not a way to discard or downgrade them.
94
+ - Run deterministic checks and implementer self-review first, and record what each proves before invoking review/challenge (this self-review-before-review ordering applies to every non-wording shared-skill change the dual-track table requires review for, not only the rows that look high-risk): `git diff --check` proves whitespace/conflict-marker hygiene only; validators prove schema/link/routing invariants; leakage/sanitization scans prove only their configured patterns; scope checks must name the changed files or expected file set; the self-review row is conclusive only when each required field is non-empty (acceptance criteria, changed-file scope, edge/failure paths, known residual risks) and the changed-file scope equals the candidate diff's changed-file set, or explicitly explains any excluded generated/irrelevant file. Persist it before the review/challenge run in a fresh, non-overwritten task-evidence path outside the candidate diff, pass that exact file as the gate's review plan, and retain the gate result that binds its profile hash; do not edit the candidate merely to record self-review or review outcome, because that creates self-referential candidate churn. A candidate-local row is appropriate only when the row itself is a substantive deliverable under review. A plain in-place-editable MR description or scratch log is not ordering proof unless its edit history is retrievable and checked; a backfilled row is invalid and forces a rerun. If the candidate diff changes after the row is saved — a file added/removed OR the content of any listed file materially changed — refresh the row; any rerun of review/challenge against the new candidate draws on the remaining cross-chain Agent budget (the self-hosted-chain rule in `references/dual-track-review-gate.md`), and at the cap, or at the effective exhaustion that rule defines, the terminal-disposition path governs instead of a rerun. Changing only the external self-review record refreshes the profile binding; it does not by itself invalidate implementation tests or the candidate packet. A missing field, "ok" placeholder, mismatched scope, or unprovable ordering makes the row inconclusive. Do not spend LLM review rounds on issues a script or implementer-side checklist can decide. If the independent pass is the first place basic scope, contract, privacy, or test issues surface, apply those findings to the diff, close the self-review gap, and rerun the deterministic gates before rerunning review/challenge; the process-defect repair is in addition to resolving the findings, not a way to discard or downgrade them.
95
95
  - Review pass: persist the complete self-review row and encode it in the review plan. For a **non-wording** lane, resolve the repository-owned `scripts/extraction_review_gate.sh` and use it from round 1; never substitute the generic controller, scan writable plugin roots, or supply a caller-selected budget. For a strictly proven **wording-only** lane, use the generic `code-review` proof-bound single-review recipe in `code-review/references/staged-review-contract.md` and record `challenge: not-required`; require its controller-derived wording scope plus the independent `wording_only_boundary` confirmation. This is the only extraction path that stays outside the multi-round wrapper and terminal ledger; the gate, not this page, decides whether a chainless review is legal, and it may still demand the tracked pair. Take all controller options from that runnable recipe, supplying the actual stage and exact candidate rather than an example default. The non-wording chain cannot be retrofitted, so a run started outside its owner wrapper is thrown away and restarted. Read the chain-opening and packet-composition rules in `references/dual-track-review-gate.md` first. Require conclusive JSON, selected-client attribution, packet/profile binding, family exclusion, and wrapper runtime evidence. When the host returns a live execution handle (`session_id`, `cell_id`, or equivalent), keep polling that exact handle until terminal exit; empty current output is progress, not a verdict, and no replacement/fallback reviewer may start while the original process is live. The result row records handle type, an opaque host transcript/tool-call reference and terminal exit status. If the handle is lost, the lane is infrastructure-inconclusive/manual-review-required and no replacement or fallback may be started or credited; process-tree and wrapper artifacts are diagnostic only. This is a procedural host obligation because the inner gate cannot observe the outer handle. Never copy a credential-like raw handle into shared evidence. `findings` is not pass; inconclusive, malformed, or free-form output stays interim. Do not add a separate behavior probe.
96
96
  - Challenge pass: for a non-wording lane, invoke `scripts/extraction_review_gate.sh` separately with the same plan, stage, candidate, family and tracked chain. Pass the next one-based index; later rounds include a distinct focus and all prior focuses. Preserve a separate result row with the same binding, exclusion, egress, attribution and conclusive checks. Review never satisfies challenge; missing or inconclusive required challenge keeps extraction interim. A wording-only lane has no challenge pass.
97
- - Treat review/challenge as batch-level gates over the landing candidate, not as a per-bullet or per-line edit loop. Apply all findings from a round; when both lenses are required, re-run both on the updated candidate before landing.
97
+ - Treat review/challenge as batch-level gates over the landing candidate, not as a per-bullet or per-line edit loop. Apply all findings from a round; when both lenses are required and cross-chain Agent budget remains, re-run both on the updated candidate before landing — every re-run sums into the same wrapper-fixed budget, and at the cap, or at the effective exhaustion that rule defines, the terminal-disposition path in `references/dual-track-review-gate.md` replaces further re-runs.
98
98
  - Skipping a required challenge = work can only land as interim, not complete.
99
99
 
100
100
  #### 3f. Apply fixes, re-sanitize
@@ -119,7 +119,7 @@ For maintainers running a fresh codebase / Figma / doc extraction. Read this fir
119
119
  - File: `~/.<host>/skills/.extraction-work/<project>-completion.md`
120
120
  - Final state: which batches done, which deferred, which sources unavailable.
121
121
  - Lessons: what surprised; what would change in next extraction; what to add to skill-extraction-workflow.
122
- - For every non-wording review chain, build the receipt-bound closeout ledger and run `scripts/validate_extraction_review_state.py <closeout.json>` before reporting a terminal state. A clean Round 2 plus its exact-candidate completion receipt may validate as `ready_for_human_decision`; Round 3 findings validate as `continuation_authorization_required`; a second ordered base drift validates as `baseline_race`. Unknown, stale, omitted, or invalid evidence remains `interim`. The strict wording-only single-review path records its independent review row but does not fabricate a multi-round ledger.
122
+ - For every non-wording review chain, build the receipt-bound closeout ledger and run `scripts/validate_extraction_review_state.py <closeout.json>` before reporting a terminal state. A clean Round 2 challenge plus its exact-candidate completion receipt may validate as `ready_for_human_decision`; Round 2 findings at the exhausted budget validate as `continuation_authorization_required`; a second ordered base drift validates as `baseline_race`. Unknown, stale, omitted, or invalid evidence remains `interim`. The strict wording-only single-review path records its independent review row but does not fabricate a multi-round ledger.
123
123
 
124
124
  ### 5. Provenance migration
125
125
 
@@ -149,7 +149,7 @@ Skip this step when nothing transferable surfaced.
149
149
  | Anti-pattern grep panel | `references/recurring-anti-patterns-checklist.md` | Every commit; ~30s |
150
150
  | `check-ccl-skills.sh` | `scripts/check-ccl-skills.sh` | Every commit; ~10s |
151
151
  | Generic `code-review` gate | repository-owned skill | Strict wording-only independent review; ~5-10 min |
152
- | `scripts/extraction_review_gate.sh` | this skill package | Non-wording review plus at most two challenges; ~5-15 min each |
152
+ | `scripts/extraction_review_gate.sh` | this skill package | Non-wording review plus the wrapper-fixed challenge budget; ~5-15 min each |
153
153
  | `scripts/validate_extraction_review_state.py <closeout.json>` | this skill package | Every non-wording terminal checkpoint |
154
154
  | Source-read fallback ladder | `SKILL.md` Source-read remediation | When a source read fails or times out |
155
155
  | Sibling mini-map | `SKILL.md` Step 4 stack-specific updates | Every stack-specific change |
@@ -24,7 +24,7 @@ The at-add-time check above decides *where* a rule lands (merge vs new bullet);
24
24
 
25
25
  | Baseline failure | Right form | Wrong form |
26
26
  |---|---|---|
27
- | Agent knows the gate and walks past it under pressure (discipline slip — our merge-authorization / R0 / done-claim class) | Prohibition + rationalization-vs-reality pairs + red-flag self-check | Soft guidance ("prefer…", "consider…") |
27
+ | Agent knows the gate and walks past it under pressure (discipline slip — our merge-authorization / R0 / done-claim class) | Prohibition + rationalization-vs-reality pairs + red-flag self-check — pairs quote excuses actually captured verbatim from baseline/pressure runs (`validation-and-landing.md` eval-first step 2), never invented ones | Soft guidance ("prefer…", "consider…") |
28
28
  | Agent complies but the artifact's SHAPE is wrong (bloated review packet, buried verdict, register row restating the source) | Positive recipe/contract: state what the artifact IS — its parts, in order | A "don't"-list about the shape — the measured backfire above |
29
29
  | Agent omits a required element from an artifact it already produces (missing status/evidence cell, absent map row) | A REQUIRED slot in the template/validator it must fill (our closeout rows and register gate are this form) | Prose reminders near the template |
30
30
  | Behavior should differ by situation | Conditional keyed to an observable predicate ("fan-out → name the tier") | Unconditional rule + exemption clauses |
@@ -412,3 +412,84 @@ and inverted the sense (production, not product), and the coordinator now shares
412
412
  | Supersedes the identity-grammar and tag-layer clauses of the shared-scan supersede row: the metadata scan recognizes current cross-provider model words, session-task URLs, and only canonically-shaped tag objects | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh; firing-path: command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh; result-class: failure | updated | `skill-extraction-workflow/SKILL.md` is the owner key. The same external round surfaced: bare `GPT-*` identities and `Gemini * Flash/Ultra` model forms escaped every surface (now in the registry and unambiguous grammar with trailer regressions); Codex task URLs on session origins (`.../codex/tasks/<id>`) matched no session-path grammar (now matched, with a product-page near-miss control); a literal tag object could duplicate its `object` header so the scanner followed a decoy chain while Git peels the first target (canonical header shape now required, duplicates fail closed); the closeout validator's completion-receipt `schema_version` guard lacked the exact-integer type check applied everywhere else (a float `3.0` validated; now rejected with a regression), and `bounded_text` accepted C1 controls and U+2028/U+2029 line separators into diagnostics (now rejected with per-codepoint regressions). The second pass additionally rejected default-ignorable format characters that could split one recurrence class into visually identical keys, and bound every counted chain round to the exact ledger candidate rather than only the final receipt. |
413
413
  | Supersedes the fixture-portability posture of the plan-intent suite row: a mode assertion probes GNU stat before BSD stat | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_update_review_plan_intent.sh; result-class: failure | updated | `code-review/SKILL.md` is the owner key. GNU stat echoes unknown BSD directives (`%Lp`) verbatim with exit 0, so the suite's BSD-first probe never reached its GNU fallback and CI shard 2 failed `plan mode was not preserved` on every Linux run while macOS stayed green. The assertion now probes `stat -c` first and falls back to `stat -f '%Lp'`, mirroring the pattern `opencode_review.sh` already runs on both platforms. |
414
414
  | Supersedes the comparison-domain clause of the obligation-preservation audit: a repository-frozen ledger pins BOTH ends of its domain, so unrelated later changes owe it nothing | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh; firing-path: command:skills/skill-extraction-workflow/scripts/test_obligation_ledger_repo_audit.sh; result-class: failure | updated | `skill-extraction-workflow/SKILL.md` is the owner key. The 065 obligation audit derived its row set from pinned-base..WORKING-TREE, so the first post-landing PR that rewrote any obligation line in any `skills/**/*.md` went red on preservation rows it never owed (observed: 12 phantom rows for this candidate's own code-review contract-paragraph rewrite). `obligation-ledger.py` now accepts `--head`, the ledger header pins `Head revision` beside the base, and the repo audit reads and requires it; carrier-drift detection still reads current files, so rewriting a BOUND carrier stays red (probed: the pinned audit still fails `CARRIER_COMPOSITE_NOT_UNIQUE` on a carrier rewrite). The synthetic suite adds the differential: a post-head non-carrier rewrite passes the pinned audit and fails the unpinned one with `ROW_SET_MISMATCH`. |
415
+ | Browser E2E business-success assertions must anchor on objective effects — weak proxy signals never suffice as the sole pass condition; billable metered-resource tests take an explicit lane with a named budget owner; the Playwright component-testing maturity claim is corrected against its primary source | `testing-strategy` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/testing-strategy/references/e2e-real-flow-testing.md#weak proxy signals are not business assertions | updated | Owner key `testing-strategy/SKILL.md` (unchanged this round; the changes are merges into two references, no new bullets). (1) Weak-proxy blacklist (page loaded / URL changed / non-empty text / generic element visible / success toast alone) merged into the existing click-through clause in `e2e-real-flow-testing.md` §E2E Scope Control, pairing the already-present objective-effect positive list in §Backend/API Real Flows with its negative list; (2) billable-resource lane semantics (paid model inference, per-call third-party APIs, real payment flows, cloud sandboxes/device farms → explicit marker/lane, out of default PR and scheduled-frequent lanes, named budget owner plus authorized test account) merged into the expensive-test clause in `test-topology-and-commands.md` §Command Tiers, with credential/provisioning discipline routed to `ci-fixtures-and-flake-control.md`; (3) factual correction in `e2e-real-flow-testing.md` §Browser E2E Rules item (d), scoped to exactly what the excerpt establishes: Playwright's current component-testing guide supersedes the experimental add-on packages — primary source `https://playwright.dev/docs/test-components` (fetched 2026-08-30): "This guide replaces the experimental `@playwright/experimental-ct-react` and `@playwright/experimental-ct-vue` packages"; the sibling Biome correction rests on `https://biomejs.dev/linter/` ("a total of 526 rules", "many of them inspired from other linters" — the latter covers the skill text's rule-origin parenthetical) plus `https://biomejs.dev/blog/biome-v1-9/` ("CSS formatter and linter are now considered stable"; the prior GraphQL-timeline clause was removed from the edited skill text rather than carried beyond its excerpt), and the S3 note on `https://docs.aws.amazon.com/AmazonS3/latest/userguide/checking-object-integrity.html` plus its upload subpage `.../checking-object-integrity-upload.html` (both fetched 2026-08-30: `CRC64NVME` "is the default checksum algorithm"; "A composite checksum is calculated based on the individual checksums of each part in a multipart upload" — establishing the composite branch the skill text describes; the per-algorithm support matrix lives on that subpage and is not restated here). RED-baseline (applied, differential; evidence scope: gate wiring and package integrity only, not per-clause semantics): on the committed candidate, `CCL_SKILL_BASE_REF=origin/dev check-ccl-skills.sh` ran green (`ccl_skill_check_clean_ok`, control); a probe commit deleting exactly this row turned the same run RED printing `impact_chain_gate_missing: upstream-owner skill changed without a matching source-register impact-chain row / missing evidence path: testing-strategy/SKILL.md` (attributable to the owning gate); restoring the row returned it to green — the oracle command is reproducible in-repo on any checkout of this candidate. Corpus-identifier scan over the five changed files (grep for the private source-vocabulary set) returned zero hits, and the repository R0 audit printed the clean private token in the same run. Clause semantics rest on this round's independent review and adversarial challenge rows. Scope of what THIS ROW asserts is deliberately narrow: the three inlined public excerpts above, the reproducible in-repo gate/scan runs, and nothing further — verification of other claims in the same round's diff is recorded in the maintainer's private round archive per the extraction-lifecycle handoff policy and is NOT certified by this row; a reviewer should judge those clauses against their own named public sources directly. Dual-track terminal record: multiple independent codex review rounds plus three full review+challenge×2 chains ran against successive candidates; every P0/P1 with a demonstrable failure path was applied (streaming retry/latch/resume/cancel semantics, canonical/hreflang interplay, submit latch, billable/payment-sandbox lanes, token-scan hardening, excerpt-scoping narrowings); the terminal chain's last three finding-fixes (cancel-vs-buffered-end, idle/heartbeat timeout, latch release on terminal state) were merged AFTER that chain per the repository's fixed-budget review-loop rule, so the exact landing candidate carries them un-rechallenged — listed for the merge decision-maker; the recurring bounded-packet self-certification findings (a packet reviewer cannot execute the in-repo oracle or see the private vocabulary) are dispositioned as verify-by-rerun: running the repository check script (`check-ccl-skills.sh` with `CCL_SKILL_BASE_REF=origin/dev`) on this candidate is the reviewer-executable oracle. Source class: local external engineering-skill corpus (sanitized to capability labels) plus public primary docs; sibling map: `web-react-dev` updated in the same round (its own package, not machine-gated), `product-ui-ux-design` unchanged (aesthetic-direction and anti-generic coverage already equal or stronger than the external source), `design-closed-contract-oracles.md` unchanged (criterion-to-proof already covered), `ci-fixtures-and-flake-control.md` unchanged (credential/provisioning face already covered). |
416
+ | Evidence-pipeline failure is a per-case third verdict: infra-error is never pass, never business-fail, never silent skip, and weaker surfaces cannot substitute for missing evidence | `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/testing-strategy/references/ci-fixtures-and-flake-control.md#never a pass, never a business fail, and never a silent skip | `updated` | Owner key `testing-strategy/SKILL.md`. Benchmark round (private provenance alias: mediagen-platform) plus primary-source verification. RED baseline (replayed): `git show origin/dev:skills/testing-strategy/references/test-code-authoring-patterns.md` asserts the xUnit smell corpus as ~18 items while xunitpatterns.com lists 15 top-level (5/6/4), and the coverage floors carried no provenance — head corrects both and adds the infra-error verdict rule; baseline grep for the anchor line is zero-hit, head exactly one. |
417
+ | A continuation gate distinguishes multi-viable-approach and no-evidence-cause stops from a single dominant reversible path, which is carried through to a reviewable draft instead of stopping at a recommendation | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#do not stop at a recommendation | `updated` | Owner key `product-rd-workflow/SKILL.md`. RED baseline (replayed): `git show origin/dev:skills/product-rd-workflow/references/delivery-lifecycle.md` attributes DORA metrics' origin to the 2018 book while dora.dev/insights/dora-metrics-history dates the research line to 2014 — head corrects it; baseline zero-hit for the new stop/carry-through predicate, head exactly one. Review checklist gains guarantee grading and disabled-path walk (same package). |
418
+ | Query/lookup evidence reports 0, 1, or N matches distinctly; silently taking the first row of N is forbidden and an empty result over a named scope is itself evidence | `defect-diagnosis` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/defect-diagnosis/SKILL.md#never silently take the first row of N | `updated` | Owner key `defect-diagnosis/SKILL.md`. Source: external production skill-pack doctrine (alias mediagen-platform), mechanism verified as stack-agnostic. RED baseline (replayed): baseline grep for the cardinality predicate is zero-hit across the package, head exactly one — the baseline tree gave no instruction against first-row-of-N, so an agent following it could silently mis-resolve identity lookups. |
419
+ | Risk classification runs on the change's objective shape: reporter tone or executive pressure never escalates tags and a small diff never de-escalates them | `feature-risk-router` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/feature-risk-router/SKILL.md#must never escalate a change's tags | `updated` | Owner key `feature-risk-router/SKILL.md`. Baseline covered only the small-diff de-escalation side (low-risk intuition clause); the tone/pressure escalation side was absent — RED baseline (replayed): baseline grep zero-hit for the anti-escalation predicate, head exactly one. Six objective evaluation axes recorded inline. |
420
+ | Reader annotations are evidence of reading breakdown: classify the root-cause class first, then sweep the whole document for the same class instead of patching only the flagged sentence | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/annotation-driven-revision.md#必须全文扫同类位置一起修,不得只改被标记的那一句 | `updated` | Owner key `tighten-doc/SKILL.md`. RED baseline (replayed): `git show origin/dev:skills/tighten-doc/references/figure-and-table-craft.md` lists the 25-word sentence limit as no-reliable-source while GOV.UK's writing guideline states it as its house style — head regrades it to a sourced single-institution style and records the 40-char claim's traceable community source; baseline zero-hit for the annotation-sweep predicate, head exactly one. |
421
+ | Review reuse depth is a deterministic function of the candidate delta with no manual downgrade, the reviewed-identity record is a freshness guard not cryptographic proof, and relayed blocking findings pass a false-positive check with auditable dispositions | `code-review` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/code-review/references/manual-invocation-and-prompts.md#deterministic function of the delta, never a manual downgrade | `updated` | Owner key `code-review/SKILL.md`. Benchmark adjudication recorded: the source pack's receipt survives rebase/amend; our stricter voids-on-any-edit line is deliberately KEPT and only the tiering-determinism and honest-trust-model clauses are absorbed (keep-stricter per Conflict Resolution). RED baseline (replayed): baseline grep zero-hit for the determinism predicate, head exactly one. |
422
+ | Observation code never intrudes on the observed path, cross-layer conclusions require aligned identifiers or time, existing signals are discovered before new ones are created, and explicit environment targets always win | `platform-observability` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-observability/references/metrics-conventions.md#must not add retries, blocking waits, or business-logic branches | `updated` | Owner key `platform-observability/SKILL.md`. RED baseline (replayed): `git show origin/dev:skills/platform-observability/SKILL.md` attributes the good/valid SLI formula to Google SRE generically while the SRE Workbook's own text is good/total (the valid-events refinement is the Art of SLOs / GCP-blog line) — head corrects the attribution and self-labels the SLO ladder as team heuristic; baseline zero-hit for the non-intrusion predicate, head exactly one. |
423
+ | Benchmark rounds issue per-mechanism P/I/M/W verdicts with grep-anchored evidence, and an M verdict passes the functional-equivalent check before any borrow lands | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/source-to-skill-extraction.md#requires the zero-hit grep recorded as replayable evidence | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Method absorbed from an external theory-audit corpus (alias mediagen-platform) whose own run found all 8 W-verdicts were internal drift rather than external knowledge gaps — the W-type internal-consistency sweep, disposition middle states (待实验/条件化采纳), and three conflict-synthesis shapes land in source-to-skill-extraction.md; keyword-activation evidence lands in description-authoring.md. RED baseline (replayed): baseline grep zero-hit for the functional-equivalent predicate, head exactly one. |
424
+ | A diff-scoped design review classifies every hard-coded visual-value hit into approved usage, pre-existing outside the change, or new violation — only new violations block, and pre-existing debt is routed, never blamed on the change | `product-ui-ux-design` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/product-ui-ux-design/references/ui-ux-audit.md#never reported as caused by this change | `updated` | Owner key `product-ui-ux-design/SKILL.md`. Changed refs: ui-ux-audit.md (diff-scoped three-bucket review), design-system-source-of-truth.md (definition-matrix completeness check), tokens-and-components.md (old code is not permission), platform-mobile-patterns.md (motion band self-labeled team heuristic per M3 tokens/Apple HIG verification; M3 Expressive research figures cited). RED baseline (replayed): baseline grep zero-hit for the three-bucket predicate, head exactly one. |
425
+ | Repo-pinned TC implementations pin the catalog revision, claim before implementing, and stop to fix a stale/contradictory source record before implementing against it | `test-artifact-management` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/test-artifact-management/references/update-lifecycle.md#停下先修源记录**,不得按旧版实现后事后补 | `updated` | Owner key `test-artifact-management/SKILL.md`. RED baseline (replayed): `git show origin/dev:skills/test-artifact-management/references/classical-test-design-techniques.md` states 2-way捕到 50–90% while NIST SP 800-142 Table 1 gives 53–97% with an explicit 10–40%+ miss warning, and tc-review-and-prioritization.md claimed a 30-year P×I consensus while ISTQB CTFL v4.0.1 §5.2 defines multiplication as the quantitative approach beside a qualitative matrix — head corrects both; baseline zero-hit for the revision-pin predicate, head exactly one. |
426
+ | Evaluation records keep human-review and machine fields with distinct writers, normalize cross-provider token accounting before comparison, and reconcile usage against billing where available | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/references/model-prompt-evaluation.md#must never overwrite a human verdict | `updated` | Owner key `llm-inference-integration/SKILL.md`. RED baseline (replayed): `git show origin/dev:skills/llm-inference-integration/references/model-prompt-evaluation.md` states extended thinking is off by default per docs while the current platform docs deprecate the manual mode on 4.6, reject it on 4.7+, and default thinking on for the Claude 5 family — head rewrites the claim per model generation; prompt-caching modes and the provider-evaluation evidence ladder (spend contract, seven layers, open-loop load per NSDI'06) added in the same package. |
427
+ | Durable-state transitions capture the clock once and validate external-response structure with a typed error path before mapping | `nodejs-service-dev` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/nodejs-service-dev/references/async-lifecycle-and-performance.md#never a crash or a silently-defaulted field | `updated` | Owner key `nodejs-service-dev/SKILL.md`. Sibling-parity landing with the go/python state-machine references (same two predicates, stack-idiomatic wording). RED baseline (replayed): baseline grep zero-hit for the predicate, head exactly one. |
428
+ | Vendor/ecosystem status claims carry their verified level and date — the RPC-alternative entry records CNCF sandbox status and the accurate interop-test wording instead of an inflated maturity tier | `go-microservice-architecture` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/go-microservice-architecture/references/architecture-playbook.md#CNCF **sandbox** project (per connectrpc.com | `updated` | Owner key `go-microservice-architecture/SKILL.md`. RED baseline (replayed): `git show origin/dev:skills/go-microservice-architecture/references/architecture-playbook.md` asserts CNCF-incubated and Google-validated interop while connectrpc.com states sandbox level and self-run extended interop tests — head corrects both; crypto-erase citation upgraded to SP 800-88 r2 + EDPB 02/2025 in the same package. |
429
+ | Crypto-erase evidence conditions cite the current sanitization standard revision and regulator guidance rather than a withdrawn revision | `python-service-architecture` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/python-service-architecture/references/multi-tenant-isolation.md#do not use CE when data predates encryption enablement | `updated` | Owner key `python-service-architecture/SKILL.md`. RED baseline (replayed): baseline cites NIST SP 800-88 generically (Rev.1 withdrawn 2025) with no CE-condition specifics; head cites Rev.2's explicit do-not-use conditions and EDPB 02/2025's encrypted-data-is-still-personal-data holding — mirrored with the go sibling. |
430
+ | Gitflow-style release trains use direction-sensitive merges (squash only feature→develop), prompt dual back-merge, human-confirmed tags, and full-ladder hotfixes; canary template thresholds are labeled as tool example values, and user-bucketed ramps follow exposure/funnel/salt data-literacy rules | `platform-release-engineering` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-release-engineering/references/promotion-gate-and-review.md#must not squash** — prefer fast-forward/plain merge | `updated` | Owner key `platform-release-engineering/SKILL.md`. RED baseline (replayed): `git show origin/dev:skills/platform-release-engineering/references/canary-and-rollout-strategy.md` presents the 1%/500ms thresholds as bare template defaults with no provenance, while Flagger's builtin checks carry exactly those example values and Argo Rollouts' official example uses 95% — head attributes and scopes them; the merge-topology section is grounded in AWS Prescriptive Guidance's Gitflow pattern (verified 2026-08). |
431
+ | Verdict assignment is decided by fault origin with evaluated-false as business fail, and required infra-error cases block aggregate readiness | `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/testing-strategy/references/ci-fixtures-and-flake-control.md#never by defaulting to whichever verdict looks better | `updated` | Owner key `testing-strategy/SKILL.md`. Fix-round rows for the review-chain findings: the pre-fix span (replayed via `git show` at the round base) carried the enumerated verdict mapping whose edges four review rounds broke in turn; the fault-origin predicate replaced the enumeration and the aggregate-readiness clause closed the clean-total false green. |
432
+ | Design-review exceptions are approved only by the design-system owner's recorded, unexpired, usage-covering approval; age is not approval; scan categories cover size/layout and motion | `product-ui-ux-design` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-ui-ux-design/references/ui-ux-audit.md#age is not approval: reusing or extending an old undocumented/expired exception | `updated` | Owner key `product-ui-ux-design/SKILL.md`. Fix-round: the pre-fix predates-arm allowed laundering an old undocumented/expired exception through a changed hunk (challenge-confirmed bypass, replayed at the round base); the arm was deleted (convergence-by-deletion) and the definition matrix aligned with the governed-category list. |
433
+ | TC claiming and write-back both ride atomic revision-conditioned updates; hash-pin mismatch always stops for source reconciliation | `test-artifact-management` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/test-artifact-management/references/update-lifecycle.md#停下改走人工/单线分配,不得按回读结果继续 | `updated` | Owner key `test-artifact-management/SKILL.md`. Fix-round: read-back-after-write was challenge-proven non-CAS (later writer silently replaces a verified claim, replayed at the round base); the landed rule requires platform CAS/optimistic-lock or a serialized coordinator, else concurrent claiming is unsupported and stops. |
434
+ | Relayed findings pass an existence-and-severity check, unverifiable stays blocking, and reuse voids on base/profile/lens change | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/code-review/references/manual-invocation-and-prompts.md#a confirmed-but-advisory or overstated-severity issue is relayed at its true severity | `updated` | Owner key `code-review/SKILL.md`. Fix-round: challenge rounds showed the fp-check validated existence only (severity inflation passed) and the unverifiable disposition could demote real defects; both closed, with the reuse boundary extended to base/profile/lens changes. |
435
+ | Auto-continue is subordinate to every stop condition, the metered-account carve-out keeps its existing/configured/self-use qualifiers, and stop wording preserves the pinned anchors | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#report interim or blocked with the next unblock step | `updated` | Owner key `product-rd-workflow/SKILL.md`. Fix-round: size-offset compression had silently widened the metered-account carve-out and detached carry-through from the stop list (review-confirmed, replayed at the round base); qualifiers restored, precedence bound, and checker-pinned anchor phrases restored after the pinned-phrase gate fired. |
436
+ | Vendor-standard citations are layered honestly: CE conditions cite Rev.1 §2.6 with Rev.2 superseding and continuing the framework | `go-microservice-architecture` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/go-microservice-architecture/references/multi-tenant-isolation.md#verify the corresponding Rev.2 section when citing it as the authority | `updated` | Owner key `go-microservice-architecture/SKILL.md`. Fix-round: the earlier candidate attributed CE do-not-use conditions to Rev.2 while the verification ledger located them in Rev.1 §2.6 (review-caught attribution mismatch); the landed text layers the citation and instructs Rev.2 section verification before citing it as authority. |
437
+ | The python sibling mirrors the layered CE citation byte-for-byte per the parity discipline | `python-service-architecture` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/python-service-architecture/references/multi-tenant-isolation.md#verify the corresponding Rev.2 section when citing it as the authority | `updated` | Owner key `python-service-architecture/SKILL.md`. Fix-round mirror of the go row above; parity gate keeps the mirrored region byte-identical. |
438
+ | Node's malformed-payload rule mirrors the go/python persisted-failure-transition semantics and single-now scopes to transition-stamped timestamps | `nodejs-service-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/nodejs-service-dev/references/async-lifecycle-and-performance.md#mirroring the go/python state-machine rendering | `updated` | Owner key `nodejs-service-dev/SKILL.md`. Fix-round: review caught the node wording drifting weaker than the go/python rendering (policy-handled vs persisted failure transition) and the single-now literalism re-stamping domain-provided times; both aligned. |
439
+ | Best-effort observation is scoped to diagnostic telemetry; audit/billing/deletion/release-gate records are business writes that never fail open | `platform-observability` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-observability/references/metrics-conventions.md#their loss is a failure, never shrugged off as telemetry | `updated` | Owner key `platform-observability/SKILL.md`. Fix-round: review showed the unbounded best-effort license could excuse a mandatory record failing open; the boundary now names the mandatory-record classes and their durable-delivery semantics. |
440
+ | Annotation-driven revision scans the whole document but edits only within authorization, and factual/citation errors route to source verification | `tighten-doc` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/tighten-doc/references/annotation-driven-revision.md#不得以「修同类」为名越权改动已定内容 | `updated` | Owner key `tighten-doc/SKILL.md`. Fix-round: review caught the unconditional same-class sweep expanding mutation past the approved scope and the root-cause classes omitting factual/citation errors; both landed with the scan-wide/edit-scoped split. |
441
+ | Keyword-activation guidance carries its sources and conditionality, and benchmark-figure reliance requires local reproduction | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/description-authoring.md#do not rely on the numbers without reproducing against your own catalog | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Fix-round: review flagged the activation claims as unsourced-in-ledger and unconditional; sources landed in the source-verification ledger and the conditionality bullet forbids relying on the numbers without local reproduction. This round also carries the crypto-erase supersede note above. |
442
+ | Hotfix back-merge covers every open release train and canary thresholds carry their tool-example provenance | `platform-release-engineering` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-release-engineering/references/promotion-gate-and-review.md#an active train missing the hotfix reverts it at that train's own merge | `updated` | Owner key `platform-release-engineering/SKILL.md`. Challenge-round: the earlier back-merge wording covered main and develop but not an open release branch (replayed at the round base), so a concurrent train could revert a shipped hotfix; the funnel invariant was also scoped as sanity-not-attribution and the error-budget comment renamed to a rolling error-rate threshold. |
443
+ | Benchmark M-verdict evidence pins command, scope, and baseline revision so replays run against the recorded baseline | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/source-to-skill-extraction.md#replay runs against the recorded baseline | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Challenge-round: a NO_HITS record without its baseline revision self-hits once the borrow lands (review-caught, replayed at the round base); this round also relocates the ledger's supersede note below the row table so machine and human readers see one uninterrupted row stream. |
444
+ | Provider-evaluation spend is enforced at admission with spent-plus-in-flight headroom and billed attempts missing usage count against the provider | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/references/model-prompt-evaluation.md#its cost enters via billing records, never silent exclusion | `updated` | Owner key `llm-inference-integration/SKILL.md`. Challenge-round: delayed usage/billing signals let an open-loop driver overshoot the cap and a provider omitting usage on billed failures flattered its ratios (replayed at the round base); the admission budget now subtracts recorded spend plus in-flight worst case, the stop latch halts admissions, and ratio exclusions are reported. |
445
+
446
+ | Spend enforcement is a reservation invariant (cap ≥ reconciled spend + outstanding reservations) and ratio exclusion never touches cost totals | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/references/model-prompt-evaluation.md#total-cost accounting is the separate aggregate that never excludes | `updated` | Owner key `llm-inference-integration/SKILL.md`. Challenge-round: three successive edge findings on the admission-budget arithmetic converged by replacing the enumeration with the reservation invariant (reserve worst-case at admission, release only on billing reconciliation), and the ratio-exclusion clause was split from total-cost accounting so neither aggregate can be flattered by usage omission. |
447
+
448
+ | The billing-records anchor phrase is preserved inside the split-aggregate wording so prior ledger locators keep resolving | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/references/model-prompt-evaluation.md#never silent exclusion from totals | `updated` | Owner key `llm-inference-integration/SKILL.md`. Locator-repair round: the split-aggregate rewrite had dropped the substring an earlier row anchors on (register_firing_path_unresolved, replayed at the round base); the phrase is restored within the new semantics so both locators resolve. |
449
+
450
+ | Spend reservations are atomic check-and-reserve on one ledger, closing the concurrent-headroom TOCTOU | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/references/inference-capacity-operations.md#read-headroom-then-reserve is a TOCTOU that lets concurrent workers jointly overshoot | `updated` | Owner key `llm-inference-integration/SKILL.md`. Challenge-round: two open-loop workers reading the same headroom could each reserve and jointly exceed the cap (replayed at the round base); the reservation is now an atomic check-and-reserve, completing the spend-invariant class alongside admission-halt and billing-release. |
451
+
452
+ | The hotfix full-ladder rule names the emergency-override section as its one sanctioned, loudly-logged exception | `platform-release-engineering` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-release-engineering/references/promotion-gate-and-review.md#the emergency-override section below is the one sanctioned, loudly-logged exception | `updated` | Owner key `platform-release-engineering/SKILL.md`. Review-round: the unconditional never-skips-a-gate wording contradicted the file's own emergency-override section during a production incident (replayed at the round base); the exception is now named inline so the two sections compose instead of conflicting. |
453
+
454
+ | Sub-threshold sample counts are reported as unreliable, never averaged into a verdict | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/references/inference-capacity-operations.md#must be reported as unreliable, never averaged into a verdict | `updated` | Owner key `llm-inference-integration/SKILL.md`. Review-round: the ~3-run floor carried no provenance label (replayed at the round base); it is now explicitly a team heuristic with a raise-per-variance instruction, and the rule moved to its own normative bullet. |
455
+
456
+ | The latency-SLI formula divides by the valid set, matching the good/valid definition in the same rule | `platform-observability` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-observability/SKILL.md#the denominator is the availability SLI's valid set, never raw total | `updated` | Owner key `platform-observability/SKILL.md`. Challenge-round: the bullet preferred good/valid while its own latency formula divided by raw total (replayed at the round base); the denominator now names the valid set explicitly. |
457
+ | Verbatim-bound obligation carriers survive wording compression only verbatim: a size-budget compression must first check the sentence against the frozen preservation mapping, and byte offsets come from sentences the same round added | `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/testing-strategy/SKILL.md#Do not describe such work as done, fixed, merge-ready, or release-ready | `updated` | Owner key `testing-strategy/SKILL.md`. Observed failure: two wording compressions in this round's size-budget offsets rewrote exact carrier sentences bound by the specs/065 obligation mapping (such-as→like; are-done→pass), and the heavy lane's real-repository obligation audit went red (CARRIER_COMPOSITE_NOT_UNIQUE count=0) while every entrypoint-scope gate stayed green — the preservation mapping is a verification surface the size-budget workflow did not consult. Fix: carrier sentences restored verbatim; equal-byte offsets taken from sentences this branch itself added (which the frozen mapping cannot bind); reader index regenerated at the mapping's pinned base/head. RED baseline (replayed): the repo-audit suite for the obligation ledger (`test_obligation_ledger_repo_audit.sh`) red before the restoration, `audit_ok domain=50 rows=1240 unresolved=0` plus `test_obligation_ledger_repo_audit_ok` after. |
458
+ | A wording compression that deletes a sentence boundary corrupts the hosting rule: the reinserted stop/continuation sentence must keep its full punctuation, and a fresh-eyes review of the reinsertion is what catches the truncation | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#explicit stop/pause needs no reconfirmation. A `continuing:` outcome | `updated` | Owner key `product-rd-workflow/SKILL.md`. Observed failure: the size-budget reinsertion of pinned continuation literals dropped a sentence boundary, leaving "needs no reconfirm A `continuing:` outcome" — two rules joined without punctuation on the entrypoint's stop/continue surface; caught by the supplementary post-delta review round (P1), invisible to every deterministic gate because pinned-phrase gates match their own literals only. Fix: boundary restored ("no reconfirmation. A"). Offset provenance, stated exactly: this entrypoint's continuation-gate block is wholly branch-rewritten (six modified base lines, no pure additions), so offsets necessarily live inside that rewritten block; a word-level diff against the base revision audited every token those compressions dropped — the two load-bearing drops it surfaced (metered model/tool scope; the missing-capability routing qualifier) are restored in their owners' rows, the ambiguity-or join was additionally reverted with its replacement byte taken from a demonstrably branch-added sentence (em-dash tightened to a colon in the dominant-approach clause), and the remaining list joins are recorded as audited-neutral. Where genuinely branch-added sentences exist, offsets come from them first. The truncation shape joins the round's carrier-restoration lesson: compression edits need a substring check against the frozen preservation mapping and plain sentence-boundary integrity. RED baseline (replayed): repo-wide grep for "needs no reconfirm A" one hit before the fix, zero after; the restored phrase greps exactly once. |
459
+ | A compression that drops a scope qualifier widens the rule it hosts: the external-pack routing clause routes only a MISSING method/tool-layer capability, and restoring the dropped word is the fix, not rewording around it | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/SKILL.md#M needs the functional-equivalent check | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: a size-budget compression rewrote the reference-only routing clause from routing a missing method/tool-layer capability to routing that layer categorically, which would displace locally covered P-verdict capabilities and contradict the functional-equivalent check landed in the same round; caught by the supplementary post-delta challenge round enumerating compressed sentences (same class as the metered model/tool qualifier drop fixed in the sibling owner). Fix: the missing-capability qualifier restored; byte offsets from this branch's own pointer sentence, whose semantics live in the owning reference. RED baseline (replayed): word-level diff against the base revision showed the dropped qualifier before the fix and shows it restored after; the class sweep over all three owner entrypoints found no further load-bearing drops. |
460
+ | Stop reporting keeps its specificity qualifiers: the entrypoint demands the concrete stop reason and the exact evidence checked, and budget offsets come from relocating clauses whose semantics already live verbatim in the owning reference | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#state the concrete stop reason and the exact evidence checked | `updated` | Owner key `product-rd-workflow/SKILL.md`. Observed failure: a size-budget compression dropped the-concrete/the-exact from the stop-reporting sentence, licensing generic stop reports; the challenger graded it load-bearing, the implementer's word-sweep had graded it neutral, and the maintainer's standing delegation resolves such token disputes by the repository's fail-closed obligation standard, so the qualifiers are restored. Byte offset: the stale-source parenthetical is removed from the entrypoint because its full sentence lives verbatim in references/pre-final-continuation-gate.md (Status-source reconciliation) which the same sentence already cites — relocation, not compression. RED baseline (replayed): grep for the restored phrase zero-hit on the pre-fix entrypoint, exactly one hit after; the removed parenthetical greps once in the owning reference. |
461
+ | The dual-track reviewer's verification scope is a documented boundary: content semantics belong to the reviewer, deterministic-gate claims to CI, historical-process claims are testimony unless receipt-bound — ruled on once so packet-verifiability findings stop recurring per round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#a finding that only restates this boundary is dispositioned against this rule, never re-litigated per round | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; the change lands in references/dual-track-review-gate.md (new Reviewer verification scope section). Observed failure: the packet-verifiability finding class recurred across four supplementary review rounds and roughly a dozen occurrences in this round's chains — every reviewer independently rediscovered that the packet cannot carry the deterministic oracles, and every round paid the same finding again because the boundary was undocumented. The maintainer confirmed the operating reality (all consumers and reviewers are agents; the human role is authority, not readership), so the boundary is now standing text agents can disposition against, with the receipt-embedding backlog item named in place. RED baseline (replayed): grep for the boundary phrase zero-hit before this change, exactly one hit after, on an added normative list line. |
462
+ | The Agent-autonomous review budget is summed across review chains, never per chain: in a self-hosted skill repository a finding fix that touches a selected-owner tree breaks the tracked chain by design (nearly every fix in such a round; a fix confined to files outside every selected owner drifts only the candidate hash and continues in-chain), so below the cap the break is recovered by a ledger-counted restart (batched dispositions first, full-context first packet, plan frozen with the candidate, final chain still able to hold the review-plus-challenge ready floor), and at the cap the designed terminal is disposition plus the honest terminal record — `continuation_authorization_required` when the final round itself returned findings, otherwise an interim record naming the last reviewed candidate and all later deltas — while restarting chains until a clean pass, or counting a restart as fresh budget, is the named contract violation | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#Sum spent rounds across all chains before opening one more | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; changes land in references/dual-track-review-gate.md — new self-hosted-chain rule with a four-step walked enumeration, the continuation bullet's dead-end sentence scoped to the at-cap case, an anti-pattern merge — and in references/extraction-quickstart.md 3e, whose two rerun clauses gain the cross-chain budget qualifier). Observed failure (production receipts, per-host archive, mechanism at the same commit as the current base): one round ran 20+ reviewer rounds, the next 12 restarted chains / 21 reviewer invocations to land a three-line diff, none human-authorized; receipts show every chain restarted at index 1 with a new candidate hash after fixes, while a control chain on an unchanged candidate ran review plus two challenges without invalidation. Root cause is a corpus-level contradiction, confirmed by frozen-criteria elicitation runs (n=3 per arm, isolated cwd, arms byte-identical to versioned text): the pre-change gate section alone elicits the STRICT reading three of three — zero autonomous restarts, interim-then-human after any break — while the quickstart page simultaneously mandated "re-run both on the updated candidate before landing" and the wrapper mechanically accepts fresh chains; jointly unsatisfiable, resolved in production by improvised unbounded restarts. RED-baseline: the occurred production failure plus the three-of-three strict/mandated contradiction on the pre-change corpus; post-change runs elicit the landed semantics three of three (cross-chain summing to the three-round cap, ledger-counted restart below it, terminal disposition at it, no laundering) with no contradiction-rationalizing text. Semantics delta declared honestly: below-cap restarts move from ask-human (strict reading) to Agent-autonomous ledger-counted rounds — a deliberate loosening grounded in the review-efficiency adjudication, the wrapper's three-round design, and two informed merges of rounds that ended in the terminal-disposition shape; the maintainer can revert to the strict reading by decision. Supersedes by pointer the recovery clause of the earlier four-process-controls row ("recovered by an interim checkpoint … plus a human continuation authorization"): that recovery is now the at-cap path only, and the row stays unedited per the append-only contract. Oracle: delete this row on the candidate and run the repo check with the base ref set — it prints the impact-chain missing token. |
463
+ | "Candidate edits do not reset Agent authority" promises no continuation: the selected-owner digest hashes each selected owner package's current working tree and owners derive from candidate paths, so a candidate edit inside any selected owner package invalidates every prior receipt and the next tracked round fails `review_chain_invalid` — for a self-hosted skill-repo candidate that is every applied fix, and a restarted chain re-enters the same cumulative Agent budget | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: file:skills/code-review/references/staged-review-contract.md#invalidates every prior receipt | `updated` | Owner key `code-review/SKILL.md` (entrypoint unchanged; the change adds two consequence bullets to the Agent review chain section of references/staged-review-contract.md). Observed failure: the section's tolerance sentence ("older candidate hashes; they remain consumed") reads as continuation-after-fix, while the stable-binding predicate in review_gate.py (this owner's script) compares controller digest, owner-selection source, owner names, and selected-owner digest on every prior receipt, with the digest computed over the owner packages' live working tree — so the documented tolerance is unreachable exactly when the candidate lives inside its owner package; archived receipts from one extraction round show 12 chains each restarted at index 1 with a new candidate hash after fixes, and a control chain on an unchanged candidate continuing three rounds. Declaration-contradicted-by-implementation class: the correction documents the implemented refusal instead of changing it — binding semantics, trust model, wrapper, and validator behavior are untouched. RED-baseline: grep for the consequence phrase is zero-hit before this change and exactly one hit after, on an added normative list line; the mechanism is reproducible from the chain-validation predicates plus the archived receipts. |
464
+ | The extraction lane's autonomous budget is one review plus one challenge, pinned in exactly two executable places and derived everywhere else: the wrapper passes the single value, the closeout validator computes every numeric bound from two module constants, and prose surfaces name the wrapper-fixed budget instead of repeating numerals — under this budget a fix-restart is never fundable, so the hold-fixes branch is the standing path, the self-hosted chain-break conflict becomes unreachable without touching the binding trust model, and a two-receipt final-round-findings chain validates as `continuation_authorization_required`, closing the cross-chain machine-terminal gap | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: command:skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (dual-track bullet numerals synced; frontmatter untouched; severe-entrypoint byte budget held at net zero by equivalent shortenings in the same bullet). The budget-size choice was maintainer-delegated in-session after two cost escalations (a review loop can burn two days on one issue) and restores the earlier one-review-plus-at-most-one-challenge adjudication for this repository; the challenge remains mandatory for every non-wording shared-skill change. Changes: extraction_review_gate.sh passes the fixed budget 1 and its guard message matches; validate_extraction_review_state.py derives round bounds, remaining counts, per-round state legitimacy, and the continuation predicate from WRAPPER_CHALLENGE_BUDGET/WRAPPER_AUTONOMOUS_ROUNDS (three hidden hardcodes found and converted during the green run: completion remaining, review_state round semantics, the continuation receipt count); both regression suites' fixtures converted from three-round to two-round shapes with occurrence semantics preserved (multi-finding rounds keep sweep-triggering occurrence counts). RED-baseline (applied): reverting the wrapper to the old budget in a throwaway edit turns test_extraction_review_gate.sh red at the budget-argument assertion and the restore is green; the validator suite was red at each hidden hardcode until converted, then fully green; catalog byte gate and implementation-gates suite green on the final candidate. Supersedes by pointer the wording of the two rows above where they cite a three-round cap or an unfundable-restart arithmetic tied to it: their production evidence and anchors are unaffected, and the budget-agnostic four-step enumeration they land is unchanged — only the numeral moved. Elicitation runs were re-taken against the final section with a corrected prompt (the scenario's own budget parenthetical had contradicted the attached rule text; the stale batch is archived unscored). |
465
+ | The closeout validator must reject a controller chain longer than the wrapper can mint: without an upper bound, a caller-supplied third receipt under budget one claims a negative remaining count and reaches completion validation as a ready-state budget bypass; and every prose surface advertising the retired budget is a laundering affordance, so the stale phrases are closed and pinned closed by documentation assertions | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; result-class: failure; firing-path: command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (unchanged this batch; changes land in validate_extraction_review_state.py — the receipts loop now fails `controller chain exceeds the wrapper budget` past WRAPPER_AUTONOMOUS_ROUNDS — its regression suite, the gate suite's documentation assertions, dual-track-review-gate.md, extraction-quickstart.md, and the register rows above). Provenance: the post-budget batch of this candidate's own two-round review chain — the final challenge (receipt archived per-host) surfaced the over-budget bypass and the stale third-round/two-challenge phrases; the review round surfaced the stale phrases independently, the uncertified-post-batch gap (closed by the pending-branch certification sentence in the terminal-disposition step), and a residual absolute in the consequence row above — superseded by pointer here, not edited in place per the ledger's append-only contract: read its "every applied fix" as "nearly every applied fix; a fix confined to files outside every selected owner drifts only the candidate hash and continues in-chain", matching the normative text it records. RED-baseline (applied, red for the right reason): the new over-budget fixture was added BEFORE the validator fix and the suite went red showing the unfixed validator accept the three-receipt budget-one ledger as ready_for_human_decision; after the one-line bound the case rejects with the named token and the full suite is green. Packet-verifiability findings about unshippable suite output remain dispositioned against the documented reviewer-verification-scope boundary with rerun oracles in the MR/PR body. |
466
+ | The assent-triggered blocked outcome states its interim classification with an explicit verb, not an elliptical fragment | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#form and classifies the turn `interim` | `updated` | Owner key `product-rd-workflow/SKILL.md`. Debt-repayment chain for the disclosed unreviewed pinned-phrase restoration commit 4e65dbc (patch-identical pre-rebase form 4b30746): replayed against its immutable base b0fe08b, `git show b0fe08b:skills/product-rd-workflow/SKILL.md` carries "form and classifies the turn `interim`." and the 4e65dbc diff drops "and classifies the", leaving the malformed "form, turn `interim`" — the independent review of this repayment chain confirmed the verb loss as the only defect of that diff still unrepaired at review time (the fused "reconfirm A" sentence boundary was repaired by 5fc25fb and the ledger-table break by the note relocation in 26b773e, both verified against the current file state); head restores the classification verb from the b0fe08b text of the same sentence, whose later specificity/carrier revisions elsewhere in the sentence are untouched. |
467
+
468
+ Supersede note (this round, before landing): the two crypto-erase rows above ("go-microservice-architecture" and "python-service-architecture") describe an earlier candidate state; the landed text attributes the CE do-not-use conditions to SP 800-88 Rev.1 §2.6 with Rev.2 (2025) superseding and continuing the framework — per the source-verification ledger row "CE conditions text location". The rows' RED-baseline probes and firing-path anchors are unaffected.
469
+
470
+ Round 073-receipt-bundling rows (new table so the entry renders as a table row after the supersede-note paragraph above):
471
+
472
+ | Upstream rule | Downstream owner | Expected executable behavior | Status (updated, unchanged, routed, or not-applicable) | Evidence |
473
+ | --- | --- | --- | --- | --- |
474
+ | Deterministic-gate process claims ride the packet as candidate-SHA-bound receipts: `gate_receipt.py` mints against the committed candidate (clean tree, HEAD, argv, exit code, output hash, bounded tail) and verifies differentially, so a pre-fix RED stops being unverifiable testimony | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#clean tree required, HEAD commit recorded | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (Reviewer verification scope — the "Standing backlog" clause is superseded in place by the landed channel) + scripts/gate_receipt.py + scripts/test_gate_receipt.sh (registered in the fast lane). Observed failure: the packet-verifiability class — historical-process claims graded implementer testimony across the prior round's review chains, terminally dispositioned accepted-limitation repeatedly, with the backlog row naming receipt embedding as the mechanical fix. Remedy re-derived against current code (a deferred registration is hypothesis, not spec): the packet already binds `--review-plan-file` evidence via review_context_sha256 and the v3 ledger already binds sibling files by hash, so no controller or validator change — the missing pieces were the receipt artifact class and its mint/verify tool, which is what lands. Trust model stated in the landed text so it cannot be oversold: candidate-bound falsifiable consistency evidence, not runner authentication; CI re-running gates stays the deterministic authority; unreceipted process claims remain testimony. RED baseline (applied mutations, differential): tampered output hash → rc1 output_hash_mismatch; tampered recorded exit code → rc1 exit_code_mismatch; foreign key → rc1 exact-key-set; wrong checked-out candidate → rc2 named no-verdict (not a false red); nondeterministic-output gate → full rerun rc1, --exit-only rc0 with scope token; control legs green. Same-round sibling disposition, no diff landed: the two-place numeric-contract backlog item (guarded-file whitelist + version-bump receipt + parse-values-from-prose) was re-probed against the current baseline and closed already-covered — the pinned/sync declared-pair registry owns the whitelist half and test_routing_pointer_integrity.sh's doc-vs-executor threshold parity check owns the parse half; that incumbent fired live twice this round against a draft duplicate (executor marker loss; a pinned firing-path phrase reworded), so the duplicate was deleted per same-class convergence-by-deletion and the final candidate leaves SKILL.md, description-authoring.md, and the checker cap logic byte-identical to base. |
475
+ | Load-bearing prose contracts are pinned declaratively: each contract-anchors.tsv row demands its pinned literal exactly once in the owner file, so deletion, semantic inversion, numeric falsification, or decoy duplication of a verdict-taxonomy discriminator, a stop-condition predicate, or an externally verified numeric tier turns the repo check red instead of passing clean | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_contract_anchors.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/check-contract-anchors.sh + contract-anchors.tsv (9 anchors) + the check-ccl-skills.sh delegation block (beside-the-validator resolution, pinned green grammar, fail-closed rc mapping) + the fast-lane suite). Observed failure: three enforcement-gap findings from the prior round's review chains, deferred needs_human_decision and scheduled into this round by the maintainer — reproduced against the round base before any implementation: deleting the fault-origin discriminator sentence, inverting the materially-differing/evidenced-cause stop predicates, and falsifying externally verified values (14.4→12.4, 97→87) each left the full check clean_ok while a control mutation (broken reference link) went red, proving the instrument could fail. RED-baseline (applied, differential): the same probe mutations now red the gate with per-anchor attribution; suite mutants M1–M8 red for their named reason (missing/duplicate/file-missing/empty-table/malformed-row/short-literal/duplicate-id/table-missing), benign neighbors B1–B3 green, D1 names only the broken anchor, and the whole suite goes red under an always-green checker stub. Scope stated honestly: anchors make contract-wording and pinned-value changes conscious (same-MR table edit), not externally re-verified — external-truth re-verification stays with the documented reviewer-verification-scope boundary, and stop-predicate semantics testability is routed to the evals layer (deferred, bound to that round's entry). Registered-remedy narrowing recorded: the N×M forbidden-token matrix is not landed (three mechanized anti-patterns, the 1×1 pending/clean exclusion, and the sync registry already own every observed shape; recurrence re-opens it) and numeric copy-prohibition narrowed to both-sides anchors (entry + reference pinned together) because repo values are already single-owner. |
476
+ | A pinned-phrase gate family must be provably able to go red for the right reason: one applied deletion mutation per family runs the FULL shipped checker against a committed fixture clone and demands that family's own red token, with the unmutated control run green on every family token; and every anti-patterns panel section must carry its Grep recipe | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_pinned_phrase_mutation_walk.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in the heavy-lane walk + the fast-lane panel structural check test_antipattern_grep_panel.sh). Observed failure: none of the ~40 required_phrase pins across the four inline gate families had ever been proven able to red — the always-green-oracle class this repository already met as the degenerate-fixture worktree-pruning precedent and the empty-glob false-green lesson (an emptied phrase list or quoting regression would certify silently). RED-baseline (applied, differential): walk legs W1–W5 each delete one currently-pinned phrase in a committed fixture clone and the full checker reds with exactly that family's token (project-assessment, task-retro/teammate-trigger, test-case-first, product-rd anchor, and the new contract-anchor delegation), ~70s total; the control leg is green with all five family green tokens and a loop-count floor guards list vacuity; the panel check reds on a stripped Grep line naming the right section and stays green for a benign non-anti-pattern section. Environment pitfall recorded for reuse: CCL_SKILL_BASE_REF must stay unset around nested gate runs — it leaks into child validators' synthetic self-test repos where HEAD always resolves and flips their no-base→degraded legs into false passes. Rule side already-covered: the killing-mutation walk, benign-near-miss precision rows, and oracle-validation duties are owned by testing-strategy and the dual-track Self-audit section — no prose added anywhere; the firing path for this failure class is now these suites in the fast/heavy lanes. Grep pattern-compile validation deliberately discarded (panel recipes mix GNU-BRE commands with prose instructions by design; a compile check would false-red on regex-dialect differences without protecting a real contract). |
477
+
478
+ | Eval-first authoring fires pre-draft: a NEW skill needs three-plus scenarios and a no-skill baseline before body text, and a non-failing control stops the draft | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#must define eval/pressure scenarios, baselines, acceptance criteria pre-draft | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. 075/R3 round (S9). Both external sources re-verified primary-source on 2026-08-31: Anthropic "Skill authoring best practices" (https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices, section "Evaluation and iteration" / "Build evaluations first") — create evaluations BEFORE writing, build three scenarios, establish the baseline without the Skill, write minimal instructions, and no built-in runner exists; superpowers plugin 6.3.0, skills/writing-skills (SKILL.md sections "The Iron Law" and "Micro-Test Wording Before Full Scenarios"; testing-skills-with-subagents.md section "RED Phase") — a failing test first for new skills AND edits, 3+ pressure scenarios, and a no-guidance control that does not exhibit the failure means stop, do not author. The commit-time RED-baseline contract is untouched in this diff (the semantic-control leg); the change moves the firing point onto the pre-draft transition per the firing-point-placement corollary, with mechanics merged into the existing canonical bullet in validation-and-landing.md rather than appended as a new rule. Zero-loss map for the rewritten Workflow step-2 sentence: the subjective/high-impact category list survives (frontend/client shortened to client, same referent), pressure scenarios widen to eval/pressure scenarios, acceptance criteria survive verbatim, and before-editing tightens to pre-draft; nothing dropped, NEW-skill coverage and baselines are the additions. |
479
+
480
+ | Deterministic anchors pin stop-predicate wording while paired body-compliance probes grade classification on the gate's own continuing:/blocked: markers | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#probe subset on this machine before landing | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Closes the 071-chainB-r1f1 semantic-testability follow-up deferred by 074 to R3. Registered form (invert predicate, gate stays clean) no longer reproduces: the entrypoint anchor gate, the register firing-path anchor, and the size ratchet each redded a separate applied polarity/reword probe in a disposable checkout at the round base, with a pristine control leg green first. Evolved form reproduced and is the RED leg: a word-compensated additive neutralization (appending an advisory-continue sentence while deleting equal unpinned words) exits 0 with ccl_skill_check_clean_ok. With-change leg: four paired prd-* probes graded 4/4 on the pristine body after instrument fixes (marker-decoration tolerance with a mention-vs-verdict grammar; shared-scaffolding single-variable isolation), and the probe set demonstrably can fail (first run graded 2/4 on real output, one miss being a genuine same-case-two-classifications observation); run reports committed under eval/evidence/stop-predicate-probes-2026-08-31/. Honest boundary recorded in the f4 layering section (eval-routing.md points there): the same applied neutralization mutants did not flip live agent behavior either (two mutants by two probes, small N) — probes carry behavior drift, not buried-sentence tripwires. |
481
+
482
+ | Negative controls and coverage-gap probes are first-class routing-bank rows (expected "none" sentinel, acceptable alternates, bait neighbors), with absorbed / ownership_split as labeled outcomes and clarify / low-confidence / replica-agreement as first-class report metrics | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#已修复的路由 miss 必须把其 utterance 冻结成 bank task; bank-evidence: command:skills/skill-extraction-workflow/scripts/eval-routing-bank.rb | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Live at the round base: grading the new gap probes with two replicas showed one replica absorbing each probe into a coordinator skill while the other rejected it — the runner's absorbed + ownership_split labels fired on real grader output before the expectations were corrected with acceptable[]; integrity-lane mutants (sentinel in must_not, acceptable restating expected, unknown acceptable target, acceptable∩must_not, empty-string fields) each turned test_routing_bank_integrity.sh red on its own named assertion with the unmutated control green, and the validator self-proof section now replays those mutants on every run. Durable artifacts (exact invocation, grader model, candidate fingerprints, raw per-replica verdicts, red/green transcript, gate exits): eval/evidence/routing-negative-controls-2026-08-31/. |
483
+
484
+ | A before/after routing comparison may vary only ONE routing variable (one description, or one skill's indivisible routing face) for its delta to be attributable | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#一次改前/改后对照只准动**一个路由变量** | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source-side paired single-description A/B is the observed working mechanism (borrow round; sanitized provenance in the private alias archive); no mis-attribution incident observed in this repository yet, so the clause lands as protocol item 5 with the existing four items unchanged as the paired control. |
485
+
486
+ | Cross-skill / cross-reference routing pointers in body text must carry the routing quadruple (trigger / scope / output / return point); a bare "refer to X if useful" pointer is never a landing shape | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/description-authoring.md#routing pointer in body text must carry the routing quadruple | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Borrowed from an adopting skill pack where unbounded pointers were the dominant dead-routing shape (two adopters verified in source; sanitized provenance in the private alias archive); landed in the routing-surface authoring reference because the entrypoint is size-ratcheted level — the reference is the required pre-edit reading for routing-surface work, and eval-routing.md's silent-skip row points back at it; the description-side Skip-when idiom already satisfies the quadruple and is named as the unchanged control. |
487
+ | Review-finding fixes are held un-applied until the full review+challenge chain has run on the frozen candidate: under the 1+1 budget apply-now is never fundable after the review round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (enumeration item 1 + cadence Round 2). Observed failure: a prior round applied its review fixes before the challenge; the tracked chain broke (challenge binds to the round-1 candidate), the round lost its double-receipt terminal, and closure required a user-granted continuation chain. The prior wording stated the rule only as a fundability conditional whose arithmetic the agent under pressure never ran; the operative unconditional form (hold all fixes; challenge on the frozen candidate; land the batch after the chain) is now explicit at both firing points. RED baseline: the recorded chain-break incident is the without-change failure; the with-change compliance surface is the explicit hold rule at the enumeration walked when a round returns findings. |
488
+ | A frozen eval case is sacred: deleting or re-scoping a bank task or golden trace requires a same-round `case-retired:`/`case-rescoped:` register adjudication row, and average improvement never offsets a frozen-case loss | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#平均改善不得抵消单条冻结案例的失守 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/eval-routing.md (冻结案例神圣 bullet), scripts/test_frozen_case_sanctity.sh (fast lane, registered), docs/f4-skill-effectiveness-harness.md (pointer line). Rule semantics: a previously-passing frozen case that degrades — including to unsure/INCONCLUSIVE — is a regression, and its sacredness attaches per case, so no aggregate improvement offsets it; the mechanism's source-verification record lives in the round's private archive, and transferred evidence does not exempt the behavioral row. RED baseline (replayed, throwaway clone at the round base): deleting the non-pinned bank case ctrl-unit-test passed the pre-change surface silently (test_routing_bank_integrity.sh exit 0) and reddens the new gate (exit 1 naming the id and the required adjudication row); re-scope and golden-trace-deletion mutants red for the right reason; adjudicated-deletion and untouched-tree control legs green; no-base and unresolvable-base legs print the explicit skip token. |
489
+ | Rationalization tables are built from excuses captured verbatim in baseline/pressure runs, never invented: counter only what a run actually said, and new observed excuses accrete counter rows | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/validation-and-landing.md#counter only what a run actually said | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/validation-and-landing.md (eval-first steps 2 and 4) with the sourcing clause mirrored into references/rule-consolidation.md's discipline-slip form row. The clause is scope-bounded to discipline-slip failures (prohibition-form counters measurably backfire on shape/omission failures per the form table it points at); the source-verification record lives in the round's private archive. RED baseline (fair tempting scenario, headless fresh-context, small-N 2x2, fully separated): asked to harden a merge-authorization gate with no run data supplied, the no-clause arm invented 12+ Excuse-Reality rows in both reps with zero mention of captured evidence; the with-clause arm produced zero invented rows in both reps and stopped to request actual run transcripts. |
490
+ | The draft-time security axis walk screens skill text as a prompt: leakage-inducing, permission-overreaching, or unsafe-automation wording is named at the axis-1 instance list and may appear only as a labelled anti-example | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#text a draft must not carry except as an explicitly labelled anti-example | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (pre-cover axis (1) instance list). Semantic control: axis (1) already required a security/authority pre-cover with at least one negative case per applicable axis; this names three instances (prompt leakage, overreach, unsafe automation) inside the existing obligation rather than adding a new gate, so the walked enumeration, the challenge mandate, and every other axis are unchanged; existing anti-example discussions stay legal via the labelled-anti-example carve-out. |
491
+
492
+ | Reference files carry a delta-ratcheted line budget: a new or crossing `skills/*/references/**/*.md` over 500 physical lines blocks, an already-over reference may shrink or stay level but never grow, the append-only ledger is excluded by gate design, and a long new reference without `##` structure draws a navigation advisory | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#stays inside the reference line budget | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in scripts/check-size-budget.sh (reference ratchet + counters + `reference_line_budget_blocking_ok` / `_failed` / `_unevaluated` verdict tokens), scripts/test_check_ccl_size_budget.sh (legs g0-g9, registered in the fast lane through the existing size-budget suite entry), and references/attention-budget-ratchet.md (the five design invariants every size/budget gate must satisfy: stable proxy estimator, anti-false-green sentinel, zero tolerance for new debt, legacy-shrink-only, missing-baseline-is-never-a-pass — plus the write-side reference norms and the per-clause verdicts on the external source). Write-side gap was measured before implementing, not assumed: 338 reference files, 4 over 500 lines, 104 between 101 and 300, and zero carrying a table-of-contents block. Paired RED at the round base in a throwaway checkout: a new 703-line reference passed the pre-change gate silently (exit 0, `entrypoint_size_blocking_ok`) and reds the new one (exit 1, naming path, head_lines=703 and the 500 allowance); on the live corpus, appending one line to the 691-line source-to-skill-extraction.md blocks as `over-limit reference grew`, deleting three lines passes with `reference_line_budget_legacy_ok allowed_lines=691`, appending 200 ledger rows passes with the `ledger_excluded` visibility token, and the untouched tree is green with over_limit_count_delta=+0 (the corpus is frozen, not retroactively reddened). Oracle self-proof (applied mutants on a copied gate, differential, control green): disabling the over-budget branch, allowing legacy growth, an off-by-one budget, removing the ledger exclusion, downgrading the unknown-base partial to a print, dropping the reference verdict from the clean-exit aggregate, and promoting the navigation advisory to a block each red on their own named leg; a first mutant attempt that broke the program was discarded as red-for-the-wrong-reason and re-applied semantically. Author-dogfood leg: the gate blocked this round's own entrypoint edit at +56 then +18 body words until the landing was funded by consolidation, which is the invariant working on its author. Review-chain repairs folded into the same round commit (tracked chain 078-r6, 1 review + 1 challenge, all four findings applied after the chain closed on the frozen candidate): line endings are normalized to LF before the line and heading counts, so a CR-delimited 501-line file can no longer read as one line and earn a clean verdict; the navigation advisory counts EXACT H2 headings, so an H3-only long reference still draws it; the head census guards every read and degrades its COUNTER to unknown rather than aborting the program before the per-file fail-closed verdicts run; and the legacy freeze gained a level-edit leg so a `>` silently becoming `>=` cannot pass. Four further mutants (H3-counts-as-section, no line-ending normalization, level-edit-blocks, census-error-reads-zero) each red on their own new leg with the control green, the last of them only after its leg was added — it survived the first re-walk, which is why the walk was re-owed after the fixes. Registered-claim narrowing recorded (a deferred registration is hypothesis, not spec): the official 500-line figure is SKILL.md-scoped and already covered more strictly by the body-word ratchet, the registered table-of-contents mandate is NOT PRESENT in the primary source (the official remedy is one-level references) so it landed as a `##`-structure advisory on repo-internal evidence only, and the three-model test matrix is not mechanized because this repo ships no model-pinned skills. |
493
+ | Core Rules own each invariant and a Workflow step only points at it: same-facet text living in both surfaces is converged toward the canonical surface rather than restated, and the sweep enumerates candidates instead of fixing whichever one a diff happens to touch | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/SKILL.md#do not make normal users route through a source name | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The Step 6 cross-section facet-ownership check already owned this rule and fires on any edit touching Core Rules or a step; this round ran it as a five-candidate enumeration rather than a spot fix (the user challenged an earlier framing that treated the duplications as a word-budget offset ledger). Verdicts: content placement (Core Rules content-placement bullet vs two Step 5 bullets) converged to a pointer; the sibling-generalization mini-map field list (Core Rules owner-generalization group vs Step 4) converged to a pointer keeping only the step-order clause; capability naming (Core Rules naming bullet vs Step 5 vs two Step 6 checklist lines) converged to one Step 5 pointer plus one merged Step 6 residual-search check; representative sampling (Core Rules full-ask prohibition vs Step 3 labeling duty) judged complementary and left unchanged; the source-register row schema (Step 3 vs references/source-register.md) left unchanged as out of this contract's Core-Rules-versus-step scope and useful where an author builds rows. Zero-loss obligation map for the four rewritten passages: entrypoint-owns-trigger/routing/non-negotiables and references-own-detail both survive in the Core Rules content-placement bullet; the mini-map field list, the `update`/`unchanged`/`route-to-shared` vocabulary, and the smallest-common-owner routing survive verbatim in the Core Rules owner-generalization bullet, with `then add stack-specific implementation notes only where needed` kept in the step; the capability-name examples, the never-name-after-source clause, and the do-not-route-users-through-a-source-name clause all survive in the step, and provenance-labelling survives in the Core Rules naming and provenance bullets; the Step 6 residual-search list gains the page-name and scenario-label terms the two merged lines carried separately, and keeps `absent from executable guidance or clearly marked as provenance`. Net effect on the frozen entrypoint: base_body_words=16759 head_body_words=16750 (-9), so the round funded its own additions and left the entrypoint smaller than it found it. |
494
+ | A base-relative gate's design-time premise is measured against EVERY base its landing faces resolve, never the current round's base alone: the set is read off CI's own base-resolution expression rather than guessed - one face per pull-request target branch plus the pushed branch's previous tip on a push build - and each resolution is a separate run of the author-dogfood leg, a difference that accumulated before the gate existed surfaces as one violation on the face nobody measured, and the repair is to shrink the frozen surface rather than add a cross-base exemption | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#enumerate the landing faces, never assume one | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key; the changed file is `references/dual-track-review-gate.md`. **Observed failure (RED, recorded incident + re-computable).** The reference line ratchet landed in the prior round measured only against the round's own base. Reproduction on the pre-fix candidate: `CCL_SKILL_BASE_REF=origin/main bash skills/skill-extraction-workflow/scripts/check-size-budget.sh .` printed `reference_line_block: ... dual-track-review-gate.md: over-limit reference grew base_lines=662 head_lines=668` and `reference_line_budget_blocking_failed`, while the same command with `CCL_SKILL_BASE_REF=origin/dev` printed `reference_line_budget_blocking_ok`. The gate behaved exactly as documented; the defect was that leg (a) said `the SAME base resolution CI uses` (singular), so the author measured one face and the six lines that older rounds had added to a frozen surface only became visible on the promotion face. **Compliance (with-change).** Same two commands on the head candidate: base=origin/main `base_lines=662 head_lines=661 allowed_lines=662` and base=origin/dev `base_lines=668 head_lines=661 allowed_lines=668`, both ending `reference_line_budget_blocking_ok`. Running BOTH faces is this round's own dogfood of the rule it lands. **Zero-loss obligation map for the five consolidations that funded the addition** (all within the same file per the frozen-reference funding rule; 668 to 660 lines, and 125978 to 126158 bytes - the line unit the ratchet measures fell, the byte count rose by the amount the new obligation text exceeds the recovered duplication, which is stated here rather than hidden). (1) The four-line raw-CLI preamble collapses to one line carrying all three of its propositions - diagnostics only, never review or challenge evidence, never a replacement for the owner wrapper on a non-wording lane. (2) The standalone do-not-iterate-to-zero-findings paragraph merges into the convergence-bar paragraph, keeping the design-tradeoff clause, the pre-existing clause, the over-correct-or-scope-creep clause, and both contrasts - the bar differs from zero findings AND from no-new-P0-P1. (3) The R0-evidence value menu was stated verbatim three times; two occurrences become pointers to the review-pass row that keeps the menu, matching the pointer form the adjacent Item-9 field already used. (4) The Rules bullet restating the behavioral-evidence table's own two rows is dropped; its one clause absent from the table, that an author cannot self-assert semantic-control, moves into the semantic-control cell itself. (5) The trailing what-a-fallback-is-worth paragraph folds into the reviewer-ladder item that already owns that question, keeping the ad-hoc-run bar, the remediation-versus-evidence distinction, and the interim rule. No obligation was dropped and no paragraph was re-wrapped to buy lines. **A sixth consolidation was attempted and reverted, which is the reusable finding.** The premise-verification bullet in the self-audit section reads as a verbatim restatement of the two paragraphs above it, and deleting it looked free; `register_firing_path_unresolved` then failed because an earlier round's register row anchors its firing path on that exact line. A rule line can be another row's evidence, so an append-only ledger makes some prose non-deletable: check the anchor set before treating any rule line as redundant, and restore rather than EXEMPT when the deletion was to fund your own budget. **Other owners.** `references/attention-budget-ratchet.md` is `unchanged: already-covered` with a real firing path - its five-invariant preamble already says the budget gates are `the budget-gate instantiation of the design-time operability check in dual-track-review-gate.md - run that check's four legs too`, so the tightened leg (a) reaches ratchet authors through that pointer and restating it there would be the same-facet drift the Step 6 check forbids. `scripts/check-size-budget.sh` is `not-applicable`: the gate is correct as shipped and judges whatever base it is handed, so multi-base topology belongs to the caller, not the script. **Residual risk, stated rather than hidden.** The firing path is a walked design-time enumeration in prose, not a mechanical multi-base run; `Makefile` still defaults `CCL_SKILL_DEFAULT_BASE_REF` to the integration branch, so a local check still measures one face unless the author enumerates. A mechanical all-faces target is deferred with an owner - it would be a new gate surface owing its own four legs, oracle self-proof and suite registration, which is a round of its own rather than a rider on this one. |
495
+ | A design-time obligation over a SET is discharged by a manifest a reviewer can diff against the set's authoritative source, never by an asserted walk: the manifest names one entry per member with the value that member resolved to and the check's verdict on it, and a set with no finite manifest is a blocking residual routed to the existing non-blocking or risk-owner-deferral exits rather than reported as coverage | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#recorded manifest, never by a walk you assert | `updated` | `skill-extraction-workflow/SKILL.md` is the owner key; this is the post-review fix batch of the same round, kept as its own commit so the externally reviewed candidate stays identifiable on the branch. **Observed failure (RED).** The first formulation of the preceding row's rule said to run the design-time leg `once per resolution CI can produce`. Both lanes of the dual-track chain independently reached the same defect on the frozen candidate: the review lane found the required set is not statically enumerable because an unrestricted pull-request trigger can resolve any target branch, so an author must either guess a subset and falsely claim completion or cannot satisfy the rule; the challenge lane found the same accumulated cross-base violation can therefore still ship despite apparent compliance, since the unchanged per-base gate cannot detect the omitted face. Two independent lanes converging on self-certifiability is the self-adjudication shape this file already names - the classification verb had no named output that produces the classification. **Fix.** The obligation now takes a manifest derived from the workflow itself, one entry per landing branch carrying the resolved base ref and the gate verdict, which a reviewer can diff against the workflow; and an unenumerable set is routed to the exits the premise-check leg already defines instead of being claimed as covered. No new mechanism is introduced - the fix converts an author-adjudicated claim into a reviewer-checkable artifact using exits that already exist. **Chain state.** Wrapper-fixed budget of one review plus one challenge, both bound to the same frozen candidate digest `70fd34678f99c444bf5b7c808a283380e99201eada6d71d076fd08e2f2ec1789`; fixes were held un-applied across both rounds. The final round itself returned findings, so the honest terminal state is `continuation_authorization_required` and this batch is post-review, not reviewed. A third review finding asked for the workflow file and captured command outputs inside the packet; it is dispositioned per this file's reviewer-verification-scope rule - deterministic-gate claims are verified by CI re-running the gates on the branch, never accepted from the implementer's prose - with the packet-composition miss recorded as a process defect for the next round rather than re-litigated here. **Supersedes two cells of the preceding row.** (i) Its compliance figures read `head_lines=661`, drafted before a later edit in the same round took the file to 660; the correct values are base=origin/main `base_lines=662 head_lines=660 allowed_lines=662` and base=origin/dev `base_lines=668 head_lines=660 allowed_lines=668`, both `reference_line_budget_blocking_ok`. (ii) Its rule cell describes the face set as read off CI's base-resolution expression; the manifest form in this row is the one that governs. The correction is recorded here rather than by editing that row: the ledger is append-only and the gate enforces it mechanically - a row edited after its own round committed no longer survives at HEAD, its round loses its only row, and the gate fails closed. That is the append-only contract catching an in-place fix, which is what supersede-by-pointer exists for. |
@@ -677,3 +677,15 @@ Use these checks before treating a source-derived skill as ready:
677
677
  - New skill: create only when the future user would naturally ask for a different task type.
678
678
  - Reference file: use for detailed checklists, variants, examples, and source-derived heuristics.
679
679
  - Script: use for repeatable validation, scanning, generation, or formatting that should be deterministic.
680
+
681
+ ## Benchmark Verdict Discipline
682
+
683
+ For an external-pack / sibling-repo benchmark round (the `SKILL.md` benchmark Core Rule owns when this fires), the per-mechanism verdict table replaces impressionistic gap lists. Each source mechanism gets one row:
684
+
685
+ - **Verdict, four values**: `P` (we cover it correctly — requires a decision point, not just the concept named), `I` (covered but incomplete — name the missing branch), `M` (missing — but see the functional-equivalent check), `W` (we state it wrong — quote both sides with file:line and propose the fix).
686
+ - **Grep-anchored evidence**: every row carries the grep terms used against our tree; an `M` verdict requires the zero-hit grep recorded as replayable evidence — terms plus the exact command/mode, path scope, and baseline revision (`NO_HITS: <terms> @ <rev>`), because once the borrow lands, a replay against HEAD hits the landed rule itself; replay runs against the recorded baseline.
687
+ - **Functional-equivalent check before any `M` borrow**: a missing *name* is not a missing *capability* — search for what our tree does in that situation under other vocabulary; when an equivalent exists, the verdict is `P(功能等价)` or `I`, and the borrow shrinks to naming/linking value (usually low). This is the mechanical guard against borrowing-for-the-sake-of-alignment.
688
+ - **Borrowing degree** per `M`/`I`: high / medium / low, scored on one axis only — how much the mechanism changes an agent's actual decisions — never on how impressive the source text reads.
689
+ - **Disposition vocabulary**: beyond keep/merge/discard/route, two middle states are legitimate and prevent binary misjudgment — `待实验` (hypothesis worth testing: record the hypothesis, the experiment, and pass/fail criteria; do not land prose) and `条件化采纳` (adopt with explicit applicability conditions, disable conditions, and a fallback path recorded next to the rule).
690
+ - **Conflict synthesis, three shapes before choosing sides**: when the source contradicts our rule, first test whether the conflict is (a) a missing tier — both are right in different regimes, so merge by adding the regime split; (b) a missing ordering — both rules survive with an explicit precedence; (c) conflated authority — one side owns proposing and the other owns deciding, so split the powers. Only when none fits is it a genuine pick-one decision (keep the stricter per Conflict Resolution).
691
+ - **W-type internal-consistency sweep**: a mature skill tree's highest-yield audit is internal drift, not external comparison. The recurring W classes to grep for: an absolutized single-case rule (one scenario promoted to an unconditional "always/never"), same-name-different-meaning across files, translation/paraphrase drift between sibling-language files, a template violating its own stated hard rule, stale-era residue (rules about tools/versions no longer in the stack), a hard-coded value where a policy/parameter reference should be, and version-lagged claims about fast-moving vendors. Route wording-level dedup to `tighten-doc`; land each confirmed W as a fix with both sides cited.
@@ -56,7 +56,7 @@ If the validator reports `missing_required_command`, keep the failure visible an
56
56
 
57
57
  ## Behavioral Validation
58
58
 
59
- - For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology before writing or finalizing the skill. The code-level RED-GREEN-REFACTOR method (write the failing case first, watch a fresh agent violate the rule WITHOUT the skill, then add the skill and watch it comply) is owned by `superpowers:writing-skills` + `superpowers:test-driven-development` — **if installed, route there; otherwise apply the RED-baseline rule inline** (manually record the without-change failure and the with-change compliance). This is the skill-authoring face of **eval-driven development** (for a behavior/routing change, run the scenario before you finalize; never special-case the scenario just to make it pass) — borrow the *principle*, not a claim of production-grade eval rigor.
59
+ - For new skills or major workflow changes, use `writing-skills` for RED-baseline/test-first methodology — **and the firing point is BEFORE drafting the body, not only before finalizing**. Eval-first authoring for a NEW skill (or a new hard-rule section): (1) write the evaluation scenarios first — **at least three** for a new skill (both the vendor's published authoring guide and the high-star practice pack converge on three-plus scenarios before body text; a single-rule edit may scope down to that rule's own scenario); (2) run them WITHOUT the skill and record the observed failures verbatim — and for a discipline-slip failure (the agent knows the rule and skips it under pressure), capture the agent's rationalizations word-for-word: each verbatim excuse is the raw material for one rationalization-vs-reality row and one red-flag line in the skill text (the discipline-slip form in `rule-consolidation.md`'s form-by-failure table); an invented hypothetical excuse does not qualify — counter only what a run actually said, and don't add rows for excuses no run produced; **a no-skill control that does not exhibit the failure is a stop signal — do not author guidance for a failure you cannot observe** (record the null finding instead; this is the pre-draft face of "Evidence must come before new rules"); (3) draft the **minimal** content that addresses the observed failures, then re-run the same scenarios WITH the skill; (4) when a later run, review round, or live miss surfaces a NEW rationalization for an existing discipline gate, add its explicit counter row to that gate's table and re-run the tempting scenario — counter tables accrete from observed excuses across rounds, never from imagination. The code-level RED-GREEN-REFACTOR method (write the failing case first, watch a fresh agent violate the rule WITHOUT the skill, then add the skill and watch it comply) is owned by `superpowers:writing-skills` + `superpowers:test-driven-development` — **if installed, route there; otherwise apply the RED-baseline rule inline** (manually record the without-change failure and the with-change compliance). This is the skill-authoring face of **eval-driven development** (for a behavior/routing change, run the scenario before you finalize; never special-case the scenario just to make it pass) — borrow the *principle*, not a claim of production-grade eval rigor.
60
60
  - **What makes a `RED-baseline` valid is executed-and-recorded vs narrated — not recorded vs live.** Any evidence form (before-after diff, golden trace, or pressure scenario) is valid when it actually records the without-change failure AND the with-change compliance with a locator + expected-vs-actual (see `dual-track-review-gate.md`). Prose that merely *describes* an expected failure without running it is not a baseline; a pressure scenario you actually executed and recorded is.
61
61
  - **A claimed impossibility is falsified in-env before it lands (authoring and reviewing alike; relocated from `SKILL.md`).** Deferring work behind a conservative-sounding stub ("not implemented", "needs a future tested wrapper") is the avoidance form of a blocked-verification claim: it *feels* safe but ships an unverified impossibility as durable behavior — distinct from a real `unavailable`/`pending`-with-remediation+residual-risk record, which is what remains after a safe attempt was genuinely impossible (the attempt itself stays within the safety boundaries the `SKILL.md` rule names). Failure shape: a fallback reviewer lane shipped as a permanent fail-closed stub on an unverified "permission probe not implemented" premise that a ~60-second live tool run refuted, re-enabling the lane.
62
62
  - **A headless code-writing RED/GREEN needs a *fair* violation-tempting scenario + an *independently-valid* objective measure — a capable agent complies on a clean task.** When the rule governs how an agent WRITES code, a clear well-specified prompt usually makes even the no-rule baseline produce compliant code (zero delta → no RED), so a clean-task baseline proves nothing. Surface the RED with **realistic inherited pressure** — a real legacy/house convention the rule must override, or a genuinely ambiguous spec — **NOT an explicit instruction to emit the anti-pattern**: leading the baseline directly into the violation launders a constructed failure into "RED" and is fabrication (the same defect as the rule above), so record why the prompt is fair and mark any direct-leading scenario synthetic/advisory, not RED. Score both runs with an **objective measure that detects the ANTI-PATTERN** — the rule's own checker counts ONLY if it was independently validated first (held-out positive/negative fixtures + a documented residual boundary, so RED genuinely fails); a same-change checker that merely whitelists the expected GREEN form makes GREEN trivially true. Run the agent **cross-model / fresh-context** so the baseline isn't primed by your session. If even the fair tempting baseline complies, that is an honest finding — the rule's marginal value is in edge/legacy cases, not the common one — record it, don't manufacture a RED.
@@ -1359,6 +1359,36 @@ for required_phrase in \
1359
1359
  done
1360
1360
  echo "test_case_first_gate_ok"
1361
1361
 
1362
+ # Contract-anchor gate: declarative wording-existence pins for load-bearing
1363
+ # prose contracts that structural checks cannot see (verdict-taxonomy
1364
+ # discriminators, stop-condition predicates, externally verified numeric
1365
+ # tiers). Checker and its contract-anchors.tsv table resolve NEXT TO THIS
1366
+ # VALIDATOR (one-versioned-unit rule, same as the sync-pointer gate — an
1367
+ # in-tree copy could be doctored to exit 0). Red = a pinned contract sentence
1368
+ # drifted, was deleted, or became ambiguous; the remedy is printed by the
1369
+ # checker (restore the wording, or update the anchor row in the same MR for an
1370
+ # intentional contract change). Self-proof: test_check_contract_anchors.sh
1371
+ # (fast) and test_pinned_phrase_mutation_walk.sh (heavy).
1372
+ contract_anchor_script="$checker_scripts_dir/check-contract-anchors.sh"
1373
+ if [[ -L "$contract_anchor_script" || ! -f "$contract_anchor_script" ]]; then
1374
+ echo "contract_anchor_infra_failed: check-contract-anchors.sh missing or not a regular file beside the validator: $contract_anchor_script" >&2
1375
+ exit 2
1376
+ fi
1377
+ contract_anchor_rc=0
1378
+ contract_anchor_out="$(bash "$contract_anchor_script" "$root")" || contract_anchor_rc=$?
1379
+ [ -n "$contract_anchor_out" ] && printf '%s\n' "$contract_anchor_out"
1380
+ if [ "$contract_anchor_rc" -eq 1 ]; then
1381
+ echo "contract_anchor_gate_blocking_failed: pinned contract wording drifted (diagnostics above)" >&2
1382
+ exit 1
1383
+ elif [ "$contract_anchor_rc" -ne 0 ]; then
1384
+ echo "contract_anchor_infra_failed: rc=$contract_anchor_rc (contract-anchor gate could not run — fail-closed)" >&2
1385
+ exit 2
1386
+ fi
1387
+ if ! printf '%s\n' "$contract_anchor_out" | grep -qE '^contract_anchor_gate_ok \([0-9]+ anchors\)$'; then
1388
+ echo "contract_anchor_infra_failed: green output grammar missing (expected: contract_anchor_gate_ok (N anchors))" >&2
1389
+ exit 2
1390
+ fi
1391
+
1362
1392
  # Post-cleanup register check: after a project-specific extraction batch is
1363
1393
  # migrated out of the shared skill tree, the source-register.md template must
1364
1394
  # not silently grow new project-specific dated entries. The rule in