@ccoalm/ccl-skills 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/README.md +2 -2
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +8 -7
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +1 -1
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +16 -17
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +1 -1
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +195 -7
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/timeout-auth-and-capabilities.md +3 -3
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +13 -5
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +9 -3
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh +9 -3
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/normalize_review_timeout.sh +22 -0
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/opencode_review.sh +9 -3
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +1540 -129
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +8 -3
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +76 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +1858 -3
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_update_review_plan_intent.sh +789 -0
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/update_review_plan_intent.py +513 -0
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +4 -1
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -1
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +11 -10
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +64 -0
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/agents/openai.yaml +4 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/async-lifecycle-and-performance.md +72 -0
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/runtime-and-project-contract.md +58 -0
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/source-map.md +41 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/references/verification-diagnostics-and-security.md +63 -0
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +8 -10
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-routing-and-readiness.md +10 -14
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/verify-developer-experience.md +1 -1
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/SKILL.md +135 -86
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/behavioral-aesthetic-logic.md +66 -80
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/delivery-contract.md +275 -0
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-execution-checklist.md +88 -214
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-impl-naming-and-versioning.md +2 -2
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-intake-and-acceptance.md +10 -8
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +4 -5
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md +112 -95
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/frontend-code-evidence-map.md +30 -21
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/interaction-design-patterns.md +22 -3
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/layout-recipes-and-screenshot-acceptance.md +20 -17
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-project-token-consistency.md +7 -9
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-stack-strategy.md +14 -10
  44. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/operational-processing-workflows.md +2 -0
  45. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +1 -1
  46. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-lifecycle-acceptance-and-iteration.md +9 -6
  47. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-surface-patterns.md +3 -0
  48. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/source-map.md +37 -10
  49. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +7 -1
  50. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +8 -5
  51. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-design-development.md +16 -5
  52. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/visual-craft.md +4 -2
  53. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +4 -1
  54. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +4 -4
  55. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +95 -5
  56. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +11 -9
  57. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/r0-leakage-audit.md +102 -0
  58. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +54 -0
  59. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +8 -0
  60. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +6 -6
  61. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +4 -3
  62. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +69 -2
  63. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/extraction_review_gate.sh +22 -0
  64. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +49 -4
  65. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/obligation-ledger.py +2748 -0
  66. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +20 -5
  67. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/shared_git_surface_gate.py +1142 -0
  68. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +17 -0
  69. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh +41 -4
  70. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ci_checkout_ref_binding.sh +120 -0
  71. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_domain_scan_terms.sh +82 -8
  72. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_extraction_review_gate.sh +336 -0
  73. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh +82 -10
  74. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh +1416 -0
  75. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_obligation_ledger_repo_audit.sh +57 -0
  76. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +141 -4
  77. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_pointer_integrity.sh +3 -1
  78. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh +1696 -0
  79. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_delivery_contract.sh +2117 -0
  80. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_uiux_loading_budget.sh +316 -0
  81. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +1176 -0
  82. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_cross_refs.sh +31 -1
  83. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate-skill.sh +9 -4
  84. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +980 -0
  85. package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +8 -6
  86. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +8 -7
  87. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/client-runtime-test-matrices.md +10 -2
  88. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +6 -5
  89. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/complex-workspace-patterns.md +1 -1
  90. package/dist/assets/release.json +175 -70
  91. package/package.json +1 -1
@@ -358,3 +358,57 @@ and inverted the sense (production, not product), and the coordinator now shares
358
358
  | 发布工程技能有推进闸、回滚契约与审计日志,却没有发布过程自身的度量反馈环——DORA 词汇全仓三个平台技能零命中;feature flag 只被当成 dynamic config 的取值实例,缺生命周期纪律(deploy≠release 解耦、四分类、transient flag 的 owner+expiry、双侧测试、并发活跃数压低) | `platform-release-engineering` | references/promotion-gate-and-review.md 新增 Release-process metrics (DORA) 节(入口 SKILL.md 为历史超限零增长预算,正文落 reference、References 行做字数中性指名):从控制平面自身 audit log 机算 DORA 当前五因子(change lead time / deployment frequency / failed deployment recovery time〔MTTR 后继〕/ change fail rate / deployment rework rate),用作过程反馈、never 作个人或团队绩效评分(评分即污染信号:deploy 改标签、rollback 改叫 roll-forward)、算不出即 audit-log 缺口须修事件采集不得估算;references/secret-and-config-management.md dynamic 层补 flag 生命周期(Fowler/Hodgson 四分类按寿命分治、transient flag 创建即带 owner+expiry、过期即债务须浮出、双侧测试与 flag 组合空间不可测故压低并发活跃数、release toggle 的退休进 feature 的 definition of done); result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/platform-release-engineering/references/promotion-gate-and-review.md#never as individual or team performance scores | updated | owner key `platform-release-engineering/SKILL.md`。**RED-baseline 为覆盖差分 + 一手源核验**:改前 grep -riE "DORA\|deployment frequency\|lead time\|change fail" 于三个平台技能零命中(红),改后 R14 命中(绿)。DORA 一手源 dora.dev/guides/dora-metrics-four-keys 本轮 fetch:五因子引文逐条取得,含 MTTR→Failed Deployment Recovery Time 的演进说明与 deployment rework rate——按当前模型落地而非记忆里的旧四键。flag 一手源 martinfowler.com/articles/feature-toggles.html(Hodgson):Release/Experiment/Ops/Permissioning 四分类、carrying cost/inventory、expiration date for short-lived toggles 引文逐条取得。同轮顺带一手源复核未改动的既有主张:R13 gitlab-runner SIGQUIT/SIGTERM 语义与 killall/pkill 告诫和 docs.gitlab.com/runner/commands 逐字吻合(unchanged: already-covered,无观察失败)。deferred(记录不落地):SLSA/构建 provenance attestation 与镜像签名准入是 R2/deploy-pipeline 的候选补强——deploy-pipeline 已有 OpenGitOps/OCI/cosign-manifest 基底,image-build attestation 面待下轮按外部源核后落。 |
359
359
  | 交付面只核「那一份变了没」核不出「放没放对地方」——挂错目录 / 空间 / 父容器的文档,内容标志核得再准也是错交付;且放置决策若依据过期快照或链接标题猜测,错位在发布前就已注定 | `tighten-doc` | delivery-face 新增 ①b 定位节(载体无关,适用任何多级容器承接面):放置前必须现读目标结构、不按链接标题猜父容器;落点选最具体的稳定容器、标题相似本身不构成父子依据(现读确认确为稳定容器的节点可作父节点——页面即容器的承接面适用,该限定由独立评审指出过宽后收窄);新建/移动后回读父容器/空间/完整路径并交付全路径;落点决定可见性——放置前记预期受众/访问边界、放置后回读生效受众/权限与 owner 并按方向分流:比预期更宽(已实际暴露)、无法排除更宽、或 owner 落到非预期主体(owner 自带控制与转授权能力)的按待安全处置转出安全/内容 owner 走收敛路径结案(closeout 触发枚举同步纳入、与「拿不准按命中」同则;预期 owner 随预期边界在放置前记录)、确知更窄(可修复交付缺陷)或边界符合但其他核验失败的按 blocked 修复后重投,两者均禁报已同步,且本节只核不改(结构移动与权限变更各需其自身授权)——该第四条由 review 与 challenge 两 lane 对 ACL 继承暴露面独立收敛后补,过宽/过窄分流由后续评审指出同判 blocked 会让真实泄露留在普通投放流后再收窄。SKILL.md 交付面 bullet 加最小 cue(①句「投放与定位都要回读远端」+ 指针「四态定义、定位判据」,后者同时修正此前登记的三态→四态指针 P2); result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/tighten-doc/references/delivery-face-closeout.md#放置前必须现读目标结构**:目录树 / 容器层级以本次实时读取为准 | `updated` | owner key `tighten-doc/SKILL.md`(轮起 base 53adde21,随并行轮 rebase 至当前 dev;派生与零损核对均以当前 origin/dev 为 base 复算)。源=对同事分享的中文技术写作技能包做 R0 复扫后剥出的载体无关方法形状(源包示例所在的具体场景域整体拒绝、仅抽象记录——域类别本身也不在共享树点名,名单在 per-host 私档;私有 denylist 审计 alias_audit_ok 由维护者侧持有、包内不可独立复核——名单本身就是需保密的标识符集合,随包公开名单即构成泄漏,故该不可复核性是脱敏设计的结构性边界而非可修缺口;包内可核的是 generic 扫描 token 与评审对全文的直接检视,此分级如实声明);三条规则不作业界 state-of-the-art 认领(generic framing,guidance stands alone)。**RED-baseline 为 headless 差分 applied differential**(claude-sonnet-5,n=3/臂,判分维度与通过线**各自冻结于其对应重跑之前**——初版三维冻结先于任何编辑,评审驱动的后增维(含两个安全维)各冻结先于其首跑、历次仪器修订全披露且时序锚于平台侧 SHA-keyed CI run 史(证据分级见包内 grading),臂快照与最终候选逐字节 diff 校验,隔离 cwd,污染 grep 零命中;判分语义为准、token grep 仅初筛代理,两例代理误报已改判并标注):行为学主张口径=规则可及性/引出差分(按清单作答能否产出义务;真实执行场景的决策行为未测、如实声明),唯一取自终态臂计分批(raw 全入包、臂逐字节校验;判分语义为准且覆盖全答;仪器修订全披露、定义冻结于对应重跑前):十一个差分维终批均 old 0/3 → new 3/3(D-verifyonly 曾有一批 old 读数 1/3——经既有删除条款类推达成,跨批漂移如实留痕于包内存档表)、C-ctl 两臂 3/3(后六维为评审指出安全关键条款需维覆盖后按冻结-先于-重跑纪律逐轮新增;D-acl-pre 初版误限 A 段判 1/3,评审对 raw 复核后按冻结全答判据改判,勘误留痕);D-acl-over 不作差分主张——校准表述:终态计分批 old 臂三跑均答清单未规定(两跑附条件性说明);此前一计分批曾观测 old 臂一跑经既有「拿不准按命中」兜底直接到达同一处置(观测保留于包内 grading 存档表与作者 charter)——该维对推理深度敏感且跨批不稳,故不以差分立论。该子句保留依据=双 lane 收敛暴露面 + closeout 触发枚举一致性修复(过宽暴露此前不在第 4 条枚举、操作者按编号流程进不了该状态,评审指出后补入并区分为收敛路径)。无角色存档 raw 按 convergence-by-deletion 移出树(三轮包内核验发现的源头;原因表与 charter 记录保留,ctl-invalid 组 raw 因自证保留)。**过程如实披露**:存档批(原因表保留、raw 除自证组外移出)(v1prompt-old、误吞②节标题的 arm-stale——由 governing-chain-diff 派生当场抓到并修复、v1prompt ctl-invalid、ACL 前 arm-stale、v1prompt-acl ctl-invalid、规则#2 收窄前的 old/new 两组)均不承载任何主张或依据,原因表在包内 grading、raw 除自证组外已按 convergence-by-deletion 移出树;终态派生 SKILL 5 行、reference 8 行(6 条改写行——含授权不投放行增「可回读授权来源」要件而原义务逐字保留——+ 2 条仅因父行改写而链变、自身文本逐字未动的子句;派生输出入包 chain-derivation.txt,base=当前 origin/dev)——closeout blocked/待安全处置行因新增触发就地改写,其既有删除类触发枚举逐字保留可 grep 复核,新增只扩触发集合不弱化既有强度,新 ①b 节为纯新增。候选绑定证据 `eval/evidence/placement-face-2026-08-28/`(REPLAY 零损对照含实测输出、臂快照、prompt 生成器、raw 12 份(计分 6 + ctl-invalid 自证 3——每个无效批保留省答那一跑 + 批 23 计分失败自证 3;判定不依赖的存档 raw 均按 convergence-by-deletion 移出树)、逐跑判分、SHA256SUMS、AGENTS 契约),除已按证据类分级如实声明者(私有 denylist 审计=维护者侧持有、冻结时序=平台侧 CI run 锚 + 包内部分自证,见包内 grading)外均包内可独立复算。 |
360
360
  | 入口字数预算下,SKILL 交付面 bullet 的新增 cue 以收缩三处纯示例括注抵消——括注载体逐字/语义保留于 reference,义务本体句中未动 | `tighten-doc` | SKILL.md 交付面 bullet 三处括注收缩(面枚举括注、被要求删除触发枚举括注、指针三态→四态与加「定位判据」),义务零损;reference 侧为 ①b 新增 + closeout 触发扩展的就地改写(既有触发逐字保留); result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/tighten-doc/SKILL.md#逐面判定,未判定不得报「已更新 / 已同步」:**①投放与定位都要回读远端 | `updated` | owner key `tighten-doc/SKILL.md`。零损对照表(每条被收缩片段 → reference 现行载体原文 → 复算命令与实测计数)在 `eval/evidence/placement-face-2026-08-28/REPLAY.md` §1;row set 由 governing-chain-diff 机械派生(SKILL 5 行=同一 bullet 改写行,reference 8 行=closeout 要件/触发扩展的 6 条就地改写行加 2 条仅因父行改写而链变、自身文本逐字未动的子句(派生输出入包 chain-derivation.txt,base=当前 origin/dev),既有触发逐字保留 grep 可核,①b 节纯新增);入口预算复测 entrypoint_word_budget_legacy_ok(9826≤9828)+ entrypoint_size_blocking_ok。保全清单三源走查:台账 firing-path 解析进本包锚点未触碰;校验脚本钉扎两遍扫无命中本轮改动短语;「三态定义」等旧短语的仓内残留仅在 062 冻结证据包(其 AGENTS 契约声明 concluded/candidate-bound,REPLAY §3 记交叉说明)。 |
361
+ | A Node.js implementation owner must resolve the repository's live runtime/module contract before code, keep asynchronous work bounded and cancellable, treat built-in TypeScript execution as distinct from type checking, and route architecture, diagnosis, test-layer policy, observability, connectivity, release, and terminal contracts to their existing owners | `nodejs-service-dev` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/nodejs-service-dev/SKILL.md#Do not swallow rejections or resume normal operation; result-class: insufficient-evidence; bank-evidence: file:eval/routing-tasks.jsonl#route-nodejs-service-impl | updated | Owner key: `nodejs-service-dev/SKILL.md`. Initial RED: the routing-bank integrity gate rejected four fixtures because `nodejs-service-dev` did not exist. Candidate adds the owner, three task references, maintainer source map, catalog/README registration, and positive plus high-overlap negative routing fixtures. Primary technical claims were checked against live Node.js/npm documentation; OWASP/OpenSSF and two public skill corpora were used only to challenge coverage and skill shape. Deterministic GREEN and repository gates prove registration/conformance, not production behavior improvement; runtime-version and framework-specific claims remain live-check obligations. Sibling disposition: `testing-strategy`, `defect-diagnosis`, `product-rd-workflow`, `terminal-cli-dev`, `platform-observability`, `platform-service-connectivity`, `platform-release-engineering`, `web-react-dev`, and `llm-inference-integration` remain unchanged because the new skill routes to their existing contracts rather than copying them. |
362
+ | A catalog regression fixture that copies candidate catalog text must snapshot the candidate skill roots too; cloning committed HEAD while a new skill exists only in the working tree creates a false pristine mismatch and cannot validate the tree being landed | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh; result-class: failure; bank-evidence: file:eval/routing-tasks.jsonl#route-nodejs-service-impl | updated | `make test` RED reproduced c6 as `extra_in_catalog=nodejs-service-dev`: the fixture cloned 32 committed skill roots but copied the 33-entry candidate catalog. `new_case` now replaces the temporary clone's `skills/` with the candidate working-tree snapshot before mutation; the full 12-case catalog suite turns GREEN. The gate assertions and failure tokens are unchanged, and all destructive fixture operations remain confined to the validated `mktemp` clone. |
363
+ | Trigger accuracy and body effectiveness are different claims for a new stack skill: route-bank examples can show owner selection while still proving nothing about the quality of the implementation advice, so both surfaces need frozen tasks and explicit failure criteria | `nodejs-service-dev` | behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/nodejs-service-dev/SKILL.md#Do not swallow rejections or resume normal operation; result-class: insufficient-evidence; bank-evidence: file:eval/routing-tasks.jsonl#route-nodejs-test-strategy | updated | Owner key: `nodejs-service-dev/SKILL.md`. User correction required an explicit effectiveness surface after the initial candidate had only routing and repository conformance evidence. Four advisory human-judgment fixtures now cover runtime/module/TypeScript contracts, streaming cancellation/backpressure/shutdown, bounded CPU workers, and the testing-strategy-to-Node-mechanics handoff. They freeze prompts, rubrics, weak-answer criteria, and owner pointers; they do not claim a measured with/without improvement until repeated paired runs and transcript/outcome review exist. The shape follows a local skill repository's useful separation of full-catalog routing trials from blinded with/without body evaluation, while its product-specific names, infrastructure, stored outputs, and version policy remain out of this shared tree. |
364
+ | A candidate-tree regression fixture must snapshot the candidate task banks as well as skill roots: a new source-register row can legitimately point at an uncommitted routing or behavior fixture, and pairing that row with the committed eval tree makes the supposed pristine arm internally impossible | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh; result-class: failure; bank-evidence: file:eval/behavior-fixtures.jsonl#F31 | updated | Final `make test` reproduced c6 after the Node.js effect fixtures were added: the temporary clone received the candidate skills/source-register but retained HEAD's older eval banks. `new_case` now snapshots both `routing-tasks.jsonl` and `behavior-fixtures.jsonl`; c13 constructs candidate-only markers under `TEST_ROOT` so deleting either copy turns the regression suite RED even after today's fixtures are committed. The scope remains the two task banks consumed by shared-skill validation; no private outputs or unrelated eval artifacts are copied. |
365
+ | Before-review implementer closure for the new Node.js owner and its candidate-snapshot regression repair: acceptance is a concrete Node.js implementation owner with evidence-bounded runtime, async lifecycle, security, verification, testing-owner handoff, routing positives/negatives, and effect-evaluation fixtures, while preserving existing owner boundaries and making the catalog fixture validate the actual candidate tree | `nodejs-service-dev` | behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/nodejs-service-dev/SKILL.md#Do not swallow rejections or resume normal operation; result-class: stable-success; bank-evidence: file:eval/routing-tasks.jsonl#route-nodejs-test-strategy | updated | Owner key: `nodejs-service-dev/SKILL.md`. **Changed-file scope, frozen before independent review/challenge:** `README.md`; `docs/SKILLS.md`; `eval/behavior-fixtures.jsonl`; `eval/routing-tasks.jsonl`; `packages/ccl-skills-npm/README.md`; `skills/nodejs-service-dev/SKILL.md`; `skills/nodejs-service-dev/agents/openai.yaml`; its four `references/*.md`; and this source register. **Load-bearing invariants and failure paths:** architecture/diagnosis/test-policy/CLI/observability/connectivity/release requests must redirect instead of being swallowed; Node test mechanics start only after `testing-strategy` chooses layers and coverage; version-sensitive claims require live repository/runtime or primary-doc evidence; TypeScript execution never substitutes for type checking; cancellation, backpressure, worker termination, fatal shutdown, dependency freezing, and secret handling stay explicit rather than happy-path examples; candidate regression clones must snapshot skills and both referenced task banks before mutations, remain confined to `mktemp`, and fail if either snapshot is omitted. **Self-check evidence:** the four Node routing rows and four F31-F34 body fixtures cover positive and high-overlap negative boundaries; catalog c6/c13 reproduced both snapshot defects before turning green; final `make test`, heavy regression lane, public sanitization, focused routing/fixture/catalog checks, and `git diff --check` were green on the then-current candidate. **Residual risks:** the body fixtures establish a falsifiable evaluation surface but no repeated blinded with/without trial has yet measured an effect delta; Node/runtime and ecosystem facts can drift and remain live-check obligations; framework-specific build conventions stay repository-local rather than being guessed into the shared owner. Dual-track classification: `dual-track-review-gate.md` first table row, non-wording new shared skill, so independent fact/consistency review plus adversarial challenge are both required; this row is the pre-review ordering record, not independent proof. |
366
+ | A self-review record written into the candidate solely to prove review readiness changes the packet it is meant to freeze, needlessly invalidates tests and review identity, and can recurse when outcome rows are added; ordering evidence and candidate evidence need separate lifecycles | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/extraction-quickstart.md#Run deterministic checks and implementer self-review first; result-class: failure | updated | The preceding Node.js closure row exposed the defect: adding it after a green exact-candidate run made the candidate dirty again and started another full suite even though no implementation, routing, runtime rule, or validator behavior had changed. That run was interrupted rather than credited. The quickstart now requires a fresh non-overwritten task-evidence path outside the candidate, passed verbatim as `--review-plan-file`; the gate result binds its profile hash and preserves the review ordering. Candidate-local rows remain valid only when the row is itself a substantive deliverable under review. Changing only external self-review evidence refreshes profile binding but does not invalidate implementation tests or candidate packet identity. The prior row remains append-only history and is superseded only for its storage mechanism; its substantive Node.js self-review claims are recopied into the external plan for this review. |
367
+ | Candidate skill-root snapshotting needs its own permanent candidate-only marker: relying on today's uncommitted new skill makes the regression turn inert as soon as that skill enters HEAD, so deleting the snapshot later can stay green until the next working-tree-only skill exposes the old false-pristine mismatch | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh; result-class: failure; bank-evidence: file:eval/routing-tasks.jsonl#route-nodejs-service-impl | updated | The adversarial challenge found that c13 permanently mutated both task banks but did not create a candidate-only skill root. `new_case` now accepts separate candidate skill and eval roots; c13 builds all three marker surfaces under `TEST_ROOT`, passes them explicitly, and asserts the skill marker reached the case clone. Removing the skill-root copy therefore turns c13 RED independently of whether `nodejs-service-dev` is already committed. Default arguments preserve every existing case, and all writes/destructive cleanup remain inside the validated temporary clone or `TEST_ROOT`. |
368
+ | A Bash test label containing Markdown backticks must be passed as a literal; otherwise command substitution can emit stderr and erase label text while the suite still exits zero | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh; result-class: failure | updated | RED reproduced the existing A13 false green: suite rc=0, the literal `` `-` `` was absent from stdout, and stderr reported `-: command not found`. The durable A13 report oracle then made an applied restoration of the old double-quoted call fail with suite rc=1 and `A13: RED (summary lost literal `-`)`; the final single-quoted call is GREEN with rc=0, the complete A13 summary present, and empty stderr. Gate verdict and fixture semantics are unchanged. |
369
+ | A closure row with one owner-scoped firing path must not combine multiple owners: the Node rule anchor can adjudicate the Node owner, while workflow verifier changes retain their own rows and executable firing paths | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh; result-class: failure; bank-evidence: file:eval/behavior-fixtures.jsonl#F31 | updated | Owner key: `skill-extraction-workflow/SKILL.md`. The explicit-base repository gate rejected the combined closure row for `skill-extraction-workflow` because its only firing path resolved inside `nodejs-service-dev`. The closure row now binds only the Node owner; the candidate-snapshot workflow repairs remain covered by their dedicated RED-baseline rows and the changed catalog regression executable. No owner requirement or verifier result was downscoped. |
370
+ | A new implementation decision owner must be registered in the impact-chain owner set before its description-level bank evidence can be resolved; otherwise the gate can demand evidence from an owner while making that owner's row unreachable | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh; result-class: failure | `updated` | Owner key: `skill-extraction-workflow/SKILL.md`. A20 changes the fixture's committed `nodejs-service-dev` description, supplies an owner-scoped firing path and external bank-evidence locator, and reproduced `impact_chain_bank_evidence_missing` before registration. Adding this implementation owner to the existing curated set makes A20 GREEN without changing generic leaf resolution or curated-set semantics for other skills. |
371
+ | A new stack implementation owner joining an already-swept class must inherit the class-wide obligations its siblings carry — the language-stack CLI carve-out with reciprocal terminal-cli-dev skip legs, and the localized-refactor trigger pair — or requests in that language route to a skill that never claims the work | `nodejs-service-dev` | behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/nodejs-service-dev/SKILL.md#language-stack CLI implementation must not be routed back; result-class: stable-success; bank-evidence: file:eval/routing-tasks.jsonl#route-nodejs-cli-not-terminal | updated | Owner key: `nodejs-service-dev/SKILL.md`. Independent review of the initial candidate flagged the asymmetry: python-service-dev and go-microservice-dev both advertise standalone CLI/tooling (with reciprocal terminal-cli-dev skip legs) and localized-refactor triggers, while the new Node owner claimed neither. RED-baseline is a coverage differential: before this change, CLI/命令行工具 and 重构/refactor wording had zero hits in the nodejs-service-dev description and routing rules; after it, the CLI description trigger, the owner-side routing rule, the reciprocal skip leg, two pointer-integrity anchors, the route-nodejs-cli-not-terminal fixture, and the localized-refactor trigger pair (重构 Node 服务里的某文件/某类(局部) / refactor a file/class within a Node.js service) are all present, mirroring the proven Python/Go legs. Per-obligation disposition for the remaining class obligations: defect-diagnosis-first, multi-stage→product-rd, and testing-strategy handoffs were already present in the initial candidate (`unchanged`); the bare service-wide refactor trigger is `not-applicable` — nodejs has no `*-architecture` sibling, and service redesigns already route to product-rd-workflow in the routing rules. The interface-contract split is preserved: the command/flag/help/exit/TTY contract stays with terminal-cli-dev. The routing fixture remains a frozen advisory input under the same evaluation boundary as the other Node fixtures in this round. |
372
+ | Extending a named skip-leg list is a routing-surface change on the skip-side owner too: the leg is only real when both sides land together, and the eval bank must assert the skip-side owner does not receive the redirected request | `terminal-cli-dev` | behavioral-evidence: semantic-control; observed-failure: no; bank-evidence: file:eval/routing-tasks.jsonl#route-nodejs-cli-not-terminal; result-class: stable-success | updated | Owner key: `terminal-cli-dev/SKILL.md`. Description-only change: the existing skip clause gains the `Node.js CLI → nodejs-service-dev` leg beside the Python and Go legs; the generic predicate ("a language whose dev skill owns it") and the no-owner fallback are unchanged, and the non-rendered command/flag/help contract stays owned here. Semantic-control: the change instantiates an already-pinned rule class for one more stack rather than introducing a new behavior claim — the Python and Go legs carry the RED history, and this leg reuses their oracle shape (reciprocal trigger anchor, skip-leg anchor, and a must-not-route fixture asserting terminal-cli-dev does not receive the Node CLI request). |
373
+ | A sibling-generalization map that only checks whether the new member copies sibling content misses the join-side dual of class-wide coverage: the obligations the class already landed on its members (triggers, skip-leg reciprocity, pinned anchors, fixtures) silently fail to transfer to the newcomer | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/source-to-skill-extraction.md#must not be counted as obligation-inheritance coverage; result-class: failure | updated | Owner key: `skill-extraction-workflow/SKILL.md`. Observed failure: the initial nodejs-service-dev candidate shipped with a sibling map recording "siblings unchanged because the new skill routes to their existing contracts", yet the new member lacked two class obligations its siblings carry (the CLI carve-out reciprocity and the localized-refactor trigger pair) — the map answered the copy-content question and never asked the obligation-inheritance question, and only independent review caught it. RED-baseline: before this change the class-wide COMPLETE-set rule fired only on change-side sweeps ("every member of class C should carry X"), with no clause firing on member-join; the new Member-Join Inheritance section makes the join-side enumeration an explicit obligation with a per-obligation disposition requirement, and the SKILL.md class-wide bullet gains a word-neutral member-JOIN pointer (two pure example parentheticals moved verbatim into the section to offset the pointer under the entrypoint's zero-growth budget). |
374
+ | A created-routing-surface obligation must be bounded to entrypoints absent at the round's base: an existing non-curated skill editing its description otherwise incurs bank evidence that no ledger row can bind, because row-to-owner resolution spans only the curated name set — the gate demands evidence while making it unreachable | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/impact-chain-gate.rb; result-class: failure | updated | Owner key: `skill-extraction-workflow/SKILL.md`. Observed: CI repository-gates went red on this branch with impact_chain_bank_evidence_missing owing the skip-side owner after the skip-leg round touched an existing non-curated description; the local run on the committed tree reproduced it, and a probe showed the owner's row set empty because name-level resolution iterates only resolvable curated names, so the appended row could never discharge the obligation. Fix bounds the created-surface pickup with an entrypoint-absent-at-base check via the existing regular-blob reader, exactly matching the block's stated brand-new-skill intent; the forcing function for genuinely new skills is preserved, since a base-absent entrypoint still joins the triggered set and still requires curation before its evidence resolves. New suite leg A21 holds an existing non-curated description edit green and turns red if the existence bound is removed. The skip-side owner's ledger row from the skip-leg round remains as unbound documentation by design. |
375
+ | Runtime-visible design work uses one context- and criterion-driven lifecycle rather than separate design checkpoints: design brief → test Phase 0 → producer/client execution records → test Phase 1/sufficiency → candidate-bound design verdict; each claim closes every required, non-substitutable evidence dimension, and heuristic review is risk discovery rather than acceptance proof | `product-ui-ux-design` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-ui-ux-design/SKILL.md#Automated accessibility checks do not replace | `updated` | Owner key `product-ui-ux-design/SKILL.md`. The pre-change focused contract test failed on the missing canonical record and reciprocal owners, four arbitrary five-second thresholds, and four eponym-only extraction anchors; later challenge also found duplicated client/test evidence ownership, reversed affected/unknown consumer status, and an over-broad rejected-surface rule. The owner entrypoint was compressed, `delivery-contract.md` added, the execution checklist converted to a profile router, theory rebuilt as an authority-classed claim ledger, source/code maps updated, and reader docs synchronized. Original-source classes include ISO human-centred/usability framing, W3C normative versus informative material, primary empirical papers with population/task limits, platform-scoped guidance, and the non-standard Design Tokens Community Group report. Current client-code extraction contributes only source-neutral static-source candidates, each limited to the implementation/test/script/CI subset actually observed; it makes no test-execution, render, runtime, or product-effect claim. The older platform-walkthrough model is explicitly superseded; its three immutable historical locators are row-digest-bound in `register-firing-path-resolution.rb`, not silently restored. Program record: [065 UI/UX evidence delivery](../../../specs/065-uiux-evidence-delivery/plan.md). |
376
+ | UI/UX testing no longer waits for a finalized design checkpoint that itself waits for test-layer selection: Phase 0 fills assertion/rendered layers and oracles from a testable design brief, while post-producer/client Phase 1 cites the complete design/test/producer/client record set, adds only test-owned evidence, and owns criterion results plus sufficiency without claiming the holistic design verdict | `testing-strategy` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/testing-strategy/SKILL.md#For every runtime-visible UI/UX slice, load the canonical sequence | `updated` | Owner key `testing-strategy/SKILL.md`. Baseline inspection found the circular checkpoint wording and no schema-level reciprocal contract; cross-owner challenge later found Phase 1 and the client owner duplicating the same evidence. The current sequence is Design brief → Phase 0 → producer/client execution → Phase 1/sufficiency → design verdict, and every design, test, producer, and client role writes its own facts once. A current bounded Design brief also satisfies testing's scope gate instead of forcing the user to restate it. The focused contract test requires both testing passes, their order, the shared-record handoff, single-write rule and bounded-scope reuse. Program record: [065 UI/UX evidence delivery](../../../specs/065-uiux-evidence-delivery/plan.md); [replayable validation](../../../specs/065-uiux-evidence-delivery/validation-evidence.md). |
377
+ | React/browser implementation consumes the shared design brief and Phase 0, writes candidate-bound Web runtime facts once into the shared client record, and lets testing Phase 1 own criterion interpretation/sufficiency before the design verdict | `web-react-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/web-react-dev/SKILL.md#For every visible UI change, load | `updated` | Owner key `web-react-dev/SKILL.md`. Baseline had one-way checkpoint consumption but no equally explicit closeout; a later challenge found direct-to-design return could skip testing sufficiency. The current Web return names route/server, viewport/container, artifacts, tested states/input, raw criterion-mapped observations, console/network observations, coverage and gaps exactly once in the shared record. Browser render remains scoped evidence rather than self-acceptance. |
378
+ | Cross-platform/native App implementation keeps its device evidence obligations, writes candidate-bound runtime facts once into the shared record, and leaves criterion sufficiency/verdict to testing/design owners instead of duplicating their state rules | `app-cross-platform-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/app-cross-platform-dev/SKILL.md#For every visible UI change, load | `updated` | Owner key `app-cross-platform-dev/SKILL.md`. Baseline App guidance already had strong post-render obligations, but cross-owner challenge showed its direct-to-design return could duplicate or bypass testing sufficiency. This round preserves device/form-factor, safe-area, keyboard, orientation, text-scale, lifecycle and artifact evidence, removes duplicated checkpoint/verdict schema, and routes the single client record through testing Phase 1 before verdict. |
379
+ | Mini-program implementation consumes the same design/Phase 0 record, translates it into each shipped host, and writes host/tool/device, capability/permission, state/dimension, artifact, coverage-boundary and gap facts once; browser/H5 preview cannot stand in for a shipped host | `miniapp-product-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/miniapp-product-dev/SKILL.md#For every visible UI change, load | `updated` | Owner key `miniapp-product-dev/SKILL.md`. Baseline repeated the design gate but lacked an equal closeout. The owner now keeps host-specific mechanics and runtime gates locally, writes raw criterion-mapped observations to the shared client record, and leaves sufficiency/verdict to testing/design owners. |
380
+ | Every user-facing terminal/CLI contract—including ordinary plain-text command trees, flags/defaults, help/output/exit behavior, confirmations, progress, and recovery—is a design surface: it consumes the shared design/Phase 0 record and writes the applicable command or PTY/runtime, dimension, fallback, interaction, artifact, coverage-boundary, and gap facts once | `terminal-cli-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; bank-evidence: file:specs/065-uiux-evidence-delivery/validation-evidence.md#基线与候选各完成 10 轮有效观测; firing-path: file:skills/terminal-cli-dev/SKILL.md#For every user-facing terminal/CLI contract change | `updated` | Owner key `terminal-cli-dev/SKILL.md`. Baseline had a one-way page-slice/checkpoint rule but no equivalent closeout. The owner now translates the shared contract to ordinary CLI semantics or cell-grid and terminal-lifecycle evidence as applicable, writes raw observations once, and leaves Phase 1 sufficiency and the design verdict to their owners. |
381
+ | Product readiness distinguishes parser/library-only CLI internals from every user-facing command/help/output/exit/confirmation/progress/recovery contract, and sequences one canonical design brief → Phase 0 → producer/client records → Phase 1/sufficiency → verdict record rather than maintaining a second checkpoint schema | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#Before coding any visible UI change, load | `updated` | Owner key `product-rd-workflow/SKILL.md` routes Web/App/mini-program/desktop/terminal visible delivery and canonical status vocabulary to the shared contract, preserves the full `visible surface: no` boundary, and permits only a labeled review-only draft MR to obtain a required independent design verdict. `design-routing-and-readiness.md` owns product sequencing, not a second evidence schema. Program record: [065 UI/UX evidence delivery](../../../specs/065-uiux-evidence-delivery/plan.md). |
382
+ | UI/UX extraction turns named theories and local source patterns into bounded observation → risk/mechanism → hypothesis → observable check → evidence-boundary records, and a deterministic regression gate protects the canonical design/testing/producer/client loop and retired folklore thresholds | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_uiux_delivery_contract.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` is unchanged. `references/uiux-judgment-extraction.md` removes eponym-only instructions; the new executable test is registered in the fast regression lane and checks reciprocal owners, five top-level stages including both testing passes, verdict binding, profiles, evidence boundaries, primary-source ledger entries, retired five-second rules and resolvable references. The test was RED on the pre-change corpus and is GREEN on the current candidate. The firing-path resolver gains three explicit row-digest-bound waivers for the superseded platform-walkthrough locators so append-only history stays intact without pretending retired criteria still execute. |
383
+ | A Go service that emits or stores possibly client-rendered text closes only after authoritative consumer-universe classification; a known rendering client is `affected`, while only an incomplete/inaccessible universe or member is `unknown-consumers` | `go-microservice-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/go-microservice-dev/SKILL.md#create the applicable full or lightweight record in | `updated` | Owner key `go-microservice-dev/SKILL.md`. Cross-skill audit found a retired checkpoint alias and no canonical pointer; later challenge found the first rewrite incorrectly classified known visible consumers as unknown. The current rule proves an authoritative universe, records each member, routes `affected` clients into the full/lightweight contract, reserves `unknown-consumers` for incomplete/inaccessible evidence, and permits backend-only closure only when the complete inventory proves no client rendering. |
384
+ | A Python service that emits or stores possibly client-rendered text uses the same authoritative consumer-universe classification and canonical design/test/producer/client handoff: known rendering clients are `affected`; incomplete/inaccessible evidence is `unknown-consumers` | `python-service-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/python-service-dev/SKILL.md#create the applicable full or lightweight record in | `updated` | Owner key `python-service-dev/SKILL.md`. The baseline allowed a current-repo search to stand in for a complete inventory; the first rewrite also conflated affected and unknown. The current rule requires an authoritative universe plus per-member disposition before backend-only closure and routes affected/unknown states without turning known UI work into a false blocker. |
385
+ | An inference path that emits or persists possibly client-rendered copy, labels or generated status uses the canonical full/lightweight record when a known client is `affected`, reserves `unknown-consumers` for incomplete/inaccessible evidence, and cannot accept a surface from inference-side evidence alone | `llm-inference-integration` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/llm-inference-integration/SKILL.md#you must load `../product-ui-ux-design/references/delivery-contract.md` | `updated` | Owner key `llm-inference-integration/SKILL.md`. The baseline allowed a bounded search over an author-chosen subset; the first rewrite conflated affected and unknown. The current rule requires an authoritative universe plus per-member disposition, routes known affected clients to their owner/testing stages, and leaves incomplete/inaccessible consumers blocked as unknown. |
386
+ | UI/UX contract and cross-owner RED-baseline claims remain independently replayable instead of living only in narrative source-register rows | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_uiux_delivery_contract.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` is unchanged. [Replayable validation evidence](../../../specs/065-uiux-evidence-delivery/validation-evidence.md) binds the unchanged `origin/dev` corpus, candidate oracle, controlled RED exit, current GREEN result, built-in five-stage, entry-router, trigger-loss, role-member-drop, dirty-as-commit, mutable-external, owner-narrowing, ordinary-CLI, negated-return, affected-vs-unknown, and user-accepted-gap mutations, historical-locator waiver mutations, commands and evidence limits. The per-obligation carrier proof is kept separately in [obligation preservation](../../../specs/065-uiux-evidence-delivery/obligation-preservation.md). |
387
+ | A generic long-digit leak guard must distinguish opaque identifiers from canonical public identifiers at the matched URL span; exempting a whole line hides a second identifier, while rejecting every long digit blocks primary-source citations | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_entrypoint_domain_scan_terms.sh | `updated` | The quick gate rejected three canonical DOI/W3C report URLs after the theory ledger restored original-source links. The checker now keeps the lexical domain scan unchanged and scans each long digit run separately. Only a run wholly inside a canonical `doi.org` or W3C Community Report URL without query/fragment is ignored; an unrelated long ID on the same line still fails. The regression covers allowed DOI/report fixtures, the same-line long-ID bypass, retained lexical terms, and end-to-end failure attribution. This is a narrow public-identifier classification, not a broad URL or line allowlist. |
388
+ | Runtime-visible delivery is a set problem, not a single-owner pipeline: producer changes can alter client state through API/event/schema/status/permission/result shape; source extraction and shared-system work compose with delivery depth; embedded content and host layers need separate owners, runtime records, and immutable binding members | `product-ui-ux-design` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-ui-ux-design/references/source-map.md#Implementation rules follow the complete affected client-owner set | `updated` | Independent cross-owner review reproduced five false-green paths: non-string producer changes bypassed the design contract; profile depth incorrectly cancelled orthogonal source/shared-system work; WebView/mini web-view/Electron collapsed content and host evidence into one owner; `base+dirty-sha256:` passed with an empty payload; and a backticked cross-skill pointer did not resolve. The contract now uses delivery-depth plus composable work-mode/risk unions, complete design/test/producer/client record and candidate-binding sets, closed non-empty binding kinds, and a dirty-bundle manifest over base, binary tracked diff, and sorted untracked path/mode/content. Go/Python/inference entries now require the applicable design record before Test Phase 0 and actual producer/client execution returns for affected consumers, or backend/product/testing handoff plus API/log/output evidence when the authoritative universe proves no client can react. Naming an owner or inventory alone is not closure. Focused mutations kill narrowed producer scope, missing pre-Phase-0 design records, name-only routing, single-choice profiles, dangling pointers, lost trigger classes, narrowed inner owner sets, dirty-as-commit or mutable-external bindings, ordinary-CLI escape, and missing design/test/producer/client members. |
389
+ | A contract oracle must parse the active contract structure it claims to protect; hard-coded test-local status sets, first-heading matches, and whole-file literals can certify extra contradictory rows, duplicate/out-of-order stages, or obligations moved into examples/comments | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_uiux_delivery_contract.sh | `updated` | Independent gate review showed the former status fixture tested its own `case` statement instead of the contract table; the stage “swap” deleted a heading rather than swapping it; duplicate headings and fenced/commented carriers survived. The focused oracle now strips fenced blocks and HTML comments, requires each active stage heading exactly once and in order, derives the exact closed verdict/state and binding-kind sets from live Markdown tables, and includes killing mutations for real swaps, duplicates, added contradictory rows/kinds, fenced/commented obligations, and fenced tables. This proves those static contract properties only; actual Agent routing/effect remains a later task-outcome measurement, not inferred from green text gates. |
390
+ | Public-identifier exceptions in a privacy scanner require identifier grammar, not merely a trusted host prefix; otherwise arbitrary long IDs hidden under a DOI/W3C-looking path bypass the rule | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_entrypoint_domain_scan_terms.sh | `updated` | Review probes showed `doi.org/not-a-doi/<opaque-id>` and `www.w3.org/community/reports/not-a-report/<opaque-id>` passed the first exception while separated IDs were blocked. The scanner now accepts only DOI paths beginning `10.<registrant>/...` and W3C Community Group `CG-FINAL-...<date>` report paths, still excluding query/fragment spans. One end-to-end fixture independently requires failures for same-line IDs, arbitrary DOI/report paths, query IDs, fragment IDs, noncanonical hosts, and IDs attached after a valid report URL; every marker must appear in the scan output so one caught class cannot mask another. |
391
+ | A shorter skill entry preserves capability only when each former task trigger still reaches its operative reference; a file that remains in the package but has no applicable inbound route is deleted behavior, not a successful rehost | `product-ui-ux-design` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-ui-ux-design/references/design-intake-and-acceptance.md#The changed producer owners, affected client owners, Test Phase 0 handoff | `updated` | Independent origin/dev-to-candidate trigger review found that the first compressed router left naming/version synchronization unreachable, limited the audit procedure to systemic redesign, and made multi-stack guidance reachable only through a token side path. A complete pass then found the same class for design-to-code primitives, narrow layout/state/screenshot work, local code-evidence audits, same-stack multi-subproject themes, and generic non-scenario surface/loop modeling. The router now chooses one delivery depth and unions every orthogonal source/code-evidence, design-to-code, audit, naming/version, same-stack, multi-stack, and risk lens whose trigger applies. Focused tests parse the live active-Markdown rows, require each old task class to resolve to its reference, and kill orphaned-pointer or narrowed-trigger mutations. The restored references were also reconciled to the canonical complete affected client-owner set and producer handoff so reactivation cannot hard-code React or mobile ownership for Vue/Svelte, Electron, terminal, desktop, or TV surfaces. This proves static reachability and owner consistency only; whether the shorter entry improves real Agent task outcomes still requires comparable task trials or longitudinal evidence. |
392
+ | Entry compression must be proved against immutable baseline obligations, not inferred from shorter files: every changed pre-existing skill obligation has one exact live carrier or a reviewed retirement, and unresolved rows block completion | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh | `updated` | The first compressed UI/UX candidate orphaned real routes, its first 544-row projection became stale after later semantic fixes, and the first parser skipped Markdown table cells. The canonical human-reviewed source is `specs/065-uiux-evidence-delivery/obligation-mapping.jsonl`; `obligation-ledger.py` derives the changed pre-existing row set from the immutable base, rejects fuzzy or provenance-only carriers, accepts a carrier bundle only when the old obligation is mechanically separable and every clause closes exactly once, and generates `obligation-preservation.md` as a reader projection. Thirty ledger mutations cover row-set drift, stale or duplicate carriers, qualifier weakening/reversal, table hiding, retirement misuse, and invalid bundles. The companion UI/UX oracle reproduces 236 baseline failures and kills 91 independent contract mutations (81 direct branches plus 10 parameterized owner/Phase-0 cases). These checks prove static preservation, reachability, and selected contract properties only; they do not prove lower task time, fewer corrections, better rendered design, or production effect. |
393
+ | Routing-bank evidence covers every skill description independently of the curated impact-chain owner set: a non-curated owner with valid evidence must pass, the same change with no row must fail, and one ambiguous row cannot discharge two owners | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; implementation lives in `skills/skill-extraction-workflow/scripts/impact-chain-gate.rb`. Hosted CI exposed that `terminal-cli-dev` had a valid owner-scoped bank locator but was looked up only in the selected-owner map. Reproduction then isolated both directions: A20 failed with valid evidence, while A21 incorrectly passed with no ledger row because the outer gate never entered. The gate now has a bank-only resolver over base/head skill names and enters on any changed skill entrypoint, while selected owners keep the original impact-chain resolver. A20/A21 turn green, A22 rejects a two-owner row, A23 proves a stale exact owner path is rejected by the earlier missing-file gate, and round-attribution Leg M remains green for a non-curated body-only change. A separate lineage trace found no reachable difference in which the selected resolver loses a valid bank row: extra in-round owners remain ambiguous, absent exact paths fail missing-file, and absent package-prefix aliases fail the earlier ambiguity gate. Thus routing-bank coverage widens without widening impact-chain jurisdiction. |
394
+ | A preservation audit that lives only as a documented manual command drifts silently, and a Markdown obligation parser that ignores CommonMark container and code-span semantics lets inert example text become obligations or lets identifier-shaped URLs launder private numeric IDs; the real mapping/ledger must be audited by a registered gate against the base commit pinned in its own header | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_obligation_ledger_repo_audit.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` is unchanged. Independent review of the delivery found the committed obligation mapping stale at the PR head: the documented completion audit exited 1 while validation evidence recorded it as passing, and no CI lane ran the real-repository audit. The repaired mapping restores `audit_ok` over 1240 rows, and the new heavy-lane suite `test_obligation_ledger_repo_audit.sh` audits the real mapping/ledger against the base SHA pinned in the ledger header, so later carrier drift turns CI red instead of shipping silently. `obligation-ledger.py` parsers were aligned with CommonMark across eight adversarial review rounds: code-span pairing in table cells, section bodies sliced from masked visible text, clause splitting on code-span-masked structural text, and container-aware indented-code masking for top level, list items, and block quotes with quote-relative indentation; the parser-boundary leg in `test_obligation_ledger.sh` replays each defect RED on the pre-change tool. The long-digit scan in `check-ccl-skills.sh` now scopes its DOI/W3C exemption to the identifier payload with a punctuation-merged nine-digit cap, closing split, padded, and report-name laundering while keeping numeric-suffix citations green; `test_entrypoint_domain_scan_terms.sh` pins eight RED laundering classes and the green controls. The regenerated ledger stayed byte-stable across every parser change, and the round-by-round record with rejected-escalation rationale lives in the delivery spec's review-fix record. |
395
+ | A merge that combines a bank-only owner resolver with a base-absent created-surface bound must re-adjudicate the bound's premise: once rows bind over base plus head skill names, exempting existing non-curated description edits only reopens the evidence-free routing-surface hole the resolver closed | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_impact_chain_self_adjudication.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` is unchanged. Integrating dev into this branch auto-merged two divergent gate changes: this branch's bank-only resolver and dev's base-absent bound on the created-surface pickup, whose stated premise was that non-curated obligations were undischargeable. Under the merged gate legs A21/A22 went RED (zero-row description edits passed). The bound is removed with its comment rewritten to record the supersession; A21/A22 return green, dev's Node owner leg is preserved as A24, and dev's exemption leg is rewritten as A25 to pin the dischargeable half: an existing non-curated owner's description edit owes bank evidence and a bank-only row discharges it. |
396
+ | Merging two branches that each rewrote one routing description resolves at the obligation level, not the text level: the union carries both rounds' obligations, and any budget trim must come out of non-carrier wording | `terminal-cli-dev` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; bank-evidence: downscoped:PR76-INTEGRATION-TRIM-NO-BANK-RERUN | `updated` | Owner key: `terminal-cli-dev/SKILL.md`. Description-only integration: the compressed entry from this branch absorbs the `Node.js CLI → nodejs-service-dev` skip leg from dev beside the Python and Go legs, and the obligation mapping row for the skip sentence records the clause on both its before and carrier sides. The OpenCode budget overflow from the union is repaid inside the uncarried ownership sentence: the pinned short contract phrase and its parenthetical stay verbatim for the cross-skill pointer pair, and the fuller enumeration folds behind it as a with-clause; all four description carriers are byte-preserved and the re-pinned ledger audit stays `audit_ok` over 1240 rows. |
397
+ | A ledger effect vocabulary without a neutral deletion value forces every retired obligation to close as strengthened, so summary counts read as zero loss while carrier-less deletions ship; deletion closes as retired, live-carrier rows may not claim it, and a migration byte budget must carry real headroom with recorded rationale instead of freezing its documents at a reverse-fitted cap | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` is unchanged. The 065 ledger recorded 98 retired-dead and 25 partial-retirement rows as effect=strengthened because the vocabulary had no other closable value, so the rendered Effects line read as zero obligation loss. EFFECTS gains retired; retired-dead rows and retired-part partitions must close as retired; a retired effect on a live-carrier row is rejected; the rendered summary reports retired separately. The specs/065 mapping is relabeled (preserved=657, strengthened=460, retired=123) and re-rendered under the pinned-base audit. Three applied mutations, each attributed differentially with the unmutated suite green: strengthened on a retired-dead row fails RETIRED_EFFECT_INVALID where it previously survived validation to STALE_LEDGER; strengthened on a retired-part partition fails PARTITION_EFFECT_MISMATCH where it was previously the required baseline; retired on a rehosted carrier row fails RETIRED_EFFECT_INVALID where it previously fell to the generic INVALID_EFFECT. The uiux loading budget's direct-runtime cap sat at exactly its measured value (58860 = 90% of 65400, a zero-byte margin that froze the entry and contract files); the cap moves to 95% with the cap-setting rationale and re-derive-or-retire policy recorded in the script, and a router-shrink-plus-contract-pad mutation kills the 95% predicate specifically while the 110% specialized cap stays green. |
398
+ | A review round's fixes are their own gate round: a budget cap verified by one off-suite mutation can be silently weakened later, and a mixed partial retirement that closes row-level as retired must not let its strengthened surviving parts vanish from the rendered summary | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` is unchanged. Independent review and adversarial challenge of the retired-effect round produced two applied findings. First, the loading budget's caps move into pinned variables with an in-suite threshold probe: the 50/95/110 contract pin plus a cap-versus-cap+1 boundary walk per predicate; an applied 95-to-100 weakening mutation turned the probe RED and was restored. Second, 24 of the 25 real partial-retirement rows carry a strengthened surviving part, and the row-level retired close dropped that dimension from the summary; the renderer now emits part-level outcome counters, the fixture gains a mixed row pinning survived-preserved=3, survived-strengthened=1, retired=2, and a new mutant proves an unreviewed partition still fails MANUAL_REVIEW_REQUIRED, hardening the refuted manual-review finding into an invariant. The round-1 claim that retired rows escape manual review was refuted empirically: flipping one real partial-retirement row's manual_reviewed to false fails the pinned-base audit at the schema4 partition entry check. |
399
+ | Bound review-plan intent: overflow preserves bytes; compaction keeps only identified core+latest | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_update_review_plan_intent.sh; result-class: failure | updated | Helper bounds no-follow UTF-8 reads, checks optional opened digest, and atomically preserves mode. Compaction requires `review-plan-intent-stable-core-v1` character-count+SHA identity; legacy plans fail closed. Pre-rename failure leaves the plan unchanged; directory-sync failure after rename is committed/durability-unknown and MUST NOT be blindly retried. Tests pin boundaries, identity, hostile inputs, and both commit states. Callers own serialization and core semantics. |
400
+ | Repeated class: bind occurrences, sweep at three, stop at round three, and terminate after drift two | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh; result-class: failure | updated | Wrapper fixes review+two challenges at capability-only budget 2. V2 validator SHA-binds same-dir v3 receipts; ready needs exact review+challenge+`complete/self_reviewed`. Round-3 findings require `continuation_authorization_required`—never an automatic fourth round; a second recorded base drift terminates `baseline_race`. Findings classify once; unresolved stays open/human-decision and `source_refuted` requires caller proof. Occurrence three binds an authoritative sweep; zero unmatched is required only for ready, while continuation/race retain evidence. Ordered raw `ls-remote` catches A→B→A. Tests pin failures. Omitted history, live authority, CAS, and classification truth remain caller-owned. |
401
+ | Scan candidate commits, destination, and exact PR text; merge base excludes ancestor debt and PR edits rerun CI | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh; result-class: failure | updated | Bounded no-echo tests cover refs/ranges, provider metadata, a cross-product of every line-oriented metadata class with Markdown list/task-list/quote/heading/nested wrappers plus malformed-checkbox precision controls, CRLF, legacy commit encodings, malformed raw messages, malformed batch framing, colliding short locators, local replacement refs, legacy grafts, and controls. Candidate metadata is requested from Git with explicit UTF-8 output; invalid UTF-8 or a replacement character fails closed with only locator and field, so ambient `i18n.logOutputEncoding` cannot erase a CJK-only match. Candidate OIDs are enumerated separately; the bounded raw-object batch is length-parsed, type/size/trailer checked, and bound in order to each full OID before payload NUL rejection and pretty formatting. Twelve-character locators are diagnostic only. Every Git object read disables `refs/replace/*`, because a clean local replacement does not change the original object a push publishes; a non-empty `info/grafts` file fails before and after the scan because it can similarly hide the pushed parent chain. Provider-shaped unambiguous AI display names plus separator-normalized known-provider `[bot]` display or exact email accounts are blocked for coauthors, authors, and committers; email accounts allow only the standard optional numeric GitHub ID prefix, so a provider token embedded at the end of another bot account is not enough. Ambiguous single-name coauthor trailers need bot-like mail; author/committer headers never use generic email shape alone because a GitHub `noreply` privacy address is normal for humans, so those ambiguous identities stay an explicit human-readback gap and unrelated automation bots remain allowed. Base order is argument, trusted event, environment, default. A direct branch-push event with a missing or all-zero `before` may use a fallback only when it still yields a non-empty candidate range; a non-zero unreachable event base fails resolution, while local and PR-only empty ranges remain valid. Hook uses target/`origin/dev` for features and remote SHA for existing `dev`/`main`; it fails closed without a new-target base, skips deletes, scans all refs, and checks worktree only for pushed HEAD. Before PR create/edit, root policy requires exact draft title/body via `--pr-text-file`. CI direct-push triggers are limited to `dev`/`main`; PR triggers are `opened`/`synchronize`/`reopened`/`edited`, including text-only edits. Comments, labels, platform-generated merge messages, host-default footer conflicts, and ambiguous personal identities stay policy-owned; annotated-tag object messages are an explicit R0 coverage gap. |
402
+ | Supersedes `Bound review-plan intent...`: compaction retains every removed byte, review inputs are frozen and bounded, and the wording-only single-review exception is mechanically candidate-bound | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh; result-class: failure | updated | `code-review/SKILL.md` is the owner key. `test_update_review_plan_intent.sh` proves reversible suffix archival and fail-closed capacity; `test_review_gate.sh` proves open-once bounded files, Git routing/textconv isolation, frozen untracked bytes, and the one-skill full-context wording proof with independent `wording_only_boundary`. Missing/invalid proof leaves release/high-risk challenge-required; wording-only stays untracked and has no completion ledger. |
403
+ | Supersedes `Repeated class...`: the extraction wrapper applies only to non-wording lanes, and terminal readiness consumes the latest attested base while a second drift forbids later receipts | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh; result-class: failure | updated | `skill-extraction-workflow/SKILL.md` is the owner key. `test_extraction_review_gate.sh` pins owner-wrapper versus proof-bound wording-only routing; the validator accepts a later same-SHA recheck, rejects an unconsumed latest base, and terminates at the second ordered drift before any new receipt. Complete-history retention and live remote authority remain caller/platform boundaries. |
404
+ | Supersedes `Scan candidate commits...`: shared Git metadata scanning clears repository-routing state and closes wrapper, suffix, scheme-port, and branch-slug bypass classes in linear time | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh; result-class: failure | updated | `skill-extraction-workflow/SKILL.md` is the owner key. Tests cover hostile Git routing variables, scheme-correct ports, repeated-origin and separator performance, arbitrary paired Markdown wrappers, whole-link labels, model suffixes (including the Fable/Mythos model words the current harness signs with), emphasis wrapped around only the identity inside an anchored trailer, and English/Chinese branch continuation forms. Annotated-tag pushes are scanned per tag-object layer — message and tagger identity, nested tags peeled with a bounded depth — because head resolution otherwise peels straight to the commit. Comments, labels, custom merge messages, future provider aliases, and ambiguous human/provider names remain explicit platform/human-readback gaps rather than clean claims. |
405
+ | Supersedes the feature-base clause of `Scan candidate commits...`: pre-push derives the feature→dev base from the pushed remote instead of assuming `origin` | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh; result-class: failure | updated | The hook uses `refs/remotes/<pushed-remote>/dev`, fails with an explicit fetch-or-`CCL_SKILL_BASE_REF` remedy when it is absent, and keeps destination SHA precedence for existing `dev`/`main`. The focused suite reproduces a clone with only `upstream/dev`; the old literal blocks it and the corrected hook passes without narrowing to the remote feature tip. |
406
+ | Supersedes only the 600-second ceiling in the earlier Kimi packet-review row: a slow reviewer uses the existing generic `--timeout` instead of a client-specific option | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh; result-class: failure | updated | The 068 round plan freezes the scope. The controller now accepts 5..1200 seconds, rejects 1201 before client execution, and still defaults to 600; all four direct wrappers keep their default and clamp only above 1200. The cumulative `--total-timeout` deadline, per-mode allocation, client order, fallback rules, and provider/model selection are unchanged. RED on the prior candidate reported exactly two new failures: the controller rejected 1200, and the four-wrapper ceiling check found every old 600 clamp. The same full gate suite is GREEN after the change; `test_review_client_compat.py` also passes. |
407
+ | Supersedes the direct-wrapper input-domain and completion claims in the preceding timeout/plan-intent rows: bounded shell arithmetic is not a decimal parser, wording-only review is never completion evidence, and duplicate-key JSON is not loss-preserving input | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh; firing-path: command:skills/code-review/scripts/test_update_review_plan_intent.sh; result-class: failure | updated | The [068 timeout correction plan](../../../specs/068-review-loop-retro/plan.md) is an addendum to the existing retrospective candidate. One shared decimal-string normalizer now rejects values below 5 and clamps `1201` or a 50-digit decimal to 1200 before any raw-value arithmetic; all four wrappers bind to it, while Kimi inline mode's independent 120-second cap stays unchanged. Completion rejects any prior receipt carrying wording-only proof/scope. The plan updater rejects duplicate keys in every JSON object before mutation. Each newly added case was RED on the prior implementation and GREEN after the fix; defaults, client routing, fallback, provider/model selection, cumulative allocation, and ordinary plan updates remain controlled. |
408
+ | Model-qualified AI Git identities are the same prohibited attribution class as their unqualified product name | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh; result-class: failure | updated | The prior exact-name grammar allowed `Claude Sonnet`, `OpenAI Codex`, `ChatGPT-5`, `Qwen2.5-Coder`, and `Gemini 2.5 Pro` through co-author trailers, and corresponding author/committer fields also escaped. The bounded display-name grammar accepts only known product/version shapes after an AI provider, including attached Qwen versions and Gemini's versioned Pro form, without turning nearby human names such as Claude Monet, Qwen Li, or Gemini Proctor into AI identities. Trailer, author, and committer regressions were RED before the change and GREEN after it. |
409
+ | Committed-range Git-surface changes remain public-sanitized, including provider vocabulary and privacy-style email fixtures | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh; result-class: failure | updated | `skill-extraction-workflow/SKILL.md` is the owner key. The committed-range private-alias audit rejected a newly added provider token, and the public sanitizer rejected a non-example fixture address that pre-commit tracked-file scans had not seen. The registry now omits the colliding token, every literal fixture email uses an allowed example domain, and the focused behavior suite still passes. |
410
+ | Supersedes the wording-only clause of `Supersedes the direct-wrapper input-domain...`: a punctuation-only proof must preserve numeric tokens byte-for-byte and never certify code-container edits, and a committed plan update whose receipt is lost is a terminal nonzero state | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh; firing-path: command:skills/code-review/scripts/test_update_review_plan_intent.sh; result-class: failure | updated | `code-review/SKILL.md` is the owner key. Independent review of the 068 candidate found the punctuation-only skeleton certifying numeric-token merges (`5.5` to `55`) — the proof kind that waives the release challenge — and token replacement running inside fenced or HTML code containers; the pre-fix predicate reproducibly certified the merged-number edit (RED probe on the prior commit) and both classes now fail closed with numeric-merge and fenced-token regressions. The `wording_only_boundary` concern names the actually permitted scope: punctuation-only, whitespace edits rejected. `update_review_plan_intent.py` maps a stdout pipe closed before the success receipt to `plan_committed_receipt_lost` on stderr with a nonzero exit instead of rc 0, mirroring the durability-unknown contract; the already-correct `plan_core_identity_mismatch` and `intent_history_evidence_overflow` branches gained killing tests (coverage additions, not behavior fixes). |
411
+ | Supersedes the packet-fidelity and waiver clauses of `Supersedes the wording-only clause...`: a frozen base-mode packet must carry the exact changed bytes and the punctuation waiver must not flip assertive polarity | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_review_gate.sh; firing-path: command:skills/code-review/scripts/test_update_review_plan_intent.sh; result-class: failure | updated | `code-review/SKILL.md` is the owner key. Cross-family external review (one review plus one challenge per partition over the four-packet partition of the exact candidate) surfaced: an in-tree `.gitattributes -diff` entry collapsed changed lines to a binary marker inside the frozen packet (now forced `--text`; a truly binary file fails the packet NUL check instead of passing unseen); repository-local `core.worktree` could redirect discovery to a decoy tree (now refused fail-closed, and the resolved root must contain `--cwd`); a punctuation proof certified `input.` to `input?` (question marks are now outside the certifiable set while exclamation marks remain pinned eligible by fixtures); the flushless success receipt let BrokenPipeError escape to interpreter shutdown, bypassing the receipt-lost handler and exiting 120 (now flushed in the guarded path with stdout parked on devnull before the terminal rc 2); and the packaged runtime closure did not require `update_review_plan_intent.py` (now in the verifier's required list with its own removal mutation test). A second adversarial pass on the corrected candidate closed the same-shape residue: `script`/`style`/`textarea` join the tracked code containers, untracked bytes split on LF only, and question/quote marks are compared with skeleton anchors across an extended Unicode question set so a mark cannot slide, appear, or vanish. Repository-local config attacks stay on the pinned neutralization posture (per-invocation `-c` disables with fixtures asserting the true packet survives a hostile include) rather than gaining refusal predicates; the clean-filter execution residue is a recorded disposition, not a silent claim. Each closed class carries a fixture that was RED against the prior behavior. |
412
+ | Supersedes the identity-grammar and tag-layer clauses of the shared-scan supersede row: the metadata scan recognizes current cross-provider model words, session-task URLs, and only canonically-shaped tag objects | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_shared_git_surface_gate.sh; firing-path: command:skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh; result-class: failure | updated | `skill-extraction-workflow/SKILL.md` is the owner key. The same external round surfaced: bare `GPT-*` identities and `Gemini * Flash/Ultra` model forms escaped every surface (now in the registry and unambiguous grammar with trailer regressions); Codex task URLs on session origins (`.../codex/tasks/<id>`) matched no session-path grammar (now matched, with a product-page near-miss control); a literal tag object could duplicate its `object` header so the scanner followed a decoy chain while Git peels the first target (canonical header shape now required, duplicates fail closed); the closeout validator's completion-receipt `schema_version` guard lacked the exact-integer type check applied everywhere else (a float `3.0` validated; now rejected with a regression), and `bounded_text` accepted C1 controls and U+2028/U+2029 line separators into diagnostics (now rejected with per-codepoint regressions). The second pass additionally rejected default-ignorable format characters that could split one recurrence class into visually identical keys, and bound every counted chain round to the exact ledger candidate rather than only the final receipt. |
413
+ | Supersedes the fixture-portability posture of the plan-intent suite row: a mode assertion probes GNU stat before BSD stat | `code-review` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_update_review_plan_intent.sh; result-class: failure | updated | `code-review/SKILL.md` is the owner key. GNU stat echoes unknown BSD directives (`%Lp`) verbatim with exit 0, so the suite's BSD-first probe never reached its GNU fallback and CI shard 2 failed `plan mode was not preserved` on every Linux run while macOS stayed green. The assertion now probes `stat -c` first and falls back to `stat -f '%Lp'`, mirroring the pattern `opencode_review.sh` already runs on both platforms. |
414
+ | Supersedes the comparison-domain clause of the obligation-preservation audit: a repository-frozen ledger pins BOTH ends of its domain, so unrelated later changes owe it nothing | `skill-extraction-workflow` | behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_obligation_ledger.sh; firing-path: command:skills/skill-extraction-workflow/scripts/test_obligation_ledger_repo_audit.sh; result-class: failure | updated | `skill-extraction-workflow/SKILL.md` is the owner key. The 065 obligation audit derived its row set from pinned-base..WORKING-TREE, so the first post-landing PR that rewrote any obligation line in any `skills/**/*.md` went red on preservation rows it never owed (observed: 12 phantom rows for this candidate's own code-review contract-paragraph rewrite). `obligation-ledger.py` now accepts `--head`, the ledger header pins `Head revision` beside the base, and the repo audit reads and requires it; carrier-drift detection still reads current files, so rewriting a BOUND carrier stays red (probed: the pinned audit still fails `CARRIER_COMPOSITE_NOT_UNIQUE` on a carrier rewrite). The synthetic suite adds the differential: a post-head non-carrier rewrite passes the pinned audit and fails the unpinned one with `ROW_SET_MISMATCH`. |
@@ -251,6 +251,14 @@ Rules:
251
251
  - Do not count source-register or source-map text as the skill itself. Provenance proves where a rule came from; the owning skill/reference must still tell a future user how to apply, implement, test, or reject the rule without reading the provenance trail.
252
252
  - **Reconcile the map against every upstream disposition surface before closeout — the map is a transcription, and transcriptions drop rows.** Walk every disposition surface still valid for the current scope — whichever round or session produced it (per-axis lesson tables, candidate ledger, a refined or replay-produced map; a cross-session resume inherits the prior round's surfaces, it does not reset them) — and list the surfaces actually reconciled: every row whose disposition carries a forward obligation — `routed`, `landed`, `updated`, `pending`, or any status naming an owner/output — must resolve to a map row (`updated`/`unchanged`/`routed`/`pending` with owner — a `pending` map row preserves the blocker per the existing pending rule instead of forcing a fake terminal status) or carry a superseded note that names the successor map row covering the same mechanism; derive the surface list from the round's own records (charter, source register, produced-artifact rows), not from memory — a disposition surface named in those records but absent from the reconciled list keeps the claim `interim` (wholesale omission of the records themselves is fabrication, out of scope for this gate and covered by the never-fabricate red line); only evidenced terminal rows (`discarded`/`no-new-lesson`/`not-applicable` with reason) need no map row, and a superseded note with no successor is `downgraded`/`pending`, not closure. A dispositioned row silently absent from the map blocks the complete claim — "diff matches map" then passes while the lesson is lost, which is the exact hole this check closes (observed shape: an axis table routed a craft lesson to two owners; the refined map omitted the row; the landing round verified diff-vs-map clean and shipped without it, caught only by user challenge).
253
253
 
254
+ ## Member-Join Inheritance
255
+
256
+ The SKILL.md class-wide COMPLETE-set rule (example of a class-wide change: "every stack `*-dev`/`*-architecture` should advertise a localized-refactor trigger"; example of a member-pair landing: Python+Go) fires on change-side sweeps. This section is its join-side dual.
257
+
258
+ - When a NEW skill joins an already-swept class (e.g. a new stack `*-dev` joining the stack-implementation-owner class), enumerate the class-wide obligations its siblings already carry — advertised triggers, skip-leg reciprocity on routing counterparties, pinned pointer-integrity anchors, and bank fixtures — and inherit each, or record a per-obligation `not-applicable` reason, before the member lands.
259
+ - A sibling-generalization map that only answers the copied-content question ("does the new skill duplicate sibling text?") must not be counted as obligation-inheritance coverage; the two questions are independent, and the join-side miss survives a clean copied-content map.
260
+ - Validation: the closeout map lists each inherited obligation with a status, exactly as the change-side sweep lists each member. Failure shape: a new stack implementation owner landed with "siblings unchanged because the new skill routes to their existing contracts" while lacking the CLI carve-out reciprocity and the localized-refactor trigger pair its siblings carry; only independent review caught it.
261
+
254
262
  ## Capability Naming And Provenance
255
263
 
256
264
  Use this when naming a new skill, naming a reference file, renaming an extracted artifact, or turning a source-specific pattern into reusable guidance.
@@ -86,7 +86,7 @@ Use source-specific judgment, but verify with concrete proxies:
86
86
  - One primary visual focus per task state; secondary controls should not compete with the main artifact or decision.
87
87
  - Use stable spacing rhythm such as 4/8px increments unless the source system clearly uses another rhythm.
88
88
  - Record token provenance before extracting visual rules: the source of typography, primary color, neutral/background scale, radius, shadow/elevation, and component density. If the source has multiple plausible visual directions or the implementation code hardcodes page-level colors, treat that as extraction evidence and decide whether the reusable rule should require a visual direction comparison, token cleanup, or both.
89
- - Touch targets should meet platform norms: at least 44pt on iOS and 48dp on Android where touch is required.
89
+ - Touch targets should follow current first-party platform guidance. For buttons, preserve Apple HIG's general-rule hit region of at least 44×44pt (60×60pt on visionOS) and Android's at-least-48×48dp touch/focusable guidance as platform-scoped inputs, not universal constants; measure the actual hit region separately from the glyph.
90
90
  - Text contrast should meet WCAG expectations for normal and small text; do not rely on brand color alone for state.
91
91
  - Dense work surfaces should keep stable row/card height, fixed context when identity would be lost, and local overflow affordance instead of decorative whitespace.
92
92
  - Modal, drawer, overlay, and floating tool visual weight should match consequence: lightweight details should not look as severe as destructive confirmation.
@@ -106,12 +106,12 @@ For each flow, inspect and record:
106
106
 
107
107
  ## Behavioral And Psychology Anchors
108
108
 
109
- Each behavioral or psychology rule needs at least one falsifiable anchor:
109
+ Each behavioral or psychology rule needs an observable risk, a falsifiable hypothesis, and a named evidence boundary. A theory name is retrieval shorthand, not a universal design instruction; classify and source theory claims through `product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md`.
110
110
 
111
- - Hick: reduce active choices while the user is deciding; reveal controls when inspecting or correcting.
112
- - Fitts: repeated or high-risk controls need stable placement and adequate hit area.
113
- - Miller/cognitive load: chunk dense information; do not force users to hold hidden context across modal or route changes.
114
- - Doherty/feedback timing: show feedback quickly for user actions; use visible pending/final states for long work.
111
+ - Choice search: when competing options slow or confuse the representative task, reduce simultaneous choices or progressively disclose secondary controls; verify the task outcome instead of invoking a named law as proof.
112
+ - Target acquisition: repeated controls need stable placement, and interactive targets need a hit area appropriate to the current platform, input mode, task frequency, and consequence. Do not derive high-risk confirmation or an exact size from Fitts's law.
113
+ - Hidden-context burden: chunk dense information and preserve task context across modal or route changes; verify recall, comparison, and recovery in representative tasks rather than asserting a fixed working-memory limit.
114
+ - State uncertainty: acknowledge actions and expose pending, success, failure, and recovery states at a pace appropriate to the operation. An arbitrary response-time threshold is not acceptance evidence.
115
115
  - Error prevention: prevent invalid input before submit when possible; when not possible, show local repair guidance.
116
116
  - User control: provide cancel, undo, retry, edit, restore, or explicit irreversible confirmation depending on consequence.
117
117
  - Trust: show source, scope, permission, timestamp, automation caveat, and consequence near the affected decision.
@@ -128,14 +128,14 @@ Before commit, for any rule that appears in more than one authored file (`SKILL.
128
128
 
129
129
  ## Bounded Independent Review Packet
130
130
 
131
- Use this when an independent review is required but a broad reviewer prompt hangs, returns no output, or starts expanding beyond the intended review scope. The gate-valid path is the provider-neutral `code-review` gate (`review_gate.sh`) with its frozen packet, family exclusion, structured validation, and client-specific recovery; raw provider CLI packets are debugging/advisory only and must not be recorded as passing review evidence.
131
+ Use this when an independent review is required but a broad reviewer prompt hangs, returns no output, or starts expanding beyond the intended review scope. For non-wording extraction work, the gate-valid path is `scripts/extraction_review_gate.sh`, which owns the fixed autonomous budget while delegating transport to the provider-neutral `code-review` controller. A strictly proven wording-only change uses the proof-bound generic single-review recipe in `code-review/references/staged-review-contract.md`; its controller-derived wording scope and independent `wording_only_boundary` result replace neither one another nor a failed semantic check. Both paths preserve the frozen packet, family exclusion, structured validation and client-specific recovery; raw provider CLI packets are debugging/advisory only and must not be recorded as passing review evidence.
132
132
 
133
133
  Required flow:
134
134
 
135
- 1. Prove the reviewer gate is available. Run `skills/code-review/scripts/review_gate.sh --help` or the repo-local equivalent and confirm it prints usage. This local availability check does not invoke a model and is not review evidence. Do not use a raw provider "ping" as gate evidence. If the gate cannot run after documented remediation, record the lane as unavailable/inconclusive with command evidence.
135
+ 1. Prove the applicable owner gate is available. For non-wording work, run `skills/skill-extraction-workflow/scripts/extraction_review_gate.sh --help`; for a strictly proven wording-only change, use the generic `code-review` help recipe. This local availability check does not invoke a model and is not review evidence. Do not use a raw provider "ping" as gate evidence. If the gate cannot run after documented remediation, record the lane as unavailable/inconclusive with command evidence.
136
136
  2. Stop the stuck review process before retrying. Do not leave background reviewer sessions running and do not count a no-output process as a completed review.
137
137
  3. Build a bounded packet from the exact changed files or excerpts being claimed. Prefer `git diff -- <files>` for local changes.
138
- 4. Feed the packet through `review_gate.sh` with the actual implementer family, required `--review-plan-file`, stage, exact base/paths or frozen diff file, and explicit non-Claude egress approval when applicable. The plan binds the target, acceptance criteria, self-review, and evidence before the independent run. The gate preserves tool posture, timeout, attribution, and output parsing while following the user's local client order. A raw Claude packet may be used only to debug a wrapper failure and remains advisory:
138
+ 4. Feed a non-wording packet through `scripts/extraction_review_gate.sh`; feed a strictly proven wording-only packet through the proof-bound generic single-review recipe. The wording-only path uses its exact canonical full-context diff and proof file, stays untracked, and does not run `complete`; a custom/context-augmented packet or missing `wording_only_boundary` result re-arms challenge. In either case supply the actual implementer family, required `--review-plan-file`, stage, exact candidate, and explicit non-Claude egress approval when applicable. The plan binds the target, acceptance criteria, self-review and evidence before the independent run. The gate preserves tool posture, timeout, attribution and output parsing while following the user's local client order. A raw Claude packet may be used only to debug a wrapper failure and remains advisory:
139
139
 
140
140
  ```sh
141
141
  git -C <repo> diff -- <files...> \
@@ -151,6 +151,7 @@ Review only stdin. Do not use tools. Do not request more context. Return finding
151
151
  5. Apply or explicitly reject actionable findings.
152
152
  6. Rerun the owning skill validators and `git diff --check`.
153
153
  7. Record the result as `findings applied`, `no blocking findings`, or `review unavailable after remediation` only when the wrapper or approved alternate produced valid structured evidence. Raw packet output is recorded separately as debugging/advisory and cannot close the gate.
154
+ 8. At a non-wording terminal checkpoint, build the receipt-bound ledger and run `scripts/validate_extraction_review_state.py <closeout.json>`. A wording-only review records its single independent-review row and does not invent a multi-round ledger.
154
155
 
155
156
  Do not count as completed review:
156
157
 
@@ -280,8 +280,11 @@ end
280
280
  # Maintain by RETIREMENT, not by growth: when a term no longer proxies anything
281
281
  # live, delete it (quantify current hits first, and keep a retained-term control
282
282
  # so the scan is proven still able to fail). Do not add generic technical words.
283
- # Both directions are pinned by test_entrypoint_domain_scan_terms.sh.
284
- if leak_output="$(rg -n 'code\.[[:alnum:].-]+|figma\.com/files|[0-9]{8,}|\x{6559}\x{5e08}|\x{5b66}\x{751f}|\x{8003}\x{8bd5}|\x{5b66}\x{6821}|\x{9605}\x{5377}|\x{51fa}\x{5377}|\x{5b66}\x{60c5}' "$root"/skills/*/SKILL.md "$root"/skills/*/references/*.md 2>/dev/null)"; then
283
+ # Both directions are pinned by test_entrypoint_domain_scan_terms.sh. Long digit
284
+ # runs are scanned separately so canonical public DOI/W3C identifiers are not
285
+ # mistaken for private object IDs while an unrelated ID on the same line still
286
+ # fails closed.
287
+ if leak_output="$(rg -n 'code\.[[:alnum:].-]+|figma\.com/files|\x{6559}\x{5e08}|\x{5b66}\x{751f}|\x{8003}\x{8bd5}|\x{5b66}\x{6821}|\x{9605}\x{5377}|\x{51fa}\x{5377}|\x{5b66}\x{60c5}' "$root"/skills/*/SKILL.md "$root"/skills/*/references/*.md 2>/dev/null)"; then
285
288
  echo "$leak_output"
286
289
  echo "entrypoint_or_reference_domain_scan_failed" >&2
287
290
  exit 1
@@ -292,6 +295,70 @@ else
292
295
  exit 1
293
296
  fi
294
297
  fi
298
+
299
+ if ! long_digit_output="$(ruby -e '
300
+ allowed_url = %r{https://(?:
301
+ doi\.org/10\.[0-9]{4,9}/(?<doipayload>[A-Za-z0-9._;()/:+\-]+)
302
+ |
303
+ www\.w3\.org/community/reports/[a-z0-9-]+/CG-FINAL-[A-Za-z0-9._-]*(?<w3cdate>[0-9]{8})/?
304
+ )}ix
305
+ ARGV.each do |path|
306
+ next unless File.file?(path)
307
+ File.foreach(path).with_index(1) do |line, line_number|
308
+ allowed_spans = []
309
+ line.to_enum(:scan, allowed_url).each do
310
+ match = Regexp.last_match
311
+ url = match[0]
312
+ next if url.include?("?") || url.include?("#")
313
+ # The allowlist is shape-only (no registrant validation), so a
314
+ # fabricated DOI-shaped wrapper could otherwise launder any private
315
+ # numeric ID. A span grants exemption only when its identifier
316
+ # payload (the DOI suffix after the registrant slash, or the W3C
317
+ # report date) carries no merged digit run above 9 digits, where a
318
+ # merged run joins digit groups across one or MORE consecutive
319
+ # punctuation separators — every punctuation char the DOI payload
320
+ # charset accepts, so no accepted punctuation can split an ID into
321
+ # exempt halves. Letters stay run boundaries because real DOI
322
+ # suffixes legitimately interleave them (s15516709cog1202_4). This
323
+ # keeps real citations green (10.1038/35057062 has an 8-digit
324
+ # payload; the registrant prefix is bounded to 9 digits by shape)
325
+ # while a split or padded identifier such as 123456789-123456789,
326
+ # 20260830--123456, or 123456789+123456789 voids its span entirely.
327
+ payload_name = match[:doipayload] ? :doipayload : :w3cdate
328
+ payload = line[match.begin(payload_name)...match.end(payload_name)]
329
+ payload_ok = payload.scan(/[0-9]+(?:[-._\/;():+]+[0-9]+)*/).all? do |run|
330
+ run.delete("^0-9").length <= 9
331
+ end
332
+ next unless payload_ok
333
+ # The exempt span is the identifier payload, not the whole URL: a
334
+ # W3C report NAME has no business carrying a long digit run, so only
335
+ # the trailing date is exempt there; a DOI suffix is the identifier
336
+ # itself, so its whole payload is exempt once it passes the cap.
337
+ allowed_spans << (match.begin(payload_name)...match.end(payload_name))
338
+ end
339
+
340
+ leaked = line.to_enum(:scan, /[0-9]{8,}/).any? do
341
+ match = Regexp.last_match
342
+ # Per-match cap: even inside a payload-clean span, a single raw run
343
+ # above 9 digits (timestamp/snowflake scale) is never exempt.
344
+ exempt = match[0].length <= 9 && allowed_spans.any? do |span|
345
+ span.begin <= match.begin(0) && match.end(0) <= span.end
346
+ end
347
+ !exempt
348
+ end
349
+ puts "#{path}:#{line_number}:#{line.chomp}" if leaked
350
+ end
351
+ end
352
+ ' "$root"/skills/*/SKILL.md "$root"/skills/*/references/*.md 2>&1)"; then
353
+ echo "$long_digit_output" >&2
354
+ echo "entrypoint_long_digit_scan_error" >&2
355
+ exit 1
356
+ fi
357
+ if [[ -n "$long_digit_output" ]]; then
358
+ echo "$long_digit_output"
359
+ echo "entrypoint_or_reference_domain_scan_failed" >&2
360
+ exit 1
361
+ fi
295
362
  echo "entrypoint_and_reference_domain_scan_ok"
296
363
 
297
364
  # Recurring anti-pattern scan: a TARGETED mechanical backstop (NOT a comprehensive
@@ -0,0 +1,22 @@
1
+ #!/usr/bin/env bash
2
+ # Extraction-owned autonomous review wrapper: one review plus two challenges.
3
+ set -euo pipefail
4
+
5
+ SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
6
+ CONTROLLER="$SCRIPT_DIR/../../code-review/scripts/review_gate.sh"
7
+
8
+ for arg in "$@"; do
9
+ case "$arg" in
10
+ --challenge-b*)
11
+ echo "extraction_review_gate_error: challenge budget is fixed at 2" >&2
12
+ exit 2
13
+ ;;
14
+ esac
15
+ done
16
+
17
+ if [[ ! -x "$CONTROLLER" ]]; then
18
+ echo "extraction_review_gate_error: code-review controller is unavailable" >&2
19
+ exit 2
20
+ fi
21
+
22
+ exec bash "$CONTROLLER" --challenge-budget 2 "$@"
@@ -132,6 +132,7 @@ upstream_owner_skills = %w[
132
132
  llm-inference-integration
133
133
  multi-agent-delegation
134
134
  multi-perspective-research
135
+ nodejs-service-dev
135
136
  platform-observability
136
137
  platform-release-engineering
137
138
  platform-service-connectivity
@@ -292,8 +293,10 @@ upstream = upstream.reject do |path|
292
293
  rename_excused[path] = true if excused
293
294
  excused
294
295
  end
295
- # The block runs when a selected owner changed (there is something to demand) OR
296
- # when the ledger itself changed (there is something to validate). It used to
296
+ # The block runs when a selected owner changed (there is something to demand),
297
+ # when any skill entrypoint changed (its description may owe routing-bank
298
+ # evidence), OR when the ledger itself changed (there is something to validate).
299
+ # It used to
297
300
  # gate on `upstream.any?` alone, which is the demand side only — and that is the
298
301
  # other half of the restored-owner hole: revert the owner and the range has no
299
302
  # changed owner at all, so the entire row evaluation was skipped and the
@@ -303,7 +306,10 @@ end
303
306
  # `eval/routing-tasks.jsonl` changes no owner package and need not touch the
304
307
  # ledger, so without this the whole block — including the bank-evidence
305
308
  # obligation written for exactly that change — was never entered.
306
- if upstream.any? || changed_paths.include?(LEDGER_PATH) || changed_paths.include?("eval/routing-tasks.jsonl")
309
+ routing_entrypoint_changed = changed_paths.any? do |path|
310
+ path.match?(%r{\Askills/[^/]+/SKILL\.md\z})
311
+ end
312
+ if upstream.any? || routing_entrypoint_changed || changed_paths.include?(LEDGER_PATH) || changed_paths.include?("eval/routing-tasks.jsonl")
307
313
  # Rows are collected PER ROUND and carry the scope they were authored against,
308
314
  # so the classifiers below judge a row against its own round's diff. Reading the
309
315
  # cumulative register diff instead would re-judge every earlier round's rows
@@ -1564,6 +1570,13 @@ if upstream.any? || changed_paths.include?(LEDGER_PATH) || changed_paths.include
1564
1570
  round_bounds.each do |span_base, span_head|
1565
1571
  scope = scope_at.call(span_base, span_head)
1566
1572
  scope_rows = rows.select { |row| row[:scope].head == span_head }
1573
+ # Routing-bank evidence applies to every skill description, not only to the
1574
+ # curated impact-chain owners. Keep a separate vocabulary for this one
1575
+ # obligation so broadening bank coverage cannot broaden the impact-chain
1576
+ # subject set or its row refusals.
1577
+ bank_owner_names = (
1578
+ owner_names_at.call(scope.base) + owner_names_at.call(scope.head)
1579
+ ).uniq
1567
1580
  # Same head-declared-grammar rule as `result-class` above, and for the same
1568
1581
  # reason: the paragraph that defines `bank-evidence` also dates it.
1569
1582
  next unless grammar_declared_at.call(scope.head)
@@ -1578,6 +1591,13 @@ if upstream.any? || changed_paths.include?(LEDGER_PATH) || changed_paths.include
1578
1591
  # being silently outside the trigger because of how the owner set is built.
1579
1592
  created_surfaces = scope.changed_paths.filter_map do |relative|
1580
1593
  next unless relative.start_with?("skills/") && relative.end_with?("/SKILL.md")
1594
+ # This pickup covers both a brand-new entrypoint and an EXISTING
1595
+ # non-curated skill editing its description — both move the routing
1596
+ # surface. An earlier round bounded it to base-absent entrypoints
1597
+ # because non-curated obligations were undischargeable then; the
1598
+ # bank-only resolver below now binds rows over base+head skill names,
1599
+ # so the obligation is dischargeable and the bound would only reopen
1600
+ # the evidence-free description-edit hole.
1581
1601
  path = relative.sub(%r{\Askills/}, "")
1582
1602
  next if triggered_owners.include?(path)
1583
1603
  next unless description_touched.call(scope, path)
@@ -1623,7 +1643,31 @@ if upstream.any? || changed_paths.include?(LEDGER_PATH) || changed_paths.include
1623
1643
  end
1624
1644
  triggered_owners.each do |path|
1625
1645
  owner = path.sub(%r{/SKILL\.md\z}, "")
1626
- owner_scope_rows = rows_by_upstream_path[path].select { |row| row[:scope].head == scope.head }
1646
+ owner_scope_rows =
1647
+ if upstream_set[path] || lineage_extra[path]
1648
+ # Preserve the existing selected-owner resolver byte-for-byte: bank
1649
+ # coverage must not weaken or widen impact-chain attribution.
1650
+ rows_by_upstream_path[path].select { |row| row[:scope].head == scope.head }
1651
+ else
1652
+ # Mirror that resolver for bank-only owners. Declaring rows use the
1653
+ # owner-package prefix convention; non-declaring rows keep the prior
1654
+ # exact `<owner>/SKILL.md` convention. A row naming more than one owner
1655
+ # satisfies none of them, so one locator cannot discharge two routing
1656
+ # surfaces.
1657
+ scope_rows.select do |row|
1658
+ declares_impact_chain = row[:behavior].split(";").any? do |fragment|
1659
+ fragment.match?(/\A\s*behavioral-evidence:/i)
1660
+ end
1661
+ candidate_names =
1662
+ if declares_impact_chain
1663
+ bank_owner_names.select { |name| owner_mentioned.call(row[:evidence], name) }
1664
+ else
1665
+ row[:evidence].scan(owner_skill_md_path).flatten.uniq
1666
+ .select { |name| bank_owner_names.include?(name) }
1667
+ end
1668
+ candidate_names == [owner]
1669
+ end
1670
+ end
1627
1671
  bank_evidence_failures << path unless owner_scope_rows.any? { |row| row_satisfies.call(row, owner) }
1628
1672
  end
1629
1673
  if bank_changed
@@ -1637,6 +1681,7 @@ if upstream.any? || changed_paths.include?(LEDGER_PATH) || changed_paths.include
1637
1681
  warn "impact_chain_bank_evidence_missing: a round that changed the routing surface carries no bank evidence for that owner"
1638
1682
  warn " note: the routing surface is the SKILL.md frontmatter `description` entry and `#{bank_relative}`. The measurement protocol says it is mandatory, but nothing produced or consumed it, so not running it left no absence to detect — the reviewer saw a candidate, not a gap"
1639
1683
  warn " fix: add `bank-evidence: command:<changed executable>` or `bank-evidence: file:<changed markdown>#<unique anchor>` to that owner's row; to skip the run deliberately use `bank-evidence: downscoped:<token>` and record the same token in this round's spec, so the downscope is itself an artifact"
1684
+ warn " note: a row without a `behavioral-evidence:` fragment attributes to its owner only via a bare `<owner>/SKILL.md` citation in the evidence cell — the full `skills/<owner>/SKILL.md` spelling is reserved for locators and does NOT attribute, so a row that looks like valid evidence but cites only the full path never reaches the owner"
1640
1685
  bank_evidence_failures.each { |path| warn " owes bank evidence: #{path}" }
1641
1686
  exit 1
1642
1687
  end