@ccoalm/ccl-skills 0.14.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +10 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +39 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +114 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +36 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +188 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +63 -5
- package/dist/assets/release.json +41 -36
- package/package.json +1 -1
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: requirement-scope
|
|
3
|
-
description: 改动范围 / 影响范围 / scope / 需求拆分 / MVP 边界 / 非目标 / 版本切片 / 变更影响 / appetite / timebox —— 交付物是**变更边界**:in/out scope、受影响对象、依赖、MVP 与后续切片、appetite 与砍项、每个切片的验收范围。前提是方向已定。Skip 方向还没定、要先弄清「到底要什么」→ requirement-intent;要的是现状清单(现在怎么运作、有什么能力)→ requirement-baseline;风险定级与要哪些 gate → feature-risk-router;实现/发布计划 → product-rd-workflow;测试范围 → testing-strategy。
|
|
3
|
+
description: 改动范围 / 影响范围 / scope / 需求拆分 / MVP 边界 / 非目标 / 版本切片 / 兼容·回滚降级边界 / 变更影响 / appetite / timebox —— 交付物是**变更边界**:in/out scope、受影响对象、依赖、MVP 与后续切片、appetite 与砍项、每个切片的验收范围。前提是方向已定。Skip 方向还没定、要先弄清「到底要什么」→ requirement-intent;要的是现状清单(现在怎么运作、有什么能力)→ requirement-baseline;风险定级与要哪些 gate → feature-risk-router;实现/发布计划 → product-rd-workflow;测试范围 → testing-strategy。
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Requirement Scope
|
|
@@ -57,8 +57,14 @@
|
|
|
57
57
|
|
|
58
58
|
以下是生成可比较证据的默认协议。**降级的是「F4 自己充当统一合并门禁」这个声称,不是「必须测、且必须有人裁决」这个义务**——这两件事分开:落地判断交给本轮实际的 owner/风险/评审门禁,但**测量本身不可选**。任何动 routing 面(SKILL.md description、task-bank 判定面)的改动都必须按下列协议产出证据;没跑就是没收敛,不得进入独立评审、也不得声称本轮无回归。采用不同样本量时,须随工件记录理由,且该理由与本轮证据一同进入独立评审——「记了理由」本身不是豁免,自审通过的理由不构成已裁决:
|
|
59
59
|
|
|
60
|
+
**筛查分辨率 ≠ 行动分辨率(是两个数,不是一个)**:全量基线默认 `--replicas 3`,它是**筛查器**——把候选捞出来,不是给结论。判定面是保守共识(任一副本 FAIL 即 FAIL),于是三副本下一条真实认领率 80–90% 的用例读起来就是红的,一次孤立偏离读起来就是一条 finding。**任何按用例采取的行动——改它 owner 的 description、判它是回归、判某条冻结期望已过期——都要求那条用例自己有 ≥10 个有效观测**;runner 对任何 `replicas < 10` 的报告打 `screening_resolution_only`、并在 JSON 里落 `action_resolution: false`,两侧的 10 由 `test_eval_routing_bank_resolution.sh` 钉在一起,防止文档与执行体漂移。实测形态(115 轮,同一轮内三次反向出错):3 副本全量基线报 14 条失败,10 副本下其中 5 条是抖动、六条已起草的 description 改动全部建立在它们上面;其中 `p3-spec-then-tc` 的单次偏离被据以论证某冻结期望已过期,10 副本下该对手 3/17、论证撤回;`ab-c5` 被判成「改前 PASS、改后 FAIL」的邻居回归,钉在未改动树上的对照臂显示它改前就是 8/10 的边缘失败。**留在 `eval/evidence/` 里的三副本基线因此是候选清单,不是 findings 清单**——引用它开轮的人要先为自己要动的每条用例补齐观测。
|
|
61
|
+
|
|
62
|
+
**地板管「能不能动手」,不等于「点估计已经准到能和阈值比」**:10 次有效观测在 70–85% 区间的抽样误差约 ±15–20 个点。实测形态(115 轮,同一个候选、同一条用例 `ab-b5`):一次 10 副本得 5/10(50%),紧接着 20 副本得 17/20(85%),合并 22/30(73%)——单看前者会判成「稳定失败」并据以改描述,单看后者会判成「健康」。所以**任何按阈值分档的判断(稳定失败 / 边缘 / 抖动)必须读合并观测,落在约 45–75% 之间的读数在 10 副本下不构成判定**,要么补到 30 次以上,要么如实记成「区间未定」。同理,改前/改后的差值也按合并观测比:本轮那条 8/10→5/10 的「回归」在 24/30 vs 22/30 下相差两次命中,不可分。
|
|
63
|
+
|
|
60
64
|
1. 动任何 description 之前必须先跑 **≥10 轮有效观测**的稳定性基线,把稳定失败与抖动分开;抖动不得作为修改依据(grader 超时/不可解析轮不算有效观测,须补跑)。
|
|
61
65
|
2. 改后通过数必须在**最终措辞**上重测:中间稿的通过数在措辞再变的那一刻作废,不得挪用到最终候选的证据里。
|
|
66
|
+
**`newly_failed` 是候选,不是回归判定。** runner 的 `--baseline` diff 在三副本下按保守共识判 status,于是一次孤立偏离就把一条用例记进 `newly_failed`;而这个集合**每跑一次就换一批**。实测(115 轮,同一条分支上四次全量三副本运行):`{route-opencode-project-config, route-nodejs-arch}`、`{skip-pytest-cmd, ctrl-ai-risk, miss-refactor-python-unqualified, route-nodejs-arch}`、`{mem-api-log-redact, route-nodejs-arch}`——除 `route-nodejs-arch` 外每一条只出现过一次、再未复现,逐条做成对 20 副本探针后**无一可归因于该轮改动**(两例两臂分布完全相同,一例两臂都红,一例合并后相差两次命中)。所以:`newly_failed` 的每一条都要按「同一用例、改前/改后两棵树、合并 ≥20 次观测」复测才能称为回归,不得直接写进轮记录当回归清单;同样地,不得因为它每轮都有内容就把整轮判红。
|
|
67
|
+
|
|
62
68
|
3. 受影响邻居用例集默认改前/改后各 **≥3 轮**,集合须含期望 owner 自己的兄弟用例与高词面重叠的他 owner 用例;邻居回归作为独立 finding 交由本轮实际门禁处置——**该 finding 须以 blocking 记入本轮 dual-track 评审记录,且只能由独立评审方豁免,不能由实现者自行判定「本轮没有门禁采用这组证据」而放行**。降级的是「F4 自己充当合并门禁」这一声称,不是「回归必须被人裁决」这一义务;后者若也随之消失,这一条就只剩被裁决方自审。
|
|
63
69
|
4. 每轮判决必须连同 **runner 调用、grader 模型身份、候选身份**(commit 或描述内容指纹)与**原始逐轮工件的持久定位符**一并记入轮记录;没有定位符的通过数只能标注为 operator-reported,不得据以宣称修复轮已 concluded。
|
|
64
70
|
5. **单变量归因**:一次改前/改后对照只准动**一个路由变量**(一条 description,或同一 skill 不可分割的一组路由面)。同时动多条 description 的批量改动,其对照差值不可归因到任何一条,只能按整包回归读——要归因就拆成逐条 A/B。(源侧实测形态:仅替换一条 description 的成对子集对照,把命中从约 2/3 提到 95%,且提升可归因到那一条改动——多条同动时这句话说不出口。)
|
|
@@ -48,7 +48,7 @@ Disposition:
|
|
|
48
48
|
- Local trust model: register rows remain honest-but-fallible workflow evidence, not a hostile-author security boundary. The gate proves that a changed, normative, owner-scoped rule line (or changed owner executable) exists for every claimed firing path; it does not prove a model run occurred, that a named executable implements the claimed enforcement (a shebang stub passes the static check), that a mangled or ambiguous ledger row was honest (those are warned, not blocked, to avoid false positives on other table shapes), author identity, or non-tampering by an authorized contributor. Independent review/challenge and the fixed checker remain the assurance case.
|
|
49
49
|
- Machine format (relocated from the `SKILL.md` firing-mechanism rule; the local evidence policy above carries the rationale): every added source-register row must carry `behavioral-evidence: RED-baseline` (any observed delta — `observed-failure: yes` requires it) or `semantic-control` (only with `observed-failure: no`), an `observed-failure: yes/no` state, and an owner-scoped `firing-path` — each declaration in its own semicolon-delimited fragment of the cell (`…prose; behavioral-evidence: …; observed-failure: …; firing-path: …`), so a key embedded mid-prose never parses as a declaration. The firing-path anchor is at least 16 characters, occurs once in the file and once in its round's added lines, and lands on a numbered/list Markdown rule with a normative action. A row that survives at HEAD must also resolve to an owner this range actually changes — an owner reverted to its base bytes by a rebase or a base-side conflict resolution leaves the changed set while its row stays behind, and the row then vouches for a change the delivered diff does not contain. There is no author-declared escape from this: a corrective rewrite that back-fills a row for a round which merged red produces the same shape, and it is a deliberate, person-adjudicated repair that can adjudicate this refusal too.
|
|
50
50
|
|
|
51
|
-
- **Round scoping — a row is judged against the round it landed in, never the accumulating range.** A row is authored against one round's diff, so reading the whole `base..HEAD` range to classify it judges the row against work it never described. That mismatch produced both directions of the same defect: an already-gated row turned red once a LATER round touched the same owner (which is what the ledger's superseded-row notes were absorbing), and a description-only round lost its routing-surface locator because an EARLIER round had edited that owner's body. The gate cuts rounds at the commits that touch the ledger
|
|
51
|
+
- **Round scoping — a row is judged against the round it landed in, never the accumulating range.** A row is authored against one round's diff, so reading the whole `base..HEAD` range to classify it judges the row against work it never described. That mismatch produced both directions of the same defect: an already-gated row turned red once a LATER round touched the same owner (which is what the ledger's superseded-row notes were absorbing), and a description-only round lost its routing-surface locator because an EARLIER round had edited that owner's body. The gate cuts rounds at the commits that touch the ledger along a first-parent line, and each round spans from the previous boundary so work commits sit in the round whose ledger append describes them. A merge that git rebuilds from its two parents is expanded into its branch's own rounds, so a merged worktree round is judged exactly as its pull request was — the same history must not partition differently after it lands; a merge git cannot rebuild (a hand resolution, a conflict) keeps a single boundary at the merge, so content that came from neither parent is never left in no round. The partition is derived from git alone — an author cannot nominate, widen, or move their own scope.
|
|
52
52
|
- Both obligations move together, in opposite directions. **Classification** narrows to the round: whether a diff is wording-only, an identifier retarget, or description-only is asked of that round's bytes, which is what makes a verdict stable once it lands. **Presence** narrows to the round too: the round that changed an owner is the round that owes the row, so owner work committed after a ledger append can no longer ride on an earlier round's row. Narrowing classification without narrowing presence would have opened exactly that laundering route.
|
|
53
53
|
- **Known exception, inherited not introduced: a renamed-away owner escapes round presence.** The round's subject set is intersected with the cumulative one, and the cumulative pass drops an owner whose package was renamed away (a row citing it would be refused for naming a SKILL.md that no longer exists). So an owner changed substantively in an earlier round, with an intervening ledger boundary that closed that round without its row, and renamed away in a later round, is demanded by neither round. Differential check on the same fixture: the pre-round-scoping gate passes it too, so this is not a regression of the narrowing — but the per-round presence contract above does not hold in this shape, and saying so is the point of this bullet. Closing it needs the evidence-file existence check to resolve at the row's round head instead of the worktree, which is its own change.
|
|
54
54
|
- The owner-level `RED-baseline` floor deliberately stays **cumulative**: it asks whether anything in the owner's whole change is left unvouched-for, and per-round would let a package self-clear on the one round that happened to be punctuation-only. Cumulative is the stricter of the two readings, so the round scoping cannot loosen it.
|
|
@@ -483,6 +483,8 @@ Round 073-receipt-bundling rows (new table so the entry renders as a table row a
|
|
|
483
483
|
|
|
484
484
|
| A before/after routing comparison may vary only ONE routing variable (one description, or one skill's indivisible routing face) for its delta to be attributable | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#一次改前/改后对照只准动**一个路由变量** | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Source-side paired single-description A/B is the observed working mechanism (borrow round; sanitized provenance in the private alias archive); no mis-attribution incident observed in this repository yet, so the clause lands as protocol item 5 with the existing four items unchanged as the paired control. |
|
|
485
485
|
|
|
486
|
+
| A reviewer-isolation gate judges only what can invoke something: the exact tool set, the tool_use scan, an empty MCP list and the pinned permission mode. Skill, command and plugin lists are vocabulary the host and its plugins own; recording them is fine, judging them by name, shape or origin is not, because nothing listed there is invocable past the pinned tools and every such predicate turns a routine CLI release or an older installed plugin into a reviewer-lane outage that proves nothing. An owner-skill binding the wrapper cannot establish costs the binding receipt, never the review | `code-review` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/code-review/scripts/test_init_policy_matrix.sh | `updated` | Owner key `code-review/SKILL.md` (contract paragraphs on the Claude wrapper, plugin binding and the binding receipt rewritten); lands in `code-review/scripts/parse_probe_result.py` (vocabulary fields become `KNOWN_VOCABULARY_INIT_FIELDS`, never judged; `REQUIRED_EMPTY_INIT_FIELDS` keeps only `mcp_servers`; the built-in name snapshots, the baseline loader, the whole-value and bare-identifier gates and the unclassifiable class are deleted), `code-review/scripts/claude_review.sh` (no acknowledgement-only baseline invocation, no help-prose probe of `--safe-mode`, `--disable-slash-commands` optional; an unverifiable installed registry or an unloadable plugin degrades to `native_skill_binding=unavailable` instead of refusing), `code-review/scripts/review_gate.py` (accepts the `unavailable` receipt as a review with `controller-profile` skill evidence and no natively reviewed skills; a wrapper attesting neither still fails closed), the oracle `init_policy_matrix.py` (policy G restated: vocabulary is data; the real 2.1.261 owner-aware init is a tolerated row beside every breach class), `test_init_policy_matrix.sh`, `test_parse_probe_result.sh`, `test_claude_review_probe.sh`, and the contract references `client-routing.md`, `manual-invocation-and-prompts.md`, `staged-review-contract.md`, `scripts/runtime-surface-verification-design.md`, `scripts/AGENTS.md`; plan in `specs/116-reviewer-vocabulary-is-data/plan.md`. Observed failure, first-hand: Claude Code 2.1.261 added the built-in skill `workflow-authoring`; the owner-aware lane classified it `unclassifiable_host_vocabulary`, reported `capability_missing`, and every review in that period fell through to another client — the third release-driven outage of the same class after `/import` and `auto-mode-setup`, and the baseline mechanism could not help because baseline skills were by design not authority. Reproduced differentially against the base parser with a success stream synthesized from the real 2.1.261 init: rejected before, accepted after, while a synthesized inherited MCP server is still refused. RED-baseline (applied mutations, each in a disposable copy, control green at 270 cases / 0 mismatches): dropping the MCP emptiness requirement flips 40 rows; reading a populated vocabulary list as a breach flips 76; reading it as schema drift flips 72; the four pre-existing mutants stay detected. Wrapper-level evidence through the fake CLI: a plugin manifest declaring hooks, an older registry whose package hash mismatches the profile, and a missing registry each produce a challenge verdict with `unavailable` and no `--plugin-dir`; every vocabulary fixture that used to refuse (empty enumeration, omitted owner, host built-ins, foreign entries, duplicates) now attests `established`; a CLI without `--disable-slash-commands` and a CLI whose `--safe-mode` prose contradicts itself both still run; a CLI without `--safe-mode` still refuses. Supersedes the four vocabulary rows of the 015 round above: their mechanism is removed rather than extended, per same-class convergence by deletion, and their transferable lesson survives in inverted form here — the predicate was never on a property the control owns, so the class was discharged by deleting the predicate. |
|
|
487
|
+
|
|
486
488
|
| Cross-skill / cross-reference routing pointers in body text must carry the routing quadruple (trigger / scope / output / return point); a bare "refer to X if useful" pointer is never a landing shape | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: semantic-control; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/description-authoring.md#routing pointer in body text must carry the routing quadruple | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Borrowed from an adopting skill pack where unbounded pointers were the dominant dead-routing shape (two adopters verified in source; sanitized provenance in the private alias archive); landed in the routing-surface authoring reference because the entrypoint is size-ratcheted level — the reference is the required pre-edit reading for routing-surface work, and eval-routing.md's silent-skip row points back at it; the description-side Skip-when idiom already satisfies the quadruple and is named as the unchanged control. |
|
|
487
489
|
| Review-finding fixes are held un-applied until the full review+challenge chain has run on the frozen candidate: under the 1+1 budget apply-now is never fundable after the review round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/skill-extraction-workflow/references/dual-track-review-gate.md#accumulate every fix unapplied, run the challenge on the frozen | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/dual-track-review-gate.md (enumeration item 1 + cadence Round 2). Observed failure: a prior round applied its review fixes before the challenge; the tracked chain broke (challenge binds to the round-1 candidate), the round lost its double-receipt terminal, and closure required a user-granted continuation chain. The prior wording stated the rule only as a fundability conditional whose arithmetic the agent under pressure never ran; the operative unconditional form (hold all fixes; challenge on the frozen candidate; land the batch after the chain) is now explicit at both firing points. RED baseline: the recorded chain-break incident is the without-change failure; the with-change compliance surface is the explicit hold rule at the enumeration walked when a round returns findings. |
|
|
488
490
|
| A frozen eval case is sacred: deleting or re-scoping a bank task or golden trace requires a same-round `case-retired:`/`case-rescoped:` register adjudication row, and average improvement never offsets a frozen-case loss | `skill-extraction-workflow` | result-class: stable-success; behavioral-evidence: RED-baseline; observed-failure: no; firing-path: file:skills/skill-extraction-workflow/references/eval-routing.md#平均改善不得抵消单条冻结案例的失守 | `updated` | Owner key `skill-extraction-workflow/SKILL.md`; lands in references/eval-routing.md (冻结案例神圣 bullet), scripts/test_frozen_case_sanctity.sh (fast lane, registered), docs/f4-skill-effectiveness-harness.md (pointer line). Rule semantics: a previously-passing frozen case that degrades — including to unsure/INCONCLUSIVE — is a regression, and its sacredness attaches per case, so no aggregate improvement offsets it; the mechanism's source-verification record lives in the round's private archive, and transferred evidence does not exempt the behavioral row. RED baseline (replayed, throwaway clone at the round base): deleting the non-pinned bank case ctrl-unit-test passed the pre-change surface silently (test_routing_bank_integrity.sh exit 0) and reddens the new gate (exit 1 naming the id and the required adjudication row); re-scope and golden-trace-deletion mutants red for the right reason; adjudicated-deletion and untouched-tree control legs green; no-base and unresolvable-base legs print the explicit skip token. |
|
|
@@ -592,3 +594,40 @@ Round 073-receipt-bundling rows (new table so the entry renders as a table row a
|
|
|
592
594
|
|
|
593
595
|
Supersede note (round 114, ledger correction with no rule change): the row above beginning "Shared-tree guidance states a usage-census conclusion qualitatively" carries, in its evidence cell, an account of which review round and which challenge produced which finding. That is conversation-level process narrative on a shared surface, and the register is append-only, so the row stays byte-identical and is corrected here by pointer rather than edited. The obligation it records, restated at artifact level: `attention-budget-ratchet.md` carries the read-shape conclusion in qualitative form with no ratio or count; the measured census figures live only in the private charter; the clean-only-oracle clause has exactly one carrier in `dual-track-review-gate.md`, recorded as row 74 of `specs/113-extraction-entry-slim/obligation-preservation.md`. The superseded row's own behavioral-evidence declaration and firing-path anchor are unaffected. This is a note rather than a table row because it changes no rule and therefore has no owner-scoped anchor of its own to declare.
|
|
594
596
|
| An instrument whose stated definitions ARE its contract owes a case per definition, and the counting rule it documents must be the rule its figures were produced by: a regular-expression alternation matches leftmost and non-overlapping, so a listed phrase absorbs the words inside it and a comment promising independent counts describes a different ruler than the one that ran | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_entrypoint_form_census.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Observed failure: two independent reviewer lenses on one candidate both read the census script's comment as promising that a phrase and the words inside it are counted separately, and no fixture exercised the overlap, so the documented rule and the reported totals were different rulers while the suite stayed green. The figures are unchanged by the correction because the behaviour was always leftmost non-overlapping and only the comment was wrong: the same run reports the same totals before and after. Two cases added, each with an applied mutation: a phrase-absorption case pins `must not` at one token, and an empty-Core-Rules case pins the false-empty guard -- disabling that guard turns exactly that case red with no other case failing, restored green. The classification artifact gains a stable per-rule identifier so each verdict maps to a rule without the generated JSON. Supporting evidence: `skill-extraction-workflow/scripts/entrypoint_form_census.py`, `skill-extraction-workflow/scripts/test_entrypoint_form_census.sh`, `specs/114-entry-form-and-routing-baselines/form-classification.md`. |
|
|
597
|
+
| A real-checkout regression accepts every documented terminal result of the candidate printer: a bounded hash, no change, or the packet-ceiling refusal, so a large merge-parent diff does not make the self-test reject correct gate behavior | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. RED baseline: a clean merge checkout whose parent diff exceeded the one-packet ceiling returned the documented `review packet exceeds 200000 bytes` refusal while the real-checkout assertion accepted only hash or no-change and failed. The same topology passes after the assertion accepts that exact refusal. Gate behavior and the packet ceiling are unchanged. Supporting evidence: `skill-extraction-workflow/scripts/test_review_ledger_binding.sh`. |
|
|
598
|
+
| A routing surface that omits a territory the owner's own body, its reference, or a sibling's Skip leg already assigns to it is a hole rather than a collision: the graders answer `none` or fall to the nearest-looking rival, and no amount of disambiguation between claimants fixes an utterance class nobody claims | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#you must use the R&D standards checklist in Workflow; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#product-rd-workflow: p3-resume-refactor moved 0/10 to 10/10| `updated` | Owner key `product-rd-workflow/SKILL.md`. Two frozen cases, each measured on paired trees differing only in the routing surface. The resume/continue utterance class was absent from a trigger list that already carried the whole redo family, and nine of ten gradings answered `none`. Separately, a stack with no architecture sibling had its service-boundary and data-ownership decisions assigned here by the dev skill's body and by this workflow's own reference, while the routing surface never said so and the two stacks that DO have architecture siblings claimed those words -- so the nearest-looking rival won 13/20 of the time. This entrypoint is at severe size debt with a zero-growth byte budget, so the two triggers were paid for rather than appended: sixteen separators compressed, one duplicate implement-phase phrasing dropped, and two sentences deleted from an unrelated standards bullet whose obligations the owning checklist reference already carries verbatim -- authority statement plus sync-gate, the testing-standard child doc with its layer and CI-gate coverage, and the execution-layer-versus-governing-Spec ruling. Net file size fell by 305 bytes. Both new triggers carry an English handle mirroring the redo family's, because dropping it would have left English-phrased requests of that class with no handle at all. The deletions are recorded as measured on the owner's whole case set rather than argued: every frozen case expecting this owner was re-run on paired trees because the edit reformatted the entire trigger list, not one token. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
599
|
+
| An owner whose territory is claimed only in the language the utterances are not written in is unreachable by keyword match, and scattered rivals at one hit each identify the defect as the owner's rather than a competitor's | `platform-observability` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-observability/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#platform-observability: p3-log-plus-test moved 2/10 to 10/10| `updated` | Owner key `platform-observability/SKILL.md`. Eight of seventeen gradings refused the compound logging/trace utterance outright and the remainder split across four rivals at one hit each. A second case had a sibling routing the production-state question here on both of that sibling's surfaces while this owner claimed it on neither of its own. Both moved to 10/10 on paired arms. The description was at 799 of 800 characters, so the addition required trimming a boilerplate sentence whose full obligation the body already carries verbatim. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
600
|
+
| A diagnosis owner that enumerates failure vocabulary but not degradation vocabulary reaches a slow-endpoint request only half the time, and the fix must be anchored to the degradation rather than to optimization, which is a delivery | `defect-diagnosis` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/defect-diagnosis/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#defect-diagnosis: p3-perf-plus-regression moved 5/10 to 10/10 | `updated` | Owner key `defect-diagnosis/SKILL.md`. The trigger list carried bug, 报错, test 挂了 and 线上问题 and nothing for an endpoint that got slower, so a request pairing that with a regression test reached the owner in five of ten gradings. The anchored form was chosen over a bare performance-optimization token deliberately: the bare token would have absorbed performance work that belongs to a multi-stage delivery owner. Paired arms, 10 replicas each, moved it to 10/10. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
601
|
+
| A client-stack owner that lists feature surfaces without the host toolchain refuses the toolchain question confidently, which is worse than refusing it with a clarify flag because nothing downstream signals that a route was missed | `miniapp-product-dev` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/miniapp-product-dev/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#miniapp-product-dev: skip-miniapp-build moved 6/10 to 10/10 | `updated` | Owner key `miniapp-product-dev/SKILL.md`. The description enumerated pages, state, auth, sharing and review but not the host developer tool's build and on-device debug path, so four of ten gradings refused the build question and one of those refusals carried high confidence with no clarify flag. Paired arms moved it to 10/10. The pinned localized-refactor literal in the contract-anchor suite survives the edit unchanged. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
602
|
+
| A bare trigger token that names an activity also names its materials, and when five Skip legs point outward and none points back, the owner absorbs requests for the materials its own body says a sibling produces | `grill-me` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/grill-me/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#grill-me: ab-c5 moved 8/10 to 10/10 | `updated` | Owner key `grill-me/SKILL.md`. Four surfaces already assigned the question-pool deliverable to the sibling that produces it, including that sibling's own body table row naming the exact two fields the utterance asks for, and the always-on entry-routing layer. Only this description claimed the bare activity token. Anchoring it to the one-question-at-a-time form and adding the reciprocal Skip leg moved the materials case from 8/10 to 10/10 while the case that genuinely belongs here stayed 10/10 on both arms -- the narrowing was measured against the case it could have cost, not only against the case it was meant to fix. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
603
|
+
| A deliverable field named twice in a body but absent from the trigger list is not owned as far as routing is concerned, and the utterance that asks for both halves splits | `requirement-scope` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/requirement-scope/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#requirement-scope: ab-d6 moved 8/10 to 10/10| `updated` | Owner key `requirement-scope/SKILL.md`. The body names the compatibility and rollback boundary as a deliverable field in both its field table and its procedure, while the description carried only the version-slice half of the same utterance. At three replicas this case had been read as a collision with a release owner; at ten that rival appeared zero times and the only deviation was the lifecycle coordinator, which is why the fix is an addition here rather than a reciprocal Skip leg there. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
604
|
+
| A Skip list that enumerates six sibling destinations and omits the lifecycle coordinator leaves every request that continues past this owner's deliverable stranded on it | `requirement-doc-writer` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/requirement-doc-writer/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#requirement-doc-writer: p3-spec-then-tc claim added, case unmoved at 7/10| `updated` | Owner key `requirement-doc-writer/SKILL.md`. The body already routes multi-stage delivery and implementation or release planning to the coordinator; the description's Skip legs named six destinations and not that one. The alternative fix -- teaching the coordinator to claim sequential compound requests -- was rejected because two other frozen cases expect the leading-deliverable owner rather than the coordinator, so it would have bought one case with two. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
605
|
+
| A territory claimed inside a deliverable clause rather than in the trigger list is claimed where routing weight does not reach it; moving the same words changes the route without claiming anything new | `requirement-baseline` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/requirement-baseline/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#requirement-baseline: ab-b5 moved 14/20 to 20/20 | `updated` | Owner key `requirement-baseline/SKILL.md`. The code-evidence claim sat in the deliverable clause and the utterance asking for exactly that reached the owner in 14 of 20 gradings, with every deviation a refusal rather than a rival. This case is also the round's caution about its own decision rule: one ten-replica reading put it at 50 percent and the next twenty put it at 85 percent on the same candidate, so the threshold that separates a stable failure from a marginal case must be read off pooled observations. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
606
|
+
| A carve-out that pushes an utterance class away from one owner is only half a route: when the owner it points to never claims that class, the class has no home and the carve-out silently becomes a hole | `platform-release-engineering` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/platform-release-engineering/SKILL.md#description; bank-evidence: file:eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md#platform-release-engineering: release-watch moved 18/20 to 20/20 | `updated` | Owner key `platform-release-engineering/SKILL.md`. Across the whole catalog only one description mentioned the on-call vocabulary, and it did so with an explicit carve-out excluding release watch and rollback; this owner, the destination of that carve-out, claimed none of it. Paired arms at twenty replicas moved the case from 18/20 to 20/20, and the one control-arm deviation had been a must-not-route-to violation rather than an ordinary miss. The pinned production-release literal in the contract-anchor suite survives the edit unchanged. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
607
|
+
| A measurement artifact produced below the resolution its own action rule requires reads as a findings list, and its consumers act on it; the floor has to be enforced by the instrument that writes the artifact, not by the prose that describes it | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The measurement protocol already required ten valid observations before a description edit, while the full-bank baseline its readers open is produced at three replicas -- a resolution the same protocol forbids acting on. In one round that ruler produced a wrong case-level reading in both directions: five cases it called failing are perfect at ten replicas, six drafted edits rested on them, one deviation supported an argument about a stale frozen expectation that had to be withdrawn, one neighbour was scored as a fresh regression that the paired control shows on the unedited tree, and three full-bank runs on near-identical candidates returned almost disjoint newly-failed sets. The runner now writes an explicit resolution field and prints a screening banner, the reference states the same floor plus the pooled-observation and candidate-list rules, and the new suite pins the doc and the executable to one number with three applied mutations each turning it red on its own assertion. Supporting evidence: `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
608
|
+
| A resolution floor stated over VALID observations but computed from the requested sample size is not the floor it claims: a run that asked for ten replicas reports itself actionable while a timeout or an unparsable answer leaves a case short, so the gate passes exactly the evidence it exists to refuse | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Independent review raised this against the first cut, and the round that landed it had already hit the shape it misses: a fourteen-task run at ten replicas returned six grader errors and left two tasks at seven valid observations, and the re-run was owed by hand rather than signalled by the report. The field now derives from the weakest task's valid-observation count -- the weakest governs because a per-case edit is licensed per case -- and the report exposes that count so a consumer is not left reading the request. Reverting the computation to the requested replica count turns the new case red on its own assertion and restoring it returns green. The regression fixture drives a grader that fails one utterance persistently, because the runner already retries a single unparsable answer as a sampling accident and a fixture that fails once is silently repaired -- a detail found by watching a ten-replica fixture spend eleven calls. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`. |
|
|
609
|
+
|
|
610
|
+
Supersede note (round 115, ledger correction with no rule change): the eleven rows appended by this round each name one owner and the case that owner's description was edited for, in a form that reads as though the measured move were attributed to that description. Independent review held the record to the single-variable rule this same round lands — a comparison that moved several descriptions at once cannot be attributed to any one of them — and the rows stay byte-identical because the register is append-only, so the correction is by pointer. Restated at the level the evidence supports: the control is the branch base and the treatment is the landing candidate carrying all eleven edits, so every per-owner move in those rows is a PACKAGE result with a per-case interpretation attached. The interpretation is supported, not established, by two facts recorded in `eval/evidence/routing-115-underclaim-fix-2026-09-03/paired-measurements.tsv`: each case's expected owner has exactly one changed description, and no other changed owner appears anywhere in that case's observed verdict distribution on either arm. Exactly one of the eleven carries the isolation the protocol asks for: `ab-c5` measured 6/9 on a four-edit tree (catalog 31c9a686…) and 10/10 on that tree plus only the grill-me description (catalog 71a6793d…). The other ten were not split into per-description A/B trees, and that is this round's residual rather than a claim it makes. The rows' behavioral-evidence declarations, firing paths and bank-evidence locators are unaffected.
|
|
611
|
+
|
|
612
|
+
Supersede note (round 115, second ledger correction with no rule change): the `product-rd-workflow` row appended by this round records that the two sentences deleted from its standards bullet had their obligations carried by the owning checklist reference. Adversarial review checked that mapping clause by clause and it was incomplete in two places: the reference required an authority statement for a multi-doc family but not that a cross-stack product Spec live in exactly one authority surface with execution slices linking back rather than redefining its goals, and it required a testing standard child doc but not that stack documents may not replace the shared test-layer and CI-gate policy that standard owns. Both obligations are now stated in `product-rd-workflow/references/rd-standards-doc-family-checklist.md` items 5 and 6, where the placement rule puts detail, rather than restored to a severe-debt entrypoint under a zero-growth byte budget. The row stays byte-identical because the register is append-only. The lesson the round records against itself: a zero-loss map asserted at paragraph granularity passed, and the same map checked at clause granularity did not.
|
|
613
|
+
| A trigger removed on a synonym argument is removed on the author's reading, not on evidence: when no frozen case exercises the removed token, the bank cannot refute the argument and the narrowing lands unmeasured | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md#must not** redefine the product goals it owns; bank-evidence: downscoped:R115-TRIGGER-RESTORE-NO-FROZEN-CASE | `updated` | Owner key `product-rd-workflow/SKILL.md`. Paying this entrypoint's zero-growth byte budget, the round dropped one trigger as a synonym of a surviving one. Adversarial review supplied the counterexample the bank could not -- a request that names the transition without the approval wording -- and the trigger is restored inside the same budget. The same review checked the round's zero-loss map for two sentences deleted from a standards bullet and found it incomplete at clause granularity: the one-authority-surface rule and the prohibition on stack documents replacing shared test-layer and CI-gate policy are restored to `product-rd-workflow/references/rd-standards-doc-family-checklist.md` items 5 and 6, where the placement rule puts detail. Supporting evidence: `product-rd-workflow/references/rd-standards-doc-family-checklist.md`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
614
|
+
| A resolution verdict published once per report cannot license anything per case: a subset run that clears the floor for the cases it graded reads, to a consumer holding only the report-level field, as licence for an edit to a case the run never measured | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Adversarial review found the field this round had just landed to be report-wide while the obligation it enforces is per case. Each result now carries its own `actionable` and `valid_observations`, the report-level field is the conjunction over the cases actually graded, and a scope note says so in the report itself. Reverting to the report-wide computation turns the new case red on its own assertion and restoring it returns green. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`. |
|
|
615
|
+
| A fixture that cannot separate the property under test from the defect it guards against turns an applied mutation into theatre: the suite goes red on an unrelated assertion, the author records a proof, and the property was never pinned | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. The resolution suite carried one case, and with one case a per-case verdict and a report-wide conjunction produce identical output, so the mutation this round ran to prove the per-case field went red on a different assertion and proved nothing. The fixture now carries two cases and degrades only one; substituting the conjunction for the per-case field fails on the assertion that the healthy case stays actionable, which is the discrimination the earlier fixture could not make. The same review found the round's own must-not claim read from one probe while a second run on the same tree carried the single forbidden verdict; the record now states the pooled counts on both arms. Supporting evidence: `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
616
|
+
| A property a reviewer cannot reach from the bounded packet is unverified even when the code is correct: the degenerate shapes that would expose an absent verdict list live in a construction site the packet excludes, so the check has to be moved into the suite rather than argued in the response | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Review could not establish from its packet that every result carries a verdict list, because the loop that builds one is outside the bounded diff; reading the source shows a single append site that always sets it, which answers the question for the author and for nobody else. The suite now exercises the two shapes that would expose an absent list -- a single-replica run, and a task whose every replica fails -- asserting neither crashes and neither reports itself actionable. Capturing the wholly-failed run's documented exit 3 required taking the status in the same command that produces it: under `set -e` the suite dies before the assignment, which is how the first version of this case reported nothing at all. Supporting evidence: `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
617
|
+
| A token that names a lifecycle phase does not only claim requests about that phase; it recolours the owner's whole description for a model reading the catalog, so the same token that catches one request class can push an unrelated frozen case toward a rival — the trade has to be measured on both, not argued from what the token says | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/SKILL.md#description; bank-evidence: downscoped:R115-TRIGGER-RESTORE-NO-FROZEN-CASE | `updated` | Owner key `product-rd-workflow/SKILL.md`. Restoring a bare implementation-phase token after review moved a frozen spec-writing case from 10/10 on the prior candidate to 13/20, with the requirement-document owner as the rival; the two trees differed by that token alone, which is the single-variable A/B the protocol asks for and the round otherwise lacked. Merging the token with its approval-wording neighbour recovers the case to 15/20 — level with its 14/20 control — while the review counterexample still routes 20/20 and the approval-wording case stays 20/20. A further attempt to lift the case by sharpening the rival's Skip leg reached 20/20 there but moved a different frozen case from 19/20 to 15/20, and was withdrawn under the rule that average improvement may not offset a single frozen case; it also rested on the rate rather than on a surface asymmetry. The downscope declaration is restated for the merged form in the round's spec. Supporting evidence: `specs/115-routing-underclaim-fix/downscope.md`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
618
|
+
| A zero-loss map that names the artifact but not its owner has not preserved the obligation: routing a template to a skill is not the same as that skill owning the standard and the policy it carries, and a reader who holds the template can approve the policy without ever satisfying the original ownership clause | `product-rd-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: file:skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md#owned by `testing-strategy`** — not by whichever document holds the template | `updated` | Owner key `product-rd-workflow/SKILL.md`. Adversarial review checked the round's second restoration of the deleted standards sentences and found a third gap: the reference said the testing-standard template routes to the testing owner, while the deleted entrypoint sentence had said the standard is OWNED by it. The reference now states the ownership of the standard and of the shared test-layer and CI-gate policy explicitly. The same review corrected four record errors the round had introduced while consolidating its evidence -- settled-tree values doubled by intermediate-tree rows in one table, a causal conclusion drawn about a must-not observation that the evidence cannot support either way, a probe utterance absent from the packet, and no per-run selection manifest against which a dropped case would show. Each is fixed in the record; none changes a routing surface. Supporting evidence: `product-rd-workflow/references/rd-standards-doc-family-checklist.md`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
619
|
+
|
|
620
|
+
Supersede note (round 115, third ledger correction with no rule change): the first supersede note above restates this round's per-owner rows as a package result supported by two facts, one of which the committed tables contradict — it says no other changed owner appears in any claimed case's observed distribution, while `p3-spec-then-tc` selects the edited `requirement-doc-writer` on both arms, `p3-log-plus-test` selects the edited `product-rd-workflow` on the control arm, and `ab-c5`'s only rival is the edited `grill-me`. The support that survives is the weaker fact alone: each claimed case's expected owner has exactly one changed description. The single isolated comparison (`ab-c5`, 6/9 → 10/10 across two trees differing only by the grill-me description) stands. Rows and the earlier note stay byte-identical because the register is append-only.
|
|
621
|
+
| A verdict that parses is not yet an observation: an answer naming a skill the catalog does not carry says nothing about the route, and a floor that counts it as usable can be satisfied by ten well-formed non-answers | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Adversarial review could not see from its packet whether the resolution counter validated a parsed selection, and reading the source showed it did not: any status other than ERROR counted, so a parseable verdict naming a non-catalog skill was a FAIL that still fed the floor. Such a verdict is now an ERROR -- absence of evidence, not evidence against the route -- and a fixture whose grader returns a well-formed answer naming no catalog skill must leave every case at zero valid observations and non-actionable; removing the validation turns that case red on its own assertion. All 4,905 non-error verdicts in the round's archived reports name a catalog skill or none, so no landed number moves. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
622
|
+
| Two lists that must agree and are built by two transformations will drift; the set of names a verdict may select has to be the same filtered list the prompt was built from, or a skill the grader was never shown becomes a valid answer | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Review noted that the selectable-name allow-list added earlier in this round was built from every directory carrying a SKILL.md while the prompt catalog was built by a separate transformation that drops entries without a usable description, so the two could disagree on exactly the entries the prompt omits. Both now derive from one filtered list. The fixture adds a skill whose description is empty and a grader that selects it; the verdict must be an ERROR, and rebuilding the allow-list from the unfiltered directory turns that case red on its own assertion. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
623
|
+
| A rewrite that keeps behaviour but raises the language floor of a shared tool is a change the tool's other hosts pay for, and the round that makes it owes either the floor or the restoration | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/eval-routing-bank.rb | `updated` | Owner key `skill-extraction-workflow/SKILL.md`. Review noted that the catalog build this round rewrote had moved from map plus compact to filter_map, which needs Ruby 2.7; the runner carried no such call at the branch base, the repository states no Ruby floor, and the round had not measured which hosts run it. The touched lines are restored to map plus compact so the round raises no floor the base did not already have; the gate script's own pre-existing uses are outside this round and unchanged. The resolution suite exercises the rebuilt catalog path on every run. Supporting evidence: `skill-extraction-workflow/scripts/eval-routing-bank.rb`, `eval/evidence/routing-115-underclaim-fix-2026-09-03/README.md`. |
|
|
624
|
+
|
|
625
|
+
Supersede note (round 115, fourth ledger correction with no rule change): the third note above names as the surviving support that each claimed case's expected owner has exactly one changed description. Adversarial review refuted it with `ab-c5`, whose expected owner `requirement-intent` is untouched and whose edit is on the rival `grill-me`. The support that survives is: for each claimed case exactly one description was edited for it, the expected owner's in ten cases and the rival's in one. The evidence directory now carries the frozen bank records for every measured case so expected owner, acceptable alternatives and must-not guards are checkable against the untouched bank rather than against this record. Rows and earlier notes stay byte-identical.
|
|
626
|
+
|
|
627
|
+
Supersede note (round 115, fifth ledger correction with no rule change): the routing-hole row above says the nearest-looking rival won 13/20 of the time for the Node architecture case. That misreads the base run: 13/20 is the count of replicas that selected the expected owner; the rival `go-microservice-architecture` took one replica and `none` took six (`paired-measurements.tsv`, run ctrl-J; the 10-replica base run split 6 expected / 4 none). The dominant failure was refusal, not the rival — which is the row's own thesis, so the lesson stands and only the distribution is corrected.
|
|
628
|
+
|
|
629
|
+
Supersede note (round 115, sixth ledger correction with no rule change, two rows): (a) the language-reachability row above says the logging/trace utterance's non-refusing gradings split across four rivals at one hit each. The pre-A and pre-A2 base runs total seventeen gradings: eight refusals, six selections of the expected owner `platform-observability`, and three rivals at one hit each (`product-rd-workflow`, `python-service-dev`, `nodejs-service-dev`). Scattered single-hit rivals remain the row's point; the count is corrected. (b) The release-watch row says the one control-arm deviation had been a must-not-route-to violation. Two base-tree control runs exist: ctrl-J (18/20, the run the row's move is measured from) deviated twice to `release-coordination`, an ordinary miss; ctrl-H (19/20) deviated once to `platform-observability`, the forbidden owner. The violation is real but belongs to the other control run.
|
|
630
|
+
|
|
631
|
+
Supersede note (round 115, seventh ledger correction with no rule change): the observation-validity row above says all 4,905 non-error verdicts in the round's archived reports name a catalog skill or `none`. That total is counted over the archived runner reports in the maintainer's scratch directory, which are not committed and cannot be recounted from this repository. What the repository does carry is `replica-verdicts.tsv`, whose 2430 non-error verdicts all resolve to a catalog skill or `none`; the archive-wide figure stands as the author's count, not as committed evidence.
|
|
632
|
+
| A round partition that depends on which ref is judged moves verdicts after they land: a branch judged per ledger commit as a pull request collapsed to one round once merged, so a row valid at pull-request time (a routing-surface `#description` anchor in a commit that changed only the description) was refused on every post-merge evaluation with nothing about it changed; a merge git rebuilds from its two parents must be expanded into the branch's own rounds so the same history partitions identically before and after it lands, and a merge git cannot rebuild keeps one boundary so content from neither parent is never left in no round | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/impact-chain-gate.rb, its round-scoping fixtures, the verdict differential's named divergences, references/external-practice-controls.md and the .github/workflows/ci.yml checkout comments). Observed failure: the integration branch's promotion pull request reported `impact_chain_firing_path_missing` for a row that was green on its own pull request (15 rounds on the branch head, one after the merge), and the integration branch's push build had been red since that merge. RED baseline: round scoping 8 now runs the branch view and the merged view on one fixture and asserts them equal; on the previous gate it fails with `expected rc=1 got rc=0`, and the real promotion shape (integration head against the target) goes from rc=1 to rc=0 with no other change. Verdict differential: 64 integration points, six newly refused, each named by sha with `impact_chain_gate_missing` — all merges from before CI checked out the branch head, whose branches carry owner work outside the round that declares it; none newly accepted. Round scoping 13 pins the observed shape (body round then description-only round, merged) green and equal to its branch view; round scoping 14 pins that a merge whose tree is not the automatic merge keeps a single boundary. Refines the row above beginning "确定性闸对历史形态有前提": the checkout ref binding stays, and the partition no longer depends on it. |
|
|
633
|
+
| A landing chain that looks for a round's review evidence only inside that round's own checkout can never bind a round that merged without its ledger, however honestly the same bytes are reviewed later: evidence is a validator-accepted closeout whose candidate hash equals the round's packet, so the chain reads the landing tree's committed evidence, and a later review of exactly those bytes, landed as a round of its own, binds the earlier round — a closeout for any other digest binds nothing, wherever it sits | `skill-extraction-workflow` | result-class: failure; behavioral-evidence: RED-baseline; observed-failure: yes; firing-path: command:skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh | `updated` | Owner key `skill-extraction-workflow/SKILL.md` (entrypoint unchanged; lands in scripts/review_ledger_binding.py and its suite). Observed failure: the same promotion pull request's chain refused at a two-file round that had merged with the binder red (`no accepted review evidence binds the landing candidate`), and no forward path existed because the rebind enumerated evidence from the round's detached checkout. RED baseline: the chain case "a validator-accepted closeout for the round's own candidate, committed on the integration branch after the merge, binds it through the chain" fails on the previous binder and passes on this one; the companion case with a closeout for a different digest is refused on both. The round's packet is still frozen from its own checkout at its own base with the landing tree's controller, and its excludes still come from its own added receipts, so a later ledger's absence in the round checkout leaves the round's hash unchanged. The retrospective review of that round's bytes is committed in this round's evidence directory (its retro-round folder) and binds its candidate hash. |
|
|
@@ -42,6 +42,14 @@ require "shellwords"
|
|
|
42
42
|
require "timeout"
|
|
43
43
|
require "digest"
|
|
44
44
|
|
|
45
|
+
# Resolution floor for ACTING on a measured case. A report taken below it
|
|
46
|
+
# LOCATES candidates; it does not license a description edit, because at three
|
|
47
|
+
# replicas a case that routes correctly 80-90% of the time reads as failing and
|
|
48
|
+
# a one-off deviation reads as a finding. `references/eval-routing.md` owns the
|
|
49
|
+
# rule and states the same number; test_eval_routing_bank_resolution.sh pins the
|
|
50
|
+
# two sides together so they cannot drift apart.
|
|
51
|
+
ACTION_RESOLUTION_MIN_REPLICAS = 10
|
|
52
|
+
|
|
45
53
|
def arg(flag, default = nil)
|
|
46
54
|
i = ARGV.index(flag)
|
|
47
55
|
i ? ARGV[i + 1] : default
|
|
@@ -183,15 +191,25 @@ desc_changed = changed_files.any? { |f| f.end_with?("/SKILL.md") }
|
|
|
183
191
|
co_change = bank_changed && desc_changed
|
|
184
192
|
|
|
185
193
|
# --- build skill routing surface (the same descriptions the agent routes on) -
|
|
186
|
-
|
|
194
|
+
# One filtered list feeds BOTH the prompt and the set of selectable names, so a
|
|
195
|
+
# skill the prompt never offered (no frontmatter, empty description) cannot be a
|
|
196
|
+
# valid selection: review noted the two were built by separate transformations
|
|
197
|
+
# whose filtering could drift apart, and a name allowed but never shown is exactly
|
|
198
|
+
# the shape that drift would let a verdict claim.
|
|
199
|
+
# map + compact rather than filter_map: the runner has to work on the oldest
|
|
200
|
+
# Ruby a host ships (macOS system Ruby is 2.6), and filter_map is 2.7+.
|
|
201
|
+
catalog_entries = Dir[File.join(root, "skills", "*", "SKILL.md")].sort.map do |path|
|
|
187
202
|
name = File.basename(File.dirname(path))
|
|
188
203
|
m = File.read(path).match(/\A---\s*\n(.*?)\n---\s*\n/m)
|
|
189
204
|
next unless m
|
|
190
205
|
desc = (YAML.safe_load(m[1]) rescue {})["description"].to_s.strip
|
|
191
206
|
next if desc.empty?
|
|
192
207
|
desc = desc[0, desc_budget] if desc_budget && desc.length > desc_budget
|
|
193
|
-
"### #{name}\n#{desc}"
|
|
194
|
-
end.compact
|
|
208
|
+
[name, "### #{name}\n#{desc}"]
|
|
209
|
+
end.compact
|
|
210
|
+
catalog = catalog_entries.map(&:last).join("\n\n")
|
|
211
|
+
# The set of names a verdict may legitimately select; "none" is accepted separately.
|
|
212
|
+
catalog_names = catalog_entries.map(&:first)
|
|
195
213
|
|
|
196
214
|
# --- optional always-on entry-routing layer -----------------------------------
|
|
197
215
|
# The catalog above is the description-only surface. Hosts ALSO inject
|
|
@@ -401,6 +419,16 @@ tasks.each do |t|
|
|
|
401
419
|
break
|
|
402
420
|
end
|
|
403
421
|
selected = parsed && parsed["selected_skill"]
|
|
422
|
+
# A parseable answer is not yet a usable observation. The prompt's contract is
|
|
423
|
+
# an exact catalog name or "none"; anything else -- a missing key, a name the
|
|
424
|
+
# catalog does not carry, a non-string -- is grader output that says nothing
|
|
425
|
+
# about routing, and counting it as a verdict would let it feed the per-case
|
|
426
|
+
# resolution floor exactly as a well-formed one does. It is an ERROR, not a
|
|
427
|
+
# FAIL: a FAIL is evidence against the route, this is absence of evidence.
|
|
428
|
+
if parsed && error.nil? && !(selected == "none" || catalog_names.include?(selected))
|
|
429
|
+
error = "invalid_selection: #{selected.inspect[0, 80]}"
|
|
430
|
+
parsed = nil
|
|
431
|
+
end
|
|
404
432
|
clarify = parsed && parsed["clarify"] == true
|
|
405
433
|
confidence = parsed && parsed["confidence"]
|
|
406
434
|
v_status =
|
|
@@ -519,6 +547,30 @@ if baseline_path && File.file?(baseline_path)
|
|
|
519
547
|
end
|
|
520
548
|
end
|
|
521
549
|
|
|
550
|
+
# The floor is on VALID observations, not on the requested replica count: a run
|
|
551
|
+
# asked for ten replicas can come back with seven usable verdicts once a grader
|
|
552
|
+
# times out or returns unparsable output, and a report that called itself
|
|
553
|
+
# actionable on the request alone would license an edit the evidence cannot
|
|
554
|
+
# support. Observed in this repository: a fourteen-task run at --replicas 10
|
|
555
|
+
# returned six grader errors and left two tasks at seven valid observations.
|
|
556
|
+
#
|
|
557
|
+
# The verdict is PER CASE, because an edit is licensed per case. A report-wide
|
|
558
|
+
# flag alone is unsound in the other direction: a subset run over case A can be
|
|
559
|
+
# actionable while saying nothing about case B, and a consumer reading only the
|
|
560
|
+
# top-level boolean would take it as licence for an edit to B. Each result
|
|
561
|
+
# therefore carries its own `actionable`, and the report-level field is the
|
|
562
|
+
# conjunction over the cases the run actually measured -- true only when every
|
|
563
|
+
# measured case clears the floor, and never a statement about a case absent from
|
|
564
|
+
# `results`.
|
|
565
|
+
results.each do |r|
|
|
566
|
+
valid = r[:verdicts].count { |v| v[:status] != "ERROR" }
|
|
567
|
+
r[:valid_observations] = valid
|
|
568
|
+
r[:actionable] = replicas >= ACTION_RESOLUTION_MIN_REPLICAS &&
|
|
569
|
+
valid >= ACTION_RESOLUTION_MIN_REPLICAS
|
|
570
|
+
end
|
|
571
|
+
min_valid_observations = results.map { |r| r[:valid_observations] }.min.to_i
|
|
572
|
+
action_resolution = !results.empty? && results.all? { |r| r[:actionable] }
|
|
573
|
+
|
|
522
574
|
report = {
|
|
523
575
|
model: model, tasks: results.size, pass: passes, fail: fails.size, error: errors.size,
|
|
524
576
|
replicas: replicas, verdicts: all_observed.size,
|
|
@@ -526,6 +578,10 @@ report = {
|
|
|
526
578
|
clarify_count: clarify_count, low_confidence_count: low_conf_count,
|
|
527
579
|
replica_agreement: (replicas >= 2 ? { agree: agreement_agree, measured: agreement_measured } : nil),
|
|
528
580
|
desc_budget_chars: desc_budget, routing_surface: routing_surface,
|
|
581
|
+
action_resolution: action_resolution,
|
|
582
|
+
action_resolution_scope: "cases measured by this run only; see each result's actionable field",
|
|
583
|
+
action_resolution_min_replicas: ACTION_RESOLUTION_MIN_REPLICAS,
|
|
584
|
+
min_valid_observations: min_valid_observations,
|
|
529
585
|
co_change_bank_and_descriptions: co_change, co_change_check_available: co_change_check_ok,
|
|
530
586
|
frozen_drift: drift.map { |r| r[:id] },
|
|
531
587
|
baseline_comparable: baseline_comparable,
|
|
@@ -535,6 +591,9 @@ report = {
|
|
|
535
591
|
File.write(json_path, JSON.pretty_generate(report)) if json_path
|
|
536
592
|
|
|
537
593
|
puts "eval-routing-bank (#{model}): #{passes}/#{results.size} pass, #{fails.size} fail, #{errors.size} grader-error"
|
|
594
|
+
unless action_resolution
|
|
595
|
+
puts " \u26a0 screening_resolution_only: replicas=#{replicas}, weakest task has #{min_valid_observations} valid observations, floor #{ACTION_RESOLUTION_MIN_REPLICAS} — this report locates candidates, it does not license a description edit; a per-case edit needs #{ACTION_RESOLUTION_MIN_REPLICAS} valid observations of that case (references/eval-routing.md)"
|
|
596
|
+
end
|
|
538
597
|
puts " arm: desc-budget-chars=#{desc_budget}" if desc_budget
|
|
539
598
|
unless all_observed.empty?
|
|
540
599
|
line = " clarify: #{clarify_count}/#{all_observed.size} verdicts, low-confidence(<0.5): #{low_conf_count}/#{all_observed.size}"
|
|
@@ -52,19 +52,123 @@ LEDGER_PATH = "skills/skill-extraction-workflow/references/source-register.md"
|
|
|
52
52
|
# the routing-surface class below was one such patch) only re-instantiates it on
|
|
53
53
|
# the next input. The predicate now reads the round.
|
|
54
54
|
#
|
|
55
|
-
# Rounds are cut at the commits that touch the ledger itself, walked
|
|
56
|
-
#
|
|
57
|
-
#
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
#
|
|
55
|
+
# Rounds are cut at the commits that touch the ledger itself, walked along a
|
|
56
|
+
# first-parent line. The partition comes from git alone: an author cannot widen,
|
|
57
|
+
# move, or nominate their own scope.
|
|
58
|
+
#
|
|
59
|
+
# THE SAME PARTITION BEFORE AND AFTER THE MERGE. A branch is judged on its own
|
|
60
|
+
# first-parent line while it is a pull request (CI checks out the branch head),
|
|
61
|
+
# and once merged that whole line sits behind ONE first-parent step of the
|
|
62
|
+
# integration branch. Reading that step as one boundary gave the same history a
|
|
63
|
+
# different partition after it landed — every round on the branch collapsed into
|
|
64
|
+
# one, and a row whose validity depends on its round being narrow (a routing-
|
|
65
|
+
# surface `#description` anchor in a commit that changed nothing else) turned red
|
|
66
|
+
# without a byte of it changing. That is the "verdict moved after it landed"
|
|
67
|
+
# defect in a new coat, and it surfaced on every post-merge evaluation: the push
|
|
68
|
+
# build of the integration branch and the promotion pull request. So a merge that
|
|
69
|
+
# git itself reproduces from its two parents is EXPANDED in place: its second
|
|
70
|
+
# parent's line, from the fork point to the merged head, is walked with the same
|
|
71
|
+
# rule, recursively, and contributes exactly the rounds it had as a branch.
|
|
72
|
+
#
|
|
73
|
+
# A merge is expanded only when git can rebuild it — two parents, a tree equal to
|
|
74
|
+
# `git merge-tree --write-tree` of those parents, and a second parent that is
|
|
75
|
+
# neither already on the base nor already on the line being walked (a sync merge
|
|
76
|
+
# brings nothing that needs a round). Anything else — a hand-resolved merge, a
|
|
77
|
+
# conflicted one, an octopus — keeps today's single boundary at the merge, so
|
|
78
|
+
# content git did not derive from the parents is never left in no round.
|
|
79
|
+
ROUND_WALK_MAX_DEPTH = 8
|
|
80
|
+
ancestor_of = lambda do |commit, tip|
|
|
81
|
+
IO.popen(["git", "-C", root, "merge-base", "--is-ancestor", commit, tip], err: File::NULL, &:read)
|
|
82
|
+
status = $?.exitstatus
|
|
83
|
+
next true if status == 0
|
|
84
|
+
next false if status == 1
|
|
85
|
+
warn "impact_chain_git_failed: git merge-base --is-ancestor #{commit} #{tip} exited #{status}"
|
|
86
|
+
exit 1
|
|
87
|
+
end
|
|
88
|
+
# The tree git produces merging `second` into `first`; nil when that merge
|
|
89
|
+
# conflicts (whoever resolved it was not git). Needs git 2.38+, the same floor
|
|
90
|
+
# the review-ledger binder already requires for the identical invariant.
|
|
91
|
+
automatic_merge_tree = lambda do |first, second|
|
|
92
|
+
out = IO.popen(["git", "-C", root, "merge-tree", "--write-tree", first, second], err: File::NULL, &:read)
|
|
93
|
+
status = $?.exitstatus
|
|
94
|
+
next out.to_s.lines.first.to_s.strip if status == 0
|
|
95
|
+
next nil if status == 1
|
|
96
|
+
warn "impact_chain_git_failed: git merge-tree --write-tree #{first} #{second} exited #{status} (git 2.38 or newer is required)"
|
|
97
|
+
exit 1
|
|
98
|
+
end
|
|
99
|
+
# [first parent, second parent, fork point] when `commit` is a merge git can
|
|
100
|
+
# rebuild from its parents and whose second parent carries a line of its own;
|
|
101
|
+
# nil when the merge keeps today's single-boundary treatment. `line_base` is the
|
|
102
|
+
# base of the line being walked: a merge whose second parent is already below
|
|
103
|
+
# that base is a sync of what the line was cut from (the target advancing under
|
|
104
|
+
# a branch), and is a sync on the branch's own line exactly as it is on the
|
|
105
|
+
# integration line — judging it against the outer base alone would expand it
|
|
106
|
+
# during promotion and strand the branch's earlier work in a rowless span.
|
|
107
|
+
expandable_merge = lambda do |commit, line_base|
|
|
108
|
+
parents = git_read.call("rev-list", "--parents", "-n", "1", commit).split[1..] || []
|
|
109
|
+
next nil unless parents.length == 2
|
|
110
|
+
first, second = parents
|
|
111
|
+
next nil if ancestor_of.call(second, base_ref) || ancestor_of.call(second, line_base) || ancestor_of.call(second, first)
|
|
112
|
+
own_tree = git_read.call("rev-parse", "#{commit}^{tree}").strip
|
|
113
|
+
next nil unless automatic_merge_tree.call(first, second) == own_tree
|
|
114
|
+
# The fork point read FAILS CLOSED like every other git read here: a lookup
|
|
115
|
+
# that errored would otherwise read as "no fork point", skip the expansion,
|
|
116
|
+
# and hand the merge the collapsed span — the lenient verdict — on a git
|
|
117
|
+
# failure nobody sees. Exit 1 is git's own "no common ancestor" and means
|
|
118
|
+
# there is genuinely no line to expand from.
|
|
119
|
+
fork_out = IO.popen(["git", "-C", root, "merge-base", first, second], err: File::NULL, &:read)
|
|
120
|
+
fork_status = $?.exitstatus
|
|
121
|
+
next nil if fork_status == 1
|
|
122
|
+
unless fork_status == 0
|
|
123
|
+
warn "impact_chain_git_failed: git merge-base #{first} #{second} exited #{fork_status}"
|
|
124
|
+
exit 1
|
|
125
|
+
end
|
|
126
|
+
fork = fork_out.to_s.split("\n").first.to_s.strip
|
|
127
|
+
next nil if fork.empty?
|
|
128
|
+
[first, second, fork]
|
|
129
|
+
end
|
|
130
|
+
# Each round spans (previous boundary, this one] so the work commits that
|
|
61
131
|
# precede a ledger append are inside the round they belong to — landing the change
|
|
62
132
|
# and appending the row in separate commits is the normal shape, not an evasion.
|
|
63
|
-
|
|
64
|
-
# The trailing span — owner changes committed after the last ledger append — is a
|
|
133
|
+
# The trailing span — owner changes committed after the last boundary — is a
|
|
65
134
|
# round too. It holds no rows, so its owners fall through to the presence check
|
|
66
|
-
# and the gate still fails closed on undeclared work.
|
|
67
|
-
|
|
135
|
+
# and the gate still fails closed on undeclared work. Both hold on every line the
|
|
136
|
+
# walk visits, the integration branch and each expanded merge alike.
|
|
137
|
+
round_bounds_for = lambda do |from, to, depth|
|
|
138
|
+
if depth > ROUND_WALK_MAX_DEPTH
|
|
139
|
+
warn "impact_chain_round_walk_too_deep: merges nested more than #{ROUND_WALK_MAX_DEPTH} levels between #{from} and #{to}"
|
|
140
|
+
exit 1
|
|
141
|
+
end
|
|
142
|
+
list = lambda do |*options, pathspec|
|
|
143
|
+
git_read.call("rev-list", "--first-parent", "--reverse", *options, "#{from}..#{to}", *pathspec)
|
|
144
|
+
.split("\n").map(&:strip).reject(&:empty?)
|
|
145
|
+
end
|
|
146
|
+
line = list.call([])
|
|
147
|
+
ledger_heads = list.call(["--", LEDGER_PATH])
|
|
148
|
+
merges = list.call("--merges", [])
|
|
149
|
+
spans = []
|
|
150
|
+
prev = from
|
|
151
|
+
line.each do |commit|
|
|
152
|
+
expansion = merges.include?(commit) ? expandable_merge.call(commit, from) : nil
|
|
153
|
+
if expansion
|
|
154
|
+
first, second, fork = expansion
|
|
155
|
+
spans << [prev, first] unless prev == first
|
|
156
|
+
spans.concat(round_bounds_for.call(fork, second, depth + 1))
|
|
157
|
+
prev = commit
|
|
158
|
+
elsif ledger_heads.include?(commit)
|
|
159
|
+
spans << [prev, commit]
|
|
160
|
+
prev = commit
|
|
161
|
+
end
|
|
162
|
+
end
|
|
163
|
+
spans << [prev, to]
|
|
164
|
+
spans
|
|
165
|
+
end
|
|
166
|
+
round_bounds = round_bounds_for.call(base_ref, "HEAD", 0)
|
|
167
|
+
# Diagnostic only: print the partition so a verdict can be read against the
|
|
168
|
+
# rounds it was judged in. Off by default so no suite's output assertions move.
|
|
169
|
+
if ENV["CCL_IMPACT_CHAIN_TRACE_ROUNDS"] == "1"
|
|
170
|
+
round_bounds.each { |span_base, span_head| warn "impact_chain_round: #{span_base[0, 12]}..#{span_head[0, 12]}" }
|
|
171
|
+
end
|
|
68
172
|
# Everything a predicate needs to judge one span. Built lazily per span and
|
|
69
173
|
# memoized: a round whose rows are all RED-baseline never pays for the rename
|
|
70
174
|
# derivation.
|
|
@@ -89,8 +89,13 @@ its own branch and a later round could restore the real one, so judging history
|
|
|
89
89
|
with history's tools would let that round's forged ledger stand forever. Using
|
|
90
90
|
the landing tree's tools means a controller or validator change between a round
|
|
91
91
|
and the promotion can stop an old round reproducing, and that reads as a refusal
|
|
92
|
-
rather than a pass. A round is
|
|
93
|
-
|
|
92
|
+
rather than a pass. A round's evidence is likewise read from the landing tree,
|
|
93
|
+
not from the round's own checkout: a ledger is evidence because the validator
|
|
94
|
+
accepts it and its candidate hash equals the round's packet, not because of
|
|
95
|
+
where it was committed, so a round that merged without its ledger is bound by a
|
|
96
|
+
later review of the same bytes committed on the integration branch -- and by
|
|
97
|
+
nothing less, since a ledger for any other bytes does not match. A round is
|
|
98
|
+
never itself a chain, so the walk is one level deep by construction. The detached checkout is released with `git worktree
|
|
94
99
|
remove` and its removal verified against the worktree list; a checkout that
|
|
95
100
|
cannot be released is an error, never a pass, and nothing prunes registrations
|
|
96
101
|
this run did not create. The chain is
|
|
@@ -581,7 +586,7 @@ def render_manifest(
|
|
|
581
586
|
|
|
582
587
|
def accepted_ledger_for(
|
|
583
588
|
evidence: list[tuple[Path, dict]],
|
|
584
|
-
|
|
589
|
+
evidence_home: Path,
|
|
585
590
|
validator: Path,
|
|
586
591
|
digest: str,
|
|
587
592
|
rejected: list[str],
|
|
@@ -590,14 +595,15 @@ def accepted_ledger_for(
|
|
|
590
595
|
|
|
591
596
|
The same criterion the single-candidate path uses: a receipt-shaped file is
|
|
592
597
|
not evidence, only a ledger the validator accepts, because this gate cannot
|
|
593
|
-
authenticate that a controller minted what it reads.
|
|
598
|
+
authenticate that a controller minted what it reads. `evidence_home` is the
|
|
599
|
+
tree the evidence was enumerated from, used only to name the ledger.
|
|
594
600
|
"""
|
|
595
601
|
for path, payload in evidence:
|
|
596
602
|
if payload.get("candidate_sha256") != digest:
|
|
597
603
|
continue
|
|
598
604
|
if "closeout_state" not in payload or "controller_receipts" not in payload:
|
|
599
605
|
continue
|
|
600
|
-
relative = str(path.relative_to(
|
|
606
|
+
relative = str(path.relative_to(evidence_home))
|
|
601
607
|
accepted, output = validator_accepts(validator, path)
|
|
602
608
|
if accepted:
|
|
603
609
|
return f"{relative} -- {output}"
|
|
@@ -613,6 +619,7 @@ def bind_manifest(
|
|
|
613
619
|
excludes: tuple[str, ...],
|
|
614
620
|
changed_all: list[str],
|
|
615
621
|
evidence: list[tuple[Path, dict]],
|
|
622
|
+
evidence_home: Path,
|
|
616
623
|
validator: Path,
|
|
617
624
|
rejected_ledgers: list[str],
|
|
618
625
|
) -> list[str]:
|
|
@@ -630,7 +637,7 @@ def bind_manifest(
|
|
|
630
637
|
raise ManifestError(
|
|
631
638
|
f"{label} recorded {recorded[:12]}... but does not reproduce: the candidate now hashes to {actual[:12]}..."
|
|
632
639
|
)
|
|
633
|
-
proof = accepted_ledger_for(evidence,
|
|
640
|
+
proof = accepted_ledger_for(evidence, evidence_home, validator, actual, rejected_ledgers)
|
|
634
641
|
if proof is None:
|
|
635
642
|
raise ManifestError(f"no accepted ledger binds {label} {actual}")
|
|
636
643
|
proofs.append(f" {label} {actual[:12]}... <- {proof}")
|
|
@@ -889,8 +896,9 @@ def bind_chain(
|
|
|
889
896
|
Returns (round count, proof lines) or raises ChainError naming the first
|
|
890
897
|
step that does not add up. Each round is rebound by this same gate in a
|
|
891
898
|
detached checkout of its head against its first parent, with THIS tree's
|
|
892
|
-
controller and
|
|
893
|
-
|
|
899
|
+
controller, validator and committed evidence (never the round's own tools;
|
|
900
|
+
the round's own evidence is part of this tree's history and is found there),
|
|
901
|
+
and never as a chain of its own.
|
|
894
902
|
"""
|
|
895
903
|
steps = walk_first_parent_chain(repo_root, base_tip)
|
|
896
904
|
proofs: list[str] = []
|
|
@@ -902,7 +910,13 @@ def bind_chain(
|
|
|
902
910
|
continue
|
|
903
911
|
with detached_checkout(repo_root, second) as round_root:
|
|
904
912
|
binding = bind_candidate(
|
|
905
|
-
round_root,
|
|
913
|
+
round_root,
|
|
914
|
+
first,
|
|
915
|
+
DEFAULT_PATHS,
|
|
916
|
+
evidence_root,
|
|
917
|
+
allow_chain=False,
|
|
918
|
+
tools_root=repo_root,
|
|
919
|
+
evidence_tree=repo_root,
|
|
906
920
|
)
|
|
907
921
|
subject = git_read(repo_root, ["log", "-1", "--format=%s", merge], f"cannot read {merge[:12]}")
|
|
908
922
|
if not binding.ok:
|
|
@@ -955,15 +969,24 @@ def bind_candidate(
|
|
|
955
969
|
evidence_root: str,
|
|
956
970
|
allow_chain: bool,
|
|
957
971
|
tools_root: Path | None = None,
|
|
972
|
+
evidence_tree: Path | None = None,
|
|
958
973
|
) -> Binding:
|
|
959
974
|
"""Evaluate one checkout against one base: single ledger, then manifest, then chain.
|
|
960
975
|
|
|
961
976
|
`tools_root` names the tree whose controller and validator judge the
|
|
962
977
|
candidate; it defaults to the checkout itself and is the landing tree when a
|
|
963
978
|
historical round is rebound, so a round never judges itself with its own tools.
|
|
979
|
+
`evidence_tree` names the tree whose committed evidence is consulted; it too
|
|
980
|
+
defaults to the checkout and is the landing tree when a historical round is
|
|
981
|
+
rebound. Evidence is a validator-accepted closeout bound to the round's own
|
|
982
|
+
candidate hash wherever it was committed: a round that merged without its
|
|
983
|
+
ledger is bound by a later review of the same bytes, committed on the
|
|
984
|
+
integration branch, and by nothing less -- the candidate hash and the
|
|
985
|
+
validator, not the file's location, are what make a ledger evidence.
|
|
964
986
|
"""
|
|
965
987
|
binding = Binding()
|
|
966
988
|
tools = tools_root if tools_root is not None else repo_root
|
|
989
|
+
evidence_home = evidence_tree if evidence_tree is not None else repo_root
|
|
967
990
|
fork, excludes, paths, changed = candidate_scope(repo_root, base_tip, user_paths)
|
|
968
991
|
binding.fork = fork
|
|
969
992
|
binding.changed = changed
|
|
@@ -987,14 +1010,14 @@ def bind_candidate(
|
|
|
987
1010
|
whole_error = str(exc)
|
|
988
1011
|
|
|
989
1012
|
validator = tools / "skills" / "skill-extraction-workflow" / "scripts" / VALIDATOR
|
|
990
|
-
evidence = scan(
|
|
1013
|
+
evidence = scan(evidence_home, evidence_root)
|
|
991
1014
|
ledgers: list[str] = []
|
|
992
1015
|
if expected is not None:
|
|
993
1016
|
# Only a validator-accepted ledger counts. A receipt-shaped file proves
|
|
994
1017
|
# nothing on its own: this gate cannot authenticate that a controller
|
|
995
1018
|
# minted it, so any branch keyed on a self-declared field is a bypass a
|
|
996
1019
|
# contributor can hand-write.
|
|
997
|
-
proof = accepted_ledger_for(evidence,
|
|
1020
|
+
proof = accepted_ledger_for(evidence, evidence_home, validator, expected, ledgers)
|
|
998
1021
|
if proof is not None:
|
|
999
1022
|
binding.ok = True
|
|
1000
1023
|
binding.summary = (
|
|
@@ -1007,10 +1030,10 @@ def bind_candidate(
|
|
|
1007
1030
|
for path, payload in evidence:
|
|
1008
1031
|
if payload.get("kind") != MANIFEST_KIND:
|
|
1009
1032
|
continue
|
|
1010
|
-
relative = str(path.relative_to(
|
|
1033
|
+
relative = str(path.relative_to(evidence_home))
|
|
1011
1034
|
try:
|
|
1012
1035
|
proofs = bind_manifest(
|
|
1013
|
-
module, repo_root, fork, payload, excludes, changed, evidence, validator, ledgers
|
|
1036
|
+
module, repo_root, fork, payload, excludes, changed, evidence, evidence_home, validator, ledgers
|
|
1014
1037
|
)
|
|
1015
1038
|
except ManifestError as exc:
|
|
1016
1039
|
manifests.append(f"{relative}: {exc}")
|