@dzhechkov/skills-bto 1.3.3 → 1.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,157 @@
1
+ {
2
+ "manifest": {
3
+ "version": 1,
4
+ "pack": "skills-bto",
5
+ "files": [
6
+ {
7
+ "path": "CHANGELOG.md",
8
+ "sha256": "3cdc74ff663f1a60b0905b26c74cb641f84e8ee89a077f479bed31b19545f359"
9
+ },
10
+ {
11
+ "path": "LICENSE",
12
+ "sha256": "3b93f7070f6843f37e0cae255f7caff22b466a4017fd2020d7d3216237e5a701"
13
+ },
14
+ {
15
+ "path": "README.md",
16
+ "sha256": "a622d32ed16bdd3b0db021ad84c5eba87da18e63fe2dd96b10ab1e3f86e24a36"
17
+ },
18
+ {
19
+ "path": "bin/cli.js",
20
+ "sha256": "a210174d6b63d053f7744757972d31acb0f4d9c59f8dacff7b62872a19ebd31a"
21
+ },
22
+ {
23
+ "path": "package.json",
24
+ "sha256": "6318b6221e222530bb00f50e978ebeed985f563b8e45f6968c0c2390354eec93"
25
+ },
26
+ {
27
+ "path": "src/cli.js",
28
+ "sha256": "a712d4840731f4974e22155dc54f7ff3478e3bd15af8e4cefeb10ee3f11b3fe7"
29
+ },
30
+ {
31
+ "path": "src/commands/doctor.js",
32
+ "sha256": "1c7d323316e502b501ce2b15f177f7cc0766c70e995092b973d7d56522bbd034"
33
+ },
34
+ {
35
+ "path": "src/commands/init.js",
36
+ "sha256": "1317cb61dc3feafaa7224fb155caaf77a7a5e1097f81070477ffdc53e5c43b22"
37
+ },
38
+ {
39
+ "path": "src/commands/list.js",
40
+ "sha256": "1ec92e70601bac6da48cddca0764663d999b0aa3cca12b785d825b88de974cfa"
41
+ },
42
+ {
43
+ "path": "src/commands/remove.js",
44
+ "sha256": "f62beed8b76b134f50d959cd70bb96674a26bcce8441baf49fe08a0780d845ea"
45
+ },
46
+ {
47
+ "path": "src/commands/update.js",
48
+ "sha256": "f3217f1601eb3fdcfb9099fa07c551e4c922544d09224c9664b4ee5ce230634c"
49
+ },
50
+ {
51
+ "path": "src/utils.js",
52
+ "sha256": "0ab07df1d28b9c2565f564e43d76c4d7f28b776d43bcb946b22ac70db821f030"
53
+ },
54
+ {
55
+ "path": "templates/.claude/agents/bto-judge-panel.md",
56
+ "sha256": "1e31c6d4caf15f0ce1202d5cbfaf17d28611e2e98f125184da9ecaa30e01d7b8"
57
+ },
58
+ {
59
+ "path": "templates/.claude/agents/bto-optimizer-worker.md",
60
+ "sha256": "b9b0522469f376135eb80d9c689c1eb0f5f07e96a3e4f4749002666a8236b64b"
61
+ },
62
+ {
63
+ "path": "templates/.claude/commands/bto-benchmark.md",
64
+ "sha256": "45a85faaaff840a44bc77e61892b3b0188ff31d4c80c2ef5438618d93a547ef7"
65
+ },
66
+ {
67
+ "path": "templates/.claude/commands/bto-build.md",
68
+ "sha256": "a1ac21e6ca097b9af1853a0cc815ff108c23eba690a76d4696e21fb4877c676c"
69
+ },
70
+ {
71
+ "path": "templates/.claude/commands/bto-optimize.md",
72
+ "sha256": "dd07ec79b88ccd3d29311325020c7fe389394e4f5d73004eadad1c605e104362"
73
+ },
74
+ {
75
+ "path": "templates/.claude/commands/bto-test.md",
76
+ "sha256": "823c509c3f0581cbd24daa1d2ce092ab4cd6f1bc60b23536ab1ce6521fadfcc1"
77
+ },
78
+ {
79
+ "path": "templates/.claude/commands/bto.md",
80
+ "sha256": "fa1a83a725e6fda68edfea7c15bce0ed688686127f2924de8983184e857fa991"
81
+ },
82
+ {
83
+ "path": "templates/.claude/commands/verify-chain.md",
84
+ "sha256": "b064f512724091908a953dde475abe1c92f63e02728846644a218144f6616e9e"
85
+ },
86
+ {
87
+ "path": "templates/.claude/rules/bto-quality-gates.md",
88
+ "sha256": "e4881acf5d15472629cea036fbea50bf2dd577b17d5c1c67730a03a5f47538b5"
89
+ },
90
+ {
91
+ "path": "templates/.claude/rules/witness-chain.md",
92
+ "sha256": "679aaed6c9b3a173d3fb2888239745d34ed1988315d4f531d14967c33193512b"
93
+ },
94
+ {
95
+ "path": "templates/.claude/shards/bto-evaluation.shard.md",
96
+ "sha256": "df1abe8d28f2a47fd24aa2da7621c9086f5208f9d675f99e25cf4889fbc6b456"
97
+ },
98
+ {
99
+ "path": "templates/.claude/skills/bto/SKILL.md",
100
+ "sha256": "cc15a07922126226776d4c48446b3d75fa8caaf16949389f448f42c9787207cb"
101
+ },
102
+ {
103
+ "path": "templates/.claude/skills/bto/examples/sample-benchmark-report.md",
104
+ "sha256": "cfea6681a75465c0372defee137a39d8fa86ed48b052f1685c96a1e96fb81f02"
105
+ },
106
+ {
107
+ "path": "templates/.claude/skills/bto/examples/sample-eval-report.md",
108
+ "sha256": "45ed20c23e42a9141de55c103a3d64917cf1f51e0d59a16fbdd9fe498632bbe5"
109
+ },
110
+ {
111
+ "path": "templates/.claude/skills/bto/modules/benchmark.md",
112
+ "sha256": "41f87e7b2cd8ba9ae2f913a586651d299052534ba4875797c79cbc577d3df4d3"
113
+ },
114
+ {
115
+ "path": "templates/.claude/skills/bto/modules/build.md",
116
+ "sha256": "d1e26e331562175b3295de3e58edfecbd80eaaed9a57d7e5cf094bc4ffae8247"
117
+ },
118
+ {
119
+ "path": "templates/.claude/skills/bto/modules/optimize.md",
120
+ "sha256": "d8737955bd02cdfa83af6d72fdbd59ac2f4a6d73b41788a4c801496654c0cc93"
121
+ },
122
+ {
123
+ "path": "templates/.claude/skills/bto/modules/test.md",
124
+ "sha256": "9daa90fd8566fe7372ad088cd489cd11caf9ba2bd2c3bf2f2b822e3d2ee0bac9"
125
+ },
126
+ {
127
+ "path": "templates/.claude/skills/bto/references/eval-patterns.md",
128
+ "sha256": "bb4e4aca4c67937cedb91e66debb0e17f84322a5d9e29d8e0ef60a1af13cc92b"
129
+ },
130
+ {
131
+ "path": "templates/.claude/skills/bto/references/golden-samples.md",
132
+ "sha256": "d98baa5d0c7816fb547ba36d2526bed6317ec9c9c6e8a7c50af2c5d7c5a22a46"
133
+ },
134
+ {
135
+ "path": "templates/.claude/skills/bto/references/judge-rubrics.md",
136
+ "sha256": "e7ee1f3802c79ce6a1966976d227a8cb060232109e487c6db45f0280cab47e20"
137
+ },
138
+ {
139
+ "path": "templates/.claude/skills/bto/references/optimization-methods.md",
140
+ "sha256": "d15962e1035e600b897384faa9ea20d3c1c687da797e63480d48e2da039b8783"
141
+ },
142
+ {
143
+ "path": "templates/.claude/skills/bto/references/quality-checklist.md",
144
+ "sha256": "611c0a0785a41b12b8dde09e9086c5e051ac0672711a4e33df5fefc3c87f7561"
145
+ },
146
+ {
147
+ "path": "templates/lib/judge-attestation.md",
148
+ "sha256": "7a5c05c6ed514256e2e394b703dc5dfd5eee292f87fe95f8f4153707ff179667"
149
+ },
150
+ {
151
+ "path": "templates/lib/witness-chain.md",
152
+ "sha256": "8d14187d213b2f4422af2699df6c940b54f5f8aba352d2e8070096b604fcba7e"
153
+ }
154
+ ]
155
+ },
156
+ "signature": "xTWXwW6XDgGgRBKacw50PZygmxm6t0dEA9kZ4FkFvI1LCzsBb7Cigip7BF4MKm4j79NBUbSvgtxwm859bKabCw=="
157
+ }
package/CHANGELOG.md ADDED
@@ -0,0 +1,75 @@
1
+ # skills-bto
2
+
3
+ ## 1.4.0 — 2026-08-18
4
+
5
+ Hardening wave from the BTO improvement review (slice D). Four adopted items; one deferred.
6
+
7
+ ### Added
8
+
9
+ - **Frontmatter fence on `SKILL.md` (P5).** The pack's own `SKILL.md` now begins with a
10
+ `--- name/description ---` fence. Before this, `dz list --skills-dir <pack>/templates/.claude/skills`
11
+ exited 1 with `SKILL.md must begin with a "---" frontmatter fence` — and because the loader walks
12
+ the whole tree, the unfenced file made *every* skill in the directory unlistable. The fix
13
+ propagates to what BTO **generates**: the BUILD Skill Template emits the fence ahead of `# <Name>`,
14
+ and a new `CHECK-S0` / `TEST-SK0` pair scores the property. Commands, rules and agent templates
15
+ deliberately keep no fence — the parser that requires it reads skills only, and that call is now
16
+ written down as a decision instead of an omission.
17
+ - **`## Agent Authoring Rule` (P1).** A new normative section in `SKILL.md`, pointed at from all
18
+ eight authoring texts (four modules, both agent templates, the driving commands): write the
19
+ skeleton first, then incremental `Edit` appends; never one giant `Write`. Carries the watchdog
20
+ numbers (180 s / 600 s) and the consequence — a killed agent loses all unwritten output. The two
21
+ genuinely short emissions (B2 probe answer, one-line variant score) are declared EXEMPT at their
22
+ sites, so an exemption and an omission are distinguishable by grep. OPTIMIZE Step 4 gains an arity
23
+ tripwire that WARNs with the missing variant ids instead of silently selecting from a short pool.
24
+ - **Layer B-1: Environment Preconditions + the INCONCLUSIVE verdict (P2).** A new first benchmark
25
+ layer probes that each downstream layer *can* execute and prints
26
+ `Preconditions: ALL_GREEN | DEGRADED(list) | ABORT` in **every** report header, clean runs
27
+ included. `INCONCLUSIVE` joins the verdict lattice, is evaluated ahead of every numeric threshold,
28
+ and is sticky downstream (an INCONCLUSIVE BENCHMARK does not enter TEST). Universal check
29
+ **U-13 `SKIPPED is not PASSED`** carries the same standard into Layer 0.
30
+ - **Judge provenance lines (P4, template half).** Every evaluation report now always emits
31
+ `Authored by:` / `Judged by:` / `Cross-family: YES|NO`, and a same-family panel prints the
32
+ `SAME-FAMILY PANEL` degradation banner. The Layer-2 judge table gains a `Family` column and
33
+ `references/eval-patterns.md` gains anti-conformity measure 6, *Model Family Diversity*.
34
+ - **Test suite + `npm test`.** `test/*.test.mjs` (5 spec files) wired to
35
+ `"test": "node --test test/*.test.mjs"`. The specs spawn the real `dz` binary as the acceptance
36
+ oracle and assert against the shipped `templates/` text.
37
+
38
+ ### Changed — user-visible recalibrations
39
+
40
+ These move numbers that existing reports display. Both are deliberate:
41
+
42
+ - **B1 skill-test denominator 5 → 6.** Adding `TEST-SK0` re-bases every skill's B1 ratio from fifths
43
+ to sixths.
44
+ - **Universal checklist total 12 → 13.** Adding `U-13` re-bases the Layer 0 pass rate; the two
45
+ worked example reports were recomputed to match (`Universal (13) + Skill-specific (16) = 29`).
46
+ - **B1 pass rate denominator is now `tests_executed`, not `tests_total`.** A check that could not run
47
+ is neither a pass nor a failure, and any `tests_inconclusive > 0` forces the layer to INCONCLUSIVE
48
+ regardless of the ratio.
49
+ - **Partial-evaluation renormalization narrowed.** Only layers the operator deliberately skipped via
50
+ `level` are renormalized away. A layer that was requested and could not run is INCONCLUSIVE and is
51
+ never renormalized — redistributing its weight can only raise the score, turning a gap in evidence
52
+ into a better grade. The `or skip B0` escape in the BENCHMARK anti-pattern table is removed for the
53
+ same reason.
54
+ - `references/golden-samples.md` documents the fence as a structural row but **excludes it from
55
+ `sections_expected`**; the required-section count stays **7**, so a conformant skill's B0 coverage
56
+ is unchanged.
57
+ - `U-03` amended to accept *fence-then-H1*, so the pack no longer fails its own fixed file. `CHECK-S2`
58
+ was re-worded ("first heading AFTER the frontmatter fence"), never dropped — the H1 is still required.
59
+
60
+ ### Not changed (deliberately deferred)
61
+
62
+ - **No hard cross-family requirement.** The `Cross-family:` line and its banner are *recorded and
63
+ advisory*: they change no score and gate nothing. Making a cross-family Critic mandatory would
64
+ block every user with a single model family, on evidence of n=1; it stays in backlog pending
65
+ 3-artifact planted-defect replication. A spec asserts that nothing in the pack blocks, fails or
66
+ refuses on `Cross-family: NO`, so the deferred half cannot arrive unannounced.
67
+ - The pre-existing **model-identity collusion BLOCK** is untouched. Identity blocks; family is
68
+ advisory; the two now read as one policy.
69
+ - **Round narrowing (P3)** — deferred, instrument first.
70
+
71
+ ## 1.3.0
72
+
73
+ Initial release — canonicalized into `@dzhechkov/skills-bto` in the dz-harness-hub monorepo.
74
+
75
+ Build-Benchmark-Test-Optimize skill pack for Claude Code — deterministic benchmarking, quality gates, witness chain, judge attestation, and optimization
package/README.md CHANGED
@@ -93,14 +93,29 @@ BUILD ──→ BENCHMARK ──→ TEST ──→ OPTIMIZE
93
93
 
94
94
  | Layer | Cost | Purpose |
95
95
  |-------|------|---------|
96
+ | **B-1** | Zero (deterministic) | **Environment preconditions** — probes that each downstream layer *can* run, and records the answer |
96
97
  | **B0** | Zero (deterministic) | Golden sample comparison — section coverage, ordering, proportions |
97
- | **B1** | Zero (deterministic) | Deterministic test suite — 5 tests per artifact type, PASS/FAIL |
98
+ | **B1** | Zero (deterministic) | Deterministic test suite — 6 tests for skills, 5 per other artifact type, PASS/FAIL |
98
99
  | **B2** | Minimal (3× haiku) | Consistency probe — 3 parallel agents, agreement measurement |
99
100
  | **B3** | Zero (deterministic) | Performance metrics — token efficiency, bloat detection, redundancy |
100
101
 
101
102
  **Scoring:** `BENCHMARK = B0×0.30 + B1×0.35 + B2×0.15 + B3×0.20`
103
+ (pass rate denominator = checks that actually **executed**, never the declared total)
102
104
 
103
- **Gate:** < 0.50 BLOCK | 0.50–0.70 WARN | > 0.70 PASS → proceed to TEST
105
+ **Gate**, first matching row wins:
106
+
107
+ | Condition | Verdict |
108
+ |-----------|---------|
109
+ | B-1 returned ABORT | **ABORT** — no score is emitted at all |
110
+ | Any layer INCONCLUSIVE | **INCONCLUSIVE** — not a pass and not a failure; TEST is not entered |
111
+ | Score < 0.50 | BLOCK |
112
+ | Score 0.50–0.70 | WARN |
113
+ | Score > 0.70 | PASS → proceed to TEST |
114
+
115
+ Layer B-1 prints `Preconditions: ALL_GREEN | DEGRADED(list) | ABORT` in the header of **every**
116
+ report, clean runs included — an absent line is itself a finding. A layer the operator deliberately
117
+ skipped via `--level` is renormalized away; a layer that was *requested and could not run* is
118
+ INCONCLUSIVE and is never renormalized, because redistributing its weight can only raise the score.
104
119
 
105
120
  ### TEST Layer Model
106
121
 
@@ -117,10 +132,35 @@ BUILD ──→ BENCHMARK ──→ TEST ──→ OPTIMIZE
117
132
  - Judges never see each other's scores before submitting
118
133
  - Standard weights: Domain Expert (0.4) / Critic (0.3) / Completeness Auditor (0.3)
119
134
  - If `max_score - min_score > 3` → meta-judge escalation
135
+ - Every evaluation report always emits three provenance lines — `Authored by:`, `Judged by:` and
136
+ `Cross-family: YES|NO` — so a report read six months later says whether it was cross-checked:
137
+
138
+ ```
139
+ Authored by: claude-opus (anthropic)
140
+ Judged by: Expert=claude-sonnet | Critic=gpt-5.5 | Auditor=claude-sonnet
141
+ Cross-family: YES (Critic=openai — different family than the author)
142
+ ```
143
+
144
+ - Running the Critic seat on a **different model family** than the author is *recommended*; when no
145
+ second family is reachable the report prints a loud `SAME-FAMILY PANEL` banner saying CORRECTNESS
146
+ and ROBUSTNESS are self-assessment. The family axis is **advisory** — it changes no score and
147
+ blocks nothing. The separate *model-identity* rule still BLOCKS: the same model must not both
148
+ generate and judge an artifact.
149
+
150
+ ### Agent Authoring Rule
151
+
152
+ Every BTO agent that writes a document writes it **incrementally**: a small `Write` carrying the
153
+ title and section skeleton first, then one `Edit` per section. An agent that emits no tool event for
154
+ the harness watchdog window (180 s inside a workflow runner, 600 s for a directly-spawned subagent)
155
+ is killed, and a killed agent loses *all* unwritten output — there is no partial save. Genuinely
156
+ short emissions (the B2 probe answer, a one-line variant score) are exempt, and each exemption is
157
+ written down at its site rather than inferred from its absence.
120
158
 
121
159
  ### Quality Gates
122
160
 
123
161
  - **BENCHMARK** must pass (score ≥ 0.50) before TEST begins
162
+ - A check or layer that could not execute is reported **INCONCLUSIVE** — never counted as a pass,
163
+ never silently dropped (`SKIPPED is not PASSED`, universal check U-13)
124
164
  - BENCHMARK score < 0.50 → BLOCK (artifact needs rework)
125
165
  - Layer 0 must pass before Layer 1
126
166
  - Layer 1 must pass before Layer 2
@@ -158,6 +198,31 @@ When installed alongside Keysarium, BTO can evaluate and optimize any skill or c
158
198
 
159
199
  ---
160
200
 
201
+ ## Generated skills carry a frontmatter fence
202
+
203
+ Skills built with `/bto-build` (and the pack's own `SKILL.md`) begin with a `---` frontmatter fence
204
+ carrying `name:` and `description:` before the `# Title` heading. This is what makes a skill loadable
205
+ — an unfenced `SKILL.md` makes the harness refuse the *entire* skills directory, not just that one
206
+ skill. Commands, rules and agent templates deliberately do **not** get a fence: the parser that
207
+ requires it reads skills only.
208
+
209
+ Verify a generated skill with the reproducer the pack itself uses:
210
+
211
+ ```bash
212
+ dz list --skills-dir <project>/.claude/skills # exit 0 and the skill named in the listing
213
+ dz info <name> --skills-dir <project>/.claude/skills
214
+ ```
215
+
216
+ ## Development
217
+
218
+ ```bash
219
+ npm test # node --test test/*.test.mjs — pack conformance specs
220
+ ```
221
+
222
+ The specs assert against the shipped `templates/` text and spawn the real `dz` binary as the
223
+ acceptance oracle. A layout-only check (`dz skills-verify --static`) is necessary but **not**
224
+ sufficient: it passes on a tree whose fence is missing, so `dz list` is the discriminating oracle.
225
+
161
226
  ## Requirements
162
227
 
163
228
  - **Claude Code CLI** — installed and configured ([installation guide](https://docs.anthropic.com/en/docs/claude-code))
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dzhechkov/skills-bto",
3
- "version": "1.3.3",
3
+ "version": "1.4.1",
4
4
  "description": "Build-Benchmark-Test-Optimize skill pack for Claude Code — deterministic benchmarking, quality gates, witness chain, judge attestation, and optimization",
5
5
  "main": "src/cli.js",
6
6
  "bin": {
@@ -10,7 +10,10 @@
10
10
  "bin/",
11
11
  "src/",
12
12
  "templates/",
13
- "LICENSE"
13
+ "CHANGELOG.md",
14
+ "LICENSE",
15
+ ".dz-manifest.json",
16
+ "sbom.json"
14
17
  ],
15
18
  "keywords": [
16
19
  "claude",
@@ -49,5 +52,8 @@
49
52
  },
50
53
  "publishConfig": {
51
54
  "access": "public"
55
+ },
56
+ "scripts": {
57
+ "test": "node --test test/*.test.mjs"
52
58
  }
53
59
  }