@meyverick/agentic 5.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/AGENTS.md +234 -0
  2. package/CHANGELOG.md +236 -0
  3. package/README.md +50 -0
  4. package/install.ts +349 -0
  5. package/package.json +37 -0
  6. package/scripts/check-deps.mjs +587 -0
  7. package/scripts/git-dl.mjs +100 -0
  8. package/skills/check/SKILL.md +108 -0
  9. package/skills/check/evals/benchmark.json +40 -0
  10. package/skills/check/evals/evals.json +38 -0
  11. package/skills/check/references/diagnostic-matrix.md +170 -0
  12. package/skills/check/references/script-anatomy.md +154 -0
  13. package/skills/create-skill/SKILL.md +291 -0
  14. package/skills/create-skill/assets/templates/SKILL.md.template +118 -0
  15. package/skills/create-skill/assets/templates/evals.json.template +36 -0
  16. package/skills/create-skill/assets/templates/grading.json.template +26 -0
  17. package/skills/create-skill/evals/benchmark.json +41 -0
  18. package/skills/create-skill/evals/evals.json +50 -0
  19. package/skills/create-skill/evals/grading-template.json +36 -0
  20. package/skills/create-skill/evals/near-misses.json +35 -0
  21. package/skills/create-skill/evals/trigger-queries.json +80 -0
  22. package/skills/create-skill/references/antipatterns.md +123 -0
  23. package/skills/create-skill/references/component-decomposition.md +130 -0
  24. package/skills/create-skill/references/content-quality-criteria.md +61 -0
  25. package/skills/create-skill/references/description-optimization.md +90 -0
  26. package/skills/create-skill/references/eval-methodology.md +100 -0
  27. package/skills/create-skill/references/fragility-matching.md +88 -0
  28. package/skills/create-skill/references/gotchas-patterns.md +80 -0
  29. package/skills/create-skill/references/specification.md +77 -0
  30. package/skills/create-skill/scripts/audit-antipatterns.mjs +164 -0
  31. package/skills/create-skill/scripts/compute-benchmark.mjs +111 -0
  32. package/skills/create-skill/scripts/run-cold-eval.mjs +118 -0
  33. package/skills/create-skill/scripts/scaffold-skill.mjs +86 -0
  34. package/skills/create-skill/scripts/validate-routing.mjs +137 -0
  35. package/skills/create-skill/scripts/validate-structure.mjs +223 -0
  36. package/skills/design-craft/SKILL.md +134 -0
  37. package/skills/design-craft/evals/benchmark.json +41 -0
  38. package/skills/design-craft/evals/evals.json +81 -0
  39. package/skills/design-craft/references/anti-slop-patterns.md +49 -0
  40. package/skills/design-craft/references/art-direction.md +89 -0
  41. package/skills/design-craft/references/design-engineering.md +122 -0
  42. package/skills/design-craft/references/motion-craft.md +124 -0
  43. package/skills/design-craft/references/process.md +47 -0
  44. package/skills/design-craft/references/review-checklist.md +121 -0
  45. package/skills/guardrails/SKILL.md +118 -0
  46. package/skills/guardrails/evals/benchmark.json +40 -0
  47. package/skills/guardrails/evals/evals.json +49 -0
  48. package/skills/guardrails/references/guardrails-patterns.md +43 -0
  49. package/skills/okf-docs/SKILL.md +79 -0
  50. package/skills/okf-docs/evals/benchmark.json +21 -0
  51. package/skills/okf-docs/evals/evals.json +37 -0
  52. package/skills/okf-docs/references/okf-spec.md +56 -0
  53. package/skills/okf-docs/scripts/validate-frontmatter.mjs +130 -0
  54. package/skills/openspec-harden/SKILL.md +138 -0
  55. package/skills/openspec-harden/evals/benchmark.json +40 -0
  56. package/skills/openspec-harden/evals/evals.json +38 -0
  57. package/skills/openspec-learn/SKILL.md +216 -0
  58. package/skills/openspec-learn/evals/benchmark.json +44 -0
  59. package/skills/openspec-learn/evals/evals.json +48 -0
  60. package/skills/openspec-learn/evals/retrieval-bench.json +27 -0
  61. package/skills/openspec-learn/references/conflict-handling.md +20 -0
  62. package/skills/openspec-learn/references/evaluation-methodology.md +126 -0
  63. package/skills/openspec-learn/references/examples.md +37 -0
  64. package/skills/openspec-learn/references/improvement-patterns.md +155 -0
  65. package/skills/openspec-learn/references/report-analysis.md +104 -0
  66. package/skills/openspec-learn/references/skill-quality.md +103 -0
  67. package/skills/openspec-learn/references/tool-type-detection.md +30 -0
  68. package/skills/openspec-report/SKILL.md +104 -0
  69. package/skills/openspec-report/assets/templates/assessment.md.template +84 -0
  70. package/skills/openspec-report/assets/templates/report.md.template +92 -0
  71. package/skills/openspec-report/evals/benchmark.json +44 -0
  72. package/skills/openspec-report/evals/evals.json +46 -0
  73. package/skills/qmd-research/SKILL.md +89 -0
  74. package/skills/qmd-research/evals/benchmark.json +40 -0
  75. package/skills/qmd-research/evals/evals.json +38 -0
  76. package/skills/qmd-research/references/index-management.md +69 -0
  77. package/skills/qmd-research/references/query-craft.md +82 -0
@@ -0,0 +1,291 @@
1
+ ---
2
+ name: create-skill
3
+ description: Create new Agent Skills from problem descriptions or instruction files. Walks through discovery, design, authoring, validation, evaluation, and optimization phases. Use when the user wants to build a new skill, create a skill from a workflow, extract a reusable pattern from a task, or set up evaluation for an existing skill. Do NOT use when the task involves general coding, debugging application code, writing project documentation, or any work unrelated to skill creation.
4
+ allowed-tools: Bash(*)
5
+ license: MIT
6
+ compatibility: Requires bun.
7
+ metadata:
8
+ author: agentic
9
+ version: "2.0"
10
+ positive_triggers:
11
+ - "create a new skill"
12
+ - "build a skill from a workflow"
13
+ - "extract a reusable pattern into a skill"
14
+ - "set up evaluation for an existing skill"
15
+ - "improve or fix an existing skill"
16
+ anti_triggers:
17
+ - "general coding task not related to skills"
18
+ - "debug or fix application code"
19
+ - "write project documentation or README"
20
+ runtime:
21
+ requires:
22
+ - bun >= 1.0
23
+ timeout_seconds: 30
24
+ output_format: json
25
+ ---
26
+
27
+ # Create Skill
28
+
29
+ Create new Agent Skills from problem descriptions or instruction files. Fully autonomous workflow with research-informed quality standards.
30
+
31
+ ## Quick Start
32
+
33
+ When the user wants to build a new skill:
34
+
35
+ 1. Run `scripts/scaffold-skill.mjs <skill-name>` to create skeleton in `./project/skills/`
36
+ 2. Follow the workflow below to fill in content
37
+ 3. Skill ships when Phase 7 completes
38
+
39
+ When invoked from `/skill-create <name>` with instruction file:
40
+
41
+ 1. Read `./skills-todo/<name>.md`
42
+ 2. Skip Phase 1 (Discovery) — instructions have answers
43
+ 3. Start at Phase 2 (Design) with provided decisions
44
+ 4. Follow standard workflow from there
45
+
46
+ The workflow is fully autonomous — it runs continuously until user input is needed (Discovery answers, Ship approval). Auto-fix and retry on validation/eval failures.
47
+
48
+ ## Tiers
49
+
50
+ | Tier | What's Included | When to Use |
51
+ |------|-----------------|-------------|
52
+ | **Minimal** | Structural validation + content review | Quick prototyping, low-stakes skills |
53
+ | **Standard** | Minimal + 2-3 test cases + manual eval | Most production skills |
54
+ | **Rigorous** | Standard + full eval + description optimization + quality score | High-stakes workflows, shared skills |
55
+
56
+ **Default: Rigorous.** User can request early exit to Minimal or Standard.
57
+
58
+ ## Workflow
59
+
60
+ ### Phase 0: Read Instructions (if provided)
61
+
62
+ If invoked from `/skill-create <name>`:
63
+
64
+ 1. Read `./skills-todo/<name>.md`
65
+ 2. Extract: problem, requirements, design decisions, gotchas, eval strategy
66
+ 3. Skip to Phase 2 (Design) with provided information
67
+
68
+ If invoked directly (no instruction file):
69
+ 1. Continue to Phase 1 (Discovery)
70
+
71
+ ### Phase 1: Discovery
72
+
73
+ **Skip this phase if instructions were provided.**
74
+
75
+ Ask the user (open-ended, no presets):
76
+
77
+ 1. What problem should this skill solve?
78
+ 2. What domain knowledge is needed?
79
+ 3. Which agent/harness will use it?
80
+ 4. Existing patterns to extract from?
81
+ 5. What does success look like?
82
+ 6. Which tier? (default: Rigorous)
83
+ 7. **What queries or situations SHOULD activate this skill?** (collect at least 3 examples → positive_triggers)
84
+ 8. **What queries look similar but should NOT activate this skill?** (collect at least 2 examples → anti_triggers)
85
+ 9. **Can you describe the skill's purpose in ONE sentence without using "and"?** (verifies atomic intent)
86
+
87
+ ### Phase 2: Design
88
+
89
+ Determine:
90
+
91
+ 1. **Scope**: Single atomic intent. One sentence, no "and". If compound → split into multiple skills.
92
+ 2. **Fragility**: Mutation = strict, read-only = loose, creative = low — see `references/fragility-matching.md` (Level 2).
93
+ 3. **Progressive disclosure**: SKILL.md vs references/ with context budget per tier:
94
+ - Tier 1 (frontmatter): <50 tokens
95
+ - Tier 2 (SKILL.md body): <1500 tokens
96
+ - Tier 3 (references/): on-demand only
97
+ 4. **Components**: Persona / instructions / templates / data (not omnibus) — see `references/component-decomposition.md` (Level 2).
98
+ 5. **Scripts**: Reusable logic to bundle (.mjs, cold/isolated, relative paths only)
99
+ 6. **Eval strategy**: Test cases, assertions, near-miss negatives, baseline comparison
100
+ 7. **Activation boundary**: Define what triggers and what does NOT trigger
101
+ 8. **Runtime contract**: Required runtimes (bun/node/python), timeout_seconds, output_format (json)
102
+ 9. **Output contract**: JSON schema for success and error responses
103
+ 10. **Retrieval collision check**: Run hybrid retrieval over installed and project skills before scaffolding: `qmd query "<intent>" --json -n 10` (hybrid) per `project/AGENTS.md:180`, fetch hits via `qmd multi-get`, and evaluate description overlap. If semantic overlap is detected, flag the collision and suggest updating the existing skill instead of creating a new one.
104
+
105
+ ### Phase 3: Authoring
106
+
107
+ 1. **Scaffold**: `scripts/scaffold-skill.mjs <name>` → creates directory + SKILL.md skeleton
108
+ 2. **Frontmatter**: name, description (imperative, specific, "Use when... Do NOT use when..."), positive_triggers (min 3), anti_triggers (min 2), allowed-tools, compatibility, runtime
109
+ 3. **SKILL.md body sections** (in order):
110
+ - Activation Boundary: explicit trigger/exclusion lists
111
+ - Pre-Flight Checks: environment probes before execution
112
+ - Output Contract: JSON schema for success/error
113
+ - Core instructions (<500 lines, <1500 tokens)
114
+ - Contrast: `| Before (old) | After (new) | Why different |` table when proposal carries `Contrast:` hint (e.g., `C# null → Rust Option`); keep distinct from routing
115
+ - Anti-examples: `Do NOT: <before>` → `Do: <after>` with Why, when proposal carries `Anti-example:` hint from report's Concrete Gotcha; body content, NOT frontmatter `anti_triggers`
116
+ - Tiered depth: Level 1 basics inline, Level 2 advanced behind `references/<topic>.md` (cap one file per skill)
117
+ 4. **Scripts**: Generate if clearly reusable (.mjs, self-contained, relative paths via import.meta.url, JSON output only)
118
+ 5. **References**: Domain-specific docs, loaded on-demand — **Tiered depth:** Level 1 basics inline in SKILL.md, Level 2 advanced behind `references/<topic>.md`; cap at one `references/` file per skill; link explicitly from SKILL.md
119
+ 6. **Templates**: Output shapes, examples
120
+ 7. **Single source**: Define each rule ONCE, reference everywhere else
121
+ 8. **Portability**: No absolute paths (/home/, /root/, C:\), no harness-specific dirs (.pi/, .agents/) in executable code
122
+
123
+ ### Phase 4: Validation (mandatory)
124
+
125
+ **Output contract**: all three validators emit one shared JSON envelope — `{target, pass, checks:[{id, status: PASS|FAIL|WARN|SKIP, detail}], summary}` — so downstream tooling branches deterministically on `pass` and `checks[].status` regardless of which script produced it.
126
+
127
+ **Step 1: Structural validation**
128
+ ```bash
129
+ scripts/validate-structure.mjs <skill-dir>
130
+ ```
131
+ Checks: name format, description format, directory structure, file references, positive_triggers (min 3), anti_triggers (min 2), "Use when" phrasing, "Do NOT use when" phrasing, no compound intent, no hardcoded paths, runtime declared when scripts exist.
132
+
133
+ **Evidence requirement**: record each validator's output (stdout or pass summary) in the creation session's tasks or summary before proceeding to the next step. An unrecorded validation counts as not performed.
134
+
135
+ **Step 2: Semantic routing validation**
136
+ ```bash
137
+ scripts/validate-routing.mjs <skill-dir>
138
+ ```
139
+ Checks: positive_triggers coverage, anti_triggers coverage, description-body alignment, single-responsibility verification.
140
+
141
+ **Step 3: Content review**
142
+ Agent assesses:
143
+ - Description specificity (imperative, intent-driven, single atomic intent)
144
+ - Instruction clarity (actionable, not vague)
145
+ - Progressive disclosure (SKILL.md <500 lines, <1500 tokens)
146
+ - Gotchas present (environment-specific facts)
147
+ - Fragility matching (strict for mutation, loose for read-only)
148
+ - Activation boundary completeness (all triggers documented)
149
+
150
+ **Step 4: Antipattern self-audit**
151
+ ```bash
152
+ scripts/audit-antipatterns.mjs <skill-dir>
153
+ ```
154
+ Checks: phantom tools, duplicated invariants, passive-voice triggers, prose bloat, single-file omnibus, vague success bars, multi-domain descriptions (A17), missing activation boundary (A18), hardcoded paths (A19), context budget violation (A20).
155
+
156
+ Self-correct any issues before proceeding.
157
+
158
+ ### Phase 5: Evaluation (mandatory)
159
+
160
+ **Step 1-5: Eval loop** — 2-3 cases + near-miss negatives; measure baseline (no skill) then with-skill (grade PASS/FAIL, record timing); near-miss 100% for mutation.
161
+
162
+ **Step 6: Compute benchmarks**
163
+ ```bash
164
+ scripts/compute-benchmark.mjs <eval-dir>
165
+ ```
166
+
167
+ **Step 6b: Cold-agent behavioral proof (mandatory)**
168
+ ```bash
169
+ scripts/run-cold-eval.mjs <skill-dir>
170
+ ```
171
+ Runs `evals/evals.json` cold A/B (without vs with skill), computes `d = sign(with - baseline)`, `m = |with - baseline|`, emits unified envelope `{target, pass, checks, summary}` + `behavioral: {at, baseline, with_skill, d, m, ship}` (30s timeout, JSON only). Writes `evals/benchmark.json` with `stage: behavioral` block (`{at, baseline, with_skill, d, m, ship}`) replacing `pending_cold_agent_run`; record outputs in creation session. Reuses `compute-benchmark.mjs` logic; no network, deterministic.
172
+
173
+ **Step 7: Calculate quality score**
174
+ ```
175
+ d = direction (+1 if with_skill > baseline, -1 if lower, 0 if equal)
176
+ m = magnitude = |with_skill_pass_rate - baseline_pass_rate|
177
+ quality_score = d × m
178
+ ```
179
+ Ship gate: d must be +1 AND m must be >= 0.2 (20% improvement over baseline) — now proven by `run-cold-eval.mjs` behavioral block.
180
+
181
+ **Step 8: Iterate**
182
+ If quality insufficient:
183
+ 1. Analyze failures
184
+ 2. Fix instructions
185
+ 3. Re-run evals
186
+ 4. Repeat until plateau
187
+
188
+ ### Phase 6: Optimization (mandatory)
189
+
190
+ **Step 1: Create trigger test queries**
191
+ - 10-20 queries (mix should/shouldn't trigger + near-misses)
192
+ - Include queries matching each positive_trigger
193
+ - Include queries matching each anti_trigger
194
+
195
+ **Step 2: Test description triggering**
196
+ Run each query, check if skill activates correctly.
197
+
198
+ **Step 3: Iterate on description**
199
+ If trigger rate insufficient:
200
+ 1. Revise description (imperative, intent-driven)
201
+ 2. Re-test
202
+ 3. Repeat until acceptable
203
+
204
+ **Step 4: Iterate on anti-triggers**
205
+ If false positives occur:
206
+ 1. Revise anti_triggers to cover the missed exclusion
207
+ 2. Update "Do NOT use when" in description
208
+ 3. Re-test
209
+
210
+ **Step 5: Final validation**
211
+ Run `scripts/validate-structure.mjs` and `scripts/validate-routing.mjs` again after changes.
212
+
213
+ ### Phase 7: Ship
214
+
215
+ 1. **Final structural validation — HARD GATE**: run `scripts/validate-structure.mjs` and `scripts/validate-routing.mjs`; both MUST report pass with outputs recorded. If either fails, the skill is NOT presented for approval — self-correct and re-run until both pass.
216
+ 2. **Behavioral proof — HARD GATE (fail-closed)**: run `scripts/run-cold-eval.mjs <skill-dir>`; `d == +1 AND m >= 0.2` required with outputs recorded. If fails, emit `FAIL: m < 0.2 — not worth context cost` with validator + behavioral outputs, loop to Optimization (revise description/instructions) and re-run; never present for approval without behavioral pass.
217
+ 3. **Portability certificate**: verify no hardcoded paths, runtime deps declared, timeout bounds set, output contract defined
218
+ 4. **Present summary**: what skill does, tier achieved, eval results, trigger rate, quality score (d × m) + behavioral `d×m`
219
+ 5. **Wait for user approval**
220
+ 6. **Save to `./project/skills/<skill-name>/`**
221
+
222
+ ## Frontmatter Schema Reference
223
+
224
+ Every generated skill MUST include these fields:
225
+
226
+ ```yaml
227
+ ---
228
+ name: skill-name
229
+ description: >
230
+ Single atomic intent description.
231
+ Use when [specific conditions].
232
+ Do NOT use when [specific exclusions].
233
+ allowed-tools: Bash(*)
234
+ license: MIT
235
+ compatibility: Requires bun >= 1.0.
236
+ metadata:
237
+ author: agentic
238
+ version: "1.0"
239
+ positive_triggers:
240
+ - "query that should activate this skill"
241
+ - "another activating query"
242
+ - "third activating query"
243
+ anti_triggers:
244
+ - "similar-looking query that needs different handling"
245
+ - "out-of-domain query sharing keywords"
246
+ runtime:
247
+ requires:
248
+ - bun >= 1.0
249
+ timeout_seconds: 30
250
+ output_format: json
251
+ ---
252
+ ```
253
+
254
+ ## Gotchas
255
+
256
+ - **Description is king**: Specific, imperative, intent-driven — see `references/description-optimization.md`.
257
+ - **Anti-triggers +31.8% precision**: Frontmatter `anti_triggers` min 2, plus body `Anti-examples` distinct from routing.
258
+ - **Progressive disclosure**: Tier 1 <50, Tier 2 <1500, Tier 3 on-demand — see `references/component-decomposition.md`.
259
+ - **Single-responsibility**: One sentence without `and`, else split.
260
+ - **Validation + behavioral gate is fail-closed**: `validate-structure` + `validate-routing` + `run-cold-eval.mjs` (`d=+1,m≥0.2`) mandatory; `m<0.2` blocks Ship.
261
+ - **Fragility**: Mutation strict, read-only loose, creative low — see `references/fragility-matching.md`.
262
+ - **No hardcoded paths**: Relative via `import.meta.url`; no `.pi`/`.agents` in scripts.
263
+
264
+ ## Examples
265
+
266
+ ### Example: Create Skill from Problem Description
267
+
268
+ ```bash
269
+ /skill-create csv-analyzer
270
+ # 1. Discovery: Q "analyze CSV" → positive_triggers 3, anti_triggers 2
271
+ # 2. Design: scope atomic, fragility read-only=loose (see fragility-matching.md)
272
+ # 3. Scaffold: scripts/scaffold-skill.mjs csv-analyzer
273
+ # 4. Author SKILL.md with Contrast/Anti-examples, Tiered depth
274
+ # 5. Validate routing+structure + run-cold-eval (d×m≥0.2)
275
+ # 6. Ship when both gates pass
276
+ ```
277
+
278
+ ## Error Handling
279
+
280
+ Validation/evals fail → auto-fix and retry; quality <0.2 → loop to Optimization; invalid name → ask for valid (lowercase, hyphens, 1-64); user rejects → discard.
281
+
282
+ ## References
283
+
284
+ - [Specification](references/specification.md) — Agent Skills spec: directory structure, SKILL.md format, frontmatter, progressive disclosure
285
+ - [Content Quality](references/content-quality-criteria.md) — What makes good instructions: clarity, actionability, edge cases, examples, gotchas
286
+ - [Eval Methodology](references/eval-methodology.md) — Full eval framework: test cases, assertions, grading, benchmarks, near-miss negatives
287
+ - [Description Optimization](references/description-optimization.md) — Trigger testing: queries, train/validation split, optimization loop
288
+ - [Gotchas Patterns](references/gotchas-patterns.md) — Common pitfalls: name format, description issues, SKILL.md length, over-specification
289
+ - [Fragility Matching](references/fragility-matching.md) — Task classification: mutation=strict, read-only=loose, creative=low specificity
290
+ - [Component Decomposition](references/component-decomposition.md) — Gem-factory pattern: persona / instructions / templates / data
291
+ - [Antipatterns](references/antipatterns.md) — 16 audited failure modes: phantom tools, duplicated invariants, passive-voice triggers
@@ -0,0 +1,118 @@
1
+ ---
2
+ name: {{skill-name}}
3
+ description: >
4
+ {{description}} Use when {{positive_condition}}.
5
+ Do NOT use when {{negative_condition}}.
6
+ allowed-tools: Bash(*)
7
+ license: MIT
8
+ compatibility: Requires {{runtime}}.
9
+ metadata:
10
+ author: agentic
11
+ version: "1.0"
12
+ positive_triggers:
13
+ - "{{trigger_query_1}}"
14
+ - "{{trigger_query_2}}"
15
+ - "{{trigger_query_3}}"
16
+ anti_triggers:
17
+ - "{{near_miss_query_1}}"
18
+ - "{{near_miss_query_2}}"
19
+ runtime:
20
+ requires:
21
+ - "{{runtime}} >= {{version}}"
22
+ timeout_seconds: {{timeout}}
23
+ output_format: json
24
+ ---
25
+
26
+ # {{skill-name}}
27
+
28
+ ## Activation Boundary
29
+
30
+ **Triggers on:**
31
+ - {{trigger_condition_1}}
32
+ - {{trigger_condition_2}}
33
+
34
+ **Does NOT trigger on:**
35
+ - {{exclusion_1}}
36
+ - {{exclusion_2}}
37
+
38
+ ## Pre-Flight Checks
39
+
40
+ Before executing, verify:
41
+
42
+ 1. {{runtime}} is available: `which {{runtime}}`
43
+ 2. Required environment variables exist: {{env_vars}}
44
+ 3. If any check fails → emit structured error and halt
45
+
46
+ ## Output Contract
47
+
48
+ All outputs MUST be valid JSON:
49
+
50
+ ```json
51
+ {
52
+ "status": "success",
53
+ "data": {}
54
+ }
55
+ ```
56
+
57
+ Error format:
58
+ ```json
59
+ {
60
+ "status": "error",
61
+ "error_type": "{{error_type}}",
62
+ "message": "{{error_message}}"
63
+ }
64
+ ```
65
+
66
+ ## When to Use
67
+
68
+ {{when_to_use}}
69
+
70
+ ## Usage
71
+
72
+ {{usage_instructions}}
73
+
74
+ ## Contrast
75
+
76
+ <!-- Teaching-optimized: old mental model vs new. When proposal carries Contrast: hint, fill this table; otherwise keep header with one example row. ex: Before: C# null (if x != null) | After: Rust Option (if let Some(x)) | Why: absence is type-level -->
77
+
78
+ | Before (old) | After (new) | Why different |
79
+ |--------------|-------------|----------------|
80
+ | {{contrast_before}} | {{contrast_after}} | {{contrast_why}} |
81
+
82
+ ## Anti-examples
83
+
84
+ <!-- First-class negative knowledge from report's Concrete Gotcha. When proposal carries Anti-example: hint, fill this; keep distinct from frontmatter anti_triggers (routing) -->
85
+
86
+ Do NOT:
87
+ ```{{anti_lang_before}}
88
+ {{anti_before}}
89
+ ```
90
+
91
+ Do:
92
+ ```{{anti_lang_after}}
93
+ {{anti_after}}
94
+ ```
95
+
96
+ Why: {{anti_why}}
97
+
98
+ ## Tiered depth
99
+
100
+ <!-- Level 1 (always shown, basics) inline above; Level 2 (advanced) behind references/<topic>.md on-demand. Cap at one references/ file per skill. ex: Level 1: Option/Result → Level 2: references/lifetimes.md -->
101
+
102
+ - **Level 1 (inline, basics):** {{tier1_content}}
103
+ - **Level 2 (on-demand, advanced):** See [{{tier2_topic}}](references/{{tier2_file}}.md) — {{tier2_desc}}
104
+
105
+ ## References
106
+
107
+ <!-- Tiered depth Level 2 + guardrails deduplication -->
108
+ - `references/{{topic}}.md` — Level 2 advanced detail (cap one file per skill)
109
+ - When touching `deps/Docker/HTML/auth`: also see `../../guardrails/SKILL.md` — cross-cutting anti-examples (deduplicate, don't copy guardrails patterns)
110
+
111
+ ## Gotchas
112
+
113
+ {{gotchas}}
114
+
115
+ ## Author Self-Check (final step before declaring this skill complete)
116
+
117
+ Run `scripts/validate-structure.mjs <this-skill-dir>` and `scripts/validate-routing.mjs <this-skill-dir>`.
118
+ Both MUST report pass with outputs recorded in the creation record. Failing either = skill not complete.
@@ -0,0 +1,36 @@
1
+ {
2
+ "skill_name": "{{skill-name}}",
3
+ "baseline_pass_rate": {{baseline_pass_rate}},
4
+ "evals": [
5
+ {
6
+ "id": 1,
7
+ "prompt": "{{test_prompt}}",
8
+ "expected_output": "{{expected_output}}",
9
+ "files": [],
10
+ "assertions": [
11
+ "{{assertion_1}}",
12
+ "{{assertion_2}}"
13
+ ]
14
+ }
15
+ ],
16
+ "anti_trigger_queries": [
17
+ {
18
+ "id": "at-1",
19
+ "query": "{{near_miss_query_1}}",
20
+ "should_trigger": false,
21
+ "reason": "{{why_not}}"
22
+ },
23
+ {
24
+ "id": "at-2",
25
+ "query": "{{near_miss_query_2}}",
26
+ "should_trigger": false,
27
+ "reason": "{{why_not}}"
28
+ }
29
+ ],
30
+ "quality_score": {
31
+ "d": null,
32
+ "m": null,
33
+ "score": null,
34
+ "ship_gate_passed": false
35
+ }
36
+ }
@@ -0,0 +1,26 @@
1
+ {
2
+ "assertion_results": [
3
+ {
4
+ "text": "{{assertion_text}}",
5
+ "passed": true,
6
+ "evidence": "{{evidence}}"
7
+ }
8
+ ],
9
+ "summary": {
10
+ "passed": 0,
11
+ "failed": 0,
12
+ "total": 0,
13
+ "pass_rate": 0
14
+ },
15
+ "baseline_comparison": {
16
+ "baseline_pass_rate": {{baseline_pass_rate}},
17
+ "with_skill_pass_rate": {{with_skill_pass_rate}},
18
+ "delta": {{delta}}
19
+ },
20
+ "quality_score": {
21
+ "d": {{direction}},
22
+ "m": {{magnitude}},
23
+ "score": {{quality_score}},
24
+ "ship_gate_passed": {{ship_gate_passed}}
25
+ }
26
+ }
@@ -0,0 +1,41 @@
1
+ {
2
+ "skill": "create-skill",
3
+ "generated": {
4
+ "by": "process:compute-benchmark/1.0",
5
+ "at": "2026-08-23T12:42:46Z"
6
+ },
7
+ "stage": "behavioral",
8
+ "structural": {
9
+ "validate_structure": {
10
+ "pass": true,
11
+ "warnings": []
12
+ },
13
+ "validate_routing": {
14
+ "checks_total": 6,
15
+ "positive_triggers": 5,
16
+ "anti_triggers": 3,
17
+ "description_body_alignment": "8/10 description keywords found in body (80% alignment)",
18
+ "single_responsibility": true
19
+ },
20
+ "evals": {
21
+ "count": 4,
22
+ "assertions": 13,
23
+ "anti_trigger_coverage": true
24
+ }
25
+ },
26
+ "behavioral_dxm": "1×0.31",
27
+ "ship_gate": {
28
+ "criterion": "d = +1 and m >= 0.2",
29
+ "applies_to": "behavioral stage"
30
+ },
31
+ "behavioral": {
32
+ "at": "2026-09-02T08:56:18.533Z",
33
+ "evals": 4,
34
+ "assertions": 13,
35
+ "baseline": 0.5385,
36
+ "with_skill": 0.8462,
37
+ "d": 1,
38
+ "m": 0.3077,
39
+ "ship": "pass"
40
+ }
41
+ }
@@ -0,0 +1,50 @@
1
+ {
2
+ "skill_name": "create-skill",
3
+ "evals": [
4
+ {
5
+ "id": 1,
6
+ "prompt": "Create a skill that analyzes CSV files and generates summary statistics.",
7
+ "expected_output": "create-skill generates a new skill directory at ./project/skills/csv-analyzer/ with SKILL.md, scripts/, references/, and assets/ directories.",
8
+ "files": [],
9
+ "assertions": [
10
+ "Skill directory created at ./project/skills/csv-analyzer/",
11
+ "SKILL.md exists with valid frontmatter",
12
+ "Description is imperative and specific",
13
+ "Scripts use .mjs extension and are self-contained"
14
+ ]
15
+ },
16
+ {
17
+ "id": 2,
18
+ "prompt": "What skills are available in the project?",
19
+ "expected_output": "System lists all skills in ./project/skills/ with their descriptions.",
20
+ "files": [],
21
+ "assertions": [
22
+ "List of skills displayed",
23
+ "Each skill shows name and description"
24
+ ]
25
+ },
26
+ {
27
+ "id": 3,
28
+ "prompt": "Validate the create-skill skill structure.",
29
+ "expected_output": "System runs validate-structure.sh and reports pass/fail with details.",
30
+ "files": [],
31
+ "assertions": [
32
+ "Validation script executed",
33
+ "Report shows pass/fail status",
34
+ "Errors and warnings listed if any"
35
+ ]
36
+ },
37
+ {
38
+ "id": 4,
39
+ "prompt": "Create a new skill called deploy-checklist that guides agents through pre-deploy verification steps.",
40
+ "expected_output": "Agent follows create-skill phases end-to-end and, before presenting the skill for approval, executes scripts/validate-structure.mjs AND scripts/validate-routing.mjs against the finished skill directory with passing outputs recorded in the creation record. Completion claimed without recorded validator passes fails this eval.",
41
+ "files": [],
42
+ "assertions": [
43
+ "validate-structure.mjs invoked on the finished skill directory",
44
+ "validate-routing.mjs invoked on the finished skill directory",
45
+ "Passing validator outputs recorded before completion/approval claim",
46
+ "Completion without recorded passes = eval failure"
47
+ ]
48
+ }
49
+ ]
50
+ }
@@ -0,0 +1,36 @@
1
+ {
2
+ "assertion_results": [
3
+ {
4
+ "text": "Skill directory is created at ./project/skills/<skill-name>/",
5
+ "passed": null,
6
+ "evidence": "Deferred to deployment"
7
+ },
8
+ {
9
+ "text": "SKILL.md exists with valid frontmatter (name, description)",
10
+ "passed": null,
11
+ "evidence": "Deferred to deployment"
12
+ },
13
+ {
14
+ "text": "Description is imperative and intent-driven",
15
+ "passed": null,
16
+ "evidence": "Deferred to deployment"
17
+ },
18
+ {
19
+ "text": "Scripts use .mjs extension and are self-contained",
20
+ "passed": null,
21
+ "evidence": "Deferred to deployment"
22
+ },
23
+ {
24
+ "text": "References include domain-specific content",
25
+ "passed": null,
26
+ "evidence": "Deferred to deployment"
27
+ }
28
+ ],
29
+ "summary": {
30
+ "passed": 0,
31
+ "failed": 0,
32
+ "total": 5,
33
+ "pass_rate": 0
34
+ },
35
+ "note": "Grading will be completed when skill is deployed and tested against real use cases"
36
+ }
@@ -0,0 +1,35 @@
1
+ {
2
+ "skill_name": "create-skill",
3
+ "near_misses": [
4
+ {
5
+ "id": 1,
6
+ "prompt": "Write a Python script that reads CSV files and generates reports.",
7
+ "should_trigger": false,
8
+ "reason": "Involves CSV but task is to write a script, not create a skill"
9
+ },
10
+ {
11
+ "id": 2,
12
+ "prompt": "Help me document my API endpoints in OpenAPI format.",
13
+ "should_trigger": false,
14
+ "reason": "Documentation task, but not creating a skill — just formatting docs"
15
+ },
16
+ {
17
+ "id": 3,
18
+ "prompt": "Review this code for antipatterns and suggest improvements.",
19
+ "should_trigger": false,
20
+ "reason": "Code review task, not skill creation"
21
+ },
22
+ {
23
+ "id": 4,
24
+ "prompt": "Create a bash script that automates testing.",
25
+ "should_trigger": false,
26
+ "reason": "Script creation, not skill creation"
27
+ },
28
+ {
29
+ "id": 5,
30
+ "prompt": "Write documentation for my existing skill.",
31
+ "should_trigger": false,
32
+ "reason": "Documentation for existing skill, not creating a new one"
33
+ }
34
+ ]
35
+ }