@meyverick/agentic 5.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +234 -0
- package/CHANGELOG.md +236 -0
- package/README.md +50 -0
- package/install.ts +349 -0
- package/package.json +37 -0
- package/scripts/check-deps.mjs +587 -0
- package/scripts/git-dl.mjs +100 -0
- package/skills/check/SKILL.md +108 -0
- package/skills/check/evals/benchmark.json +40 -0
- package/skills/check/evals/evals.json +38 -0
- package/skills/check/references/diagnostic-matrix.md +170 -0
- package/skills/check/references/script-anatomy.md +154 -0
- package/skills/create-skill/SKILL.md +291 -0
- package/skills/create-skill/assets/templates/SKILL.md.template +118 -0
- package/skills/create-skill/assets/templates/evals.json.template +36 -0
- package/skills/create-skill/assets/templates/grading.json.template +26 -0
- package/skills/create-skill/evals/benchmark.json +41 -0
- package/skills/create-skill/evals/evals.json +50 -0
- package/skills/create-skill/evals/grading-template.json +36 -0
- package/skills/create-skill/evals/near-misses.json +35 -0
- package/skills/create-skill/evals/trigger-queries.json +80 -0
- package/skills/create-skill/references/antipatterns.md +123 -0
- package/skills/create-skill/references/component-decomposition.md +130 -0
- package/skills/create-skill/references/content-quality-criteria.md +61 -0
- package/skills/create-skill/references/description-optimization.md +90 -0
- package/skills/create-skill/references/eval-methodology.md +100 -0
- package/skills/create-skill/references/fragility-matching.md +88 -0
- package/skills/create-skill/references/gotchas-patterns.md +80 -0
- package/skills/create-skill/references/specification.md +77 -0
- package/skills/create-skill/scripts/audit-antipatterns.mjs +164 -0
- package/skills/create-skill/scripts/compute-benchmark.mjs +111 -0
- package/skills/create-skill/scripts/run-cold-eval.mjs +118 -0
- package/skills/create-skill/scripts/scaffold-skill.mjs +86 -0
- package/skills/create-skill/scripts/validate-routing.mjs +137 -0
- package/skills/create-skill/scripts/validate-structure.mjs +223 -0
- package/skills/design-craft/SKILL.md +134 -0
- package/skills/design-craft/evals/benchmark.json +41 -0
- package/skills/design-craft/evals/evals.json +81 -0
- package/skills/design-craft/references/anti-slop-patterns.md +49 -0
- package/skills/design-craft/references/art-direction.md +89 -0
- package/skills/design-craft/references/design-engineering.md +122 -0
- package/skills/design-craft/references/motion-craft.md +124 -0
- package/skills/design-craft/references/process.md +47 -0
- package/skills/design-craft/references/review-checklist.md +121 -0
- package/skills/guardrails/SKILL.md +118 -0
- package/skills/guardrails/evals/benchmark.json +40 -0
- package/skills/guardrails/evals/evals.json +49 -0
- package/skills/guardrails/references/guardrails-patterns.md +43 -0
- package/skills/okf-docs/SKILL.md +79 -0
- package/skills/okf-docs/evals/benchmark.json +21 -0
- package/skills/okf-docs/evals/evals.json +37 -0
- package/skills/okf-docs/references/okf-spec.md +56 -0
- package/skills/okf-docs/scripts/validate-frontmatter.mjs +130 -0
- package/skills/openspec-harden/SKILL.md +138 -0
- package/skills/openspec-harden/evals/benchmark.json +40 -0
- package/skills/openspec-harden/evals/evals.json +38 -0
- package/skills/openspec-learn/SKILL.md +216 -0
- package/skills/openspec-learn/evals/benchmark.json +44 -0
- package/skills/openspec-learn/evals/evals.json +48 -0
- package/skills/openspec-learn/evals/retrieval-bench.json +27 -0
- package/skills/openspec-learn/references/conflict-handling.md +20 -0
- package/skills/openspec-learn/references/evaluation-methodology.md +126 -0
- package/skills/openspec-learn/references/examples.md +37 -0
- package/skills/openspec-learn/references/improvement-patterns.md +155 -0
- package/skills/openspec-learn/references/report-analysis.md +104 -0
- package/skills/openspec-learn/references/skill-quality.md +103 -0
- package/skills/openspec-learn/references/tool-type-detection.md +30 -0
- package/skills/openspec-report/SKILL.md +104 -0
- package/skills/openspec-report/assets/templates/assessment.md.template +84 -0
- package/skills/openspec-report/assets/templates/report.md.template +92 -0
- package/skills/openspec-report/evals/benchmark.json +44 -0
- package/skills/openspec-report/evals/evals.json +46 -0
- package/skills/qmd-research/SKILL.md +89 -0
- package/skills/qmd-research/evals/benchmark.json +40 -0
- package/skills/qmd-research/evals/evals.json +38 -0
- package/skills/qmd-research/references/index-management.md +69 -0
- package/skills/qmd-research/references/query-craft.md +82 -0
|
@@ -0,0 +1,291 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: create-skill
|
|
3
|
+
description: Create new Agent Skills from problem descriptions or instruction files. Walks through discovery, design, authoring, validation, evaluation, and optimization phases. Use when the user wants to build a new skill, create a skill from a workflow, extract a reusable pattern from a task, or set up evaluation for an existing skill. Do NOT use when the task involves general coding, debugging application code, writing project documentation, or any work unrelated to skill creation.
|
|
4
|
+
allowed-tools: Bash(*)
|
|
5
|
+
license: MIT
|
|
6
|
+
compatibility: Requires bun.
|
|
7
|
+
metadata:
|
|
8
|
+
author: agentic
|
|
9
|
+
version: "2.0"
|
|
10
|
+
positive_triggers:
|
|
11
|
+
- "create a new skill"
|
|
12
|
+
- "build a skill from a workflow"
|
|
13
|
+
- "extract a reusable pattern into a skill"
|
|
14
|
+
- "set up evaluation for an existing skill"
|
|
15
|
+
- "improve or fix an existing skill"
|
|
16
|
+
anti_triggers:
|
|
17
|
+
- "general coding task not related to skills"
|
|
18
|
+
- "debug or fix application code"
|
|
19
|
+
- "write project documentation or README"
|
|
20
|
+
runtime:
|
|
21
|
+
requires:
|
|
22
|
+
- bun >= 1.0
|
|
23
|
+
timeout_seconds: 30
|
|
24
|
+
output_format: json
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
# Create Skill
|
|
28
|
+
|
|
29
|
+
Create new Agent Skills from problem descriptions or instruction files. Fully autonomous workflow with research-informed quality standards.
|
|
30
|
+
|
|
31
|
+
## Quick Start
|
|
32
|
+
|
|
33
|
+
When the user wants to build a new skill:
|
|
34
|
+
|
|
35
|
+
1. Run `scripts/scaffold-skill.mjs <skill-name>` to create skeleton in `./project/skills/`
|
|
36
|
+
2. Follow the workflow below to fill in content
|
|
37
|
+
3. Skill ships when Phase 7 completes
|
|
38
|
+
|
|
39
|
+
When invoked from `/skill-create <name>` with instruction file:
|
|
40
|
+
|
|
41
|
+
1. Read `./skills-todo/<name>.md`
|
|
42
|
+
2. Skip Phase 1 (Discovery) — instructions have answers
|
|
43
|
+
3. Start at Phase 2 (Design) with provided decisions
|
|
44
|
+
4. Follow standard workflow from there
|
|
45
|
+
|
|
46
|
+
The workflow is fully autonomous — it runs continuously until user input is needed (Discovery answers, Ship approval). Auto-fix and retry on validation/eval failures.
|
|
47
|
+
|
|
48
|
+
## Tiers
|
|
49
|
+
|
|
50
|
+
| Tier | What's Included | When to Use |
|
|
51
|
+
|------|-----------------|-------------|
|
|
52
|
+
| **Minimal** | Structural validation + content review | Quick prototyping, low-stakes skills |
|
|
53
|
+
| **Standard** | Minimal + 2-3 test cases + manual eval | Most production skills |
|
|
54
|
+
| **Rigorous** | Standard + full eval + description optimization + quality score | High-stakes workflows, shared skills |
|
|
55
|
+
|
|
56
|
+
**Default: Rigorous.** User can request early exit to Minimal or Standard.
|
|
57
|
+
|
|
58
|
+
## Workflow
|
|
59
|
+
|
|
60
|
+
### Phase 0: Read Instructions (if provided)
|
|
61
|
+
|
|
62
|
+
If invoked from `/skill-create <name>`:
|
|
63
|
+
|
|
64
|
+
1. Read `./skills-todo/<name>.md`
|
|
65
|
+
2. Extract: problem, requirements, design decisions, gotchas, eval strategy
|
|
66
|
+
3. Skip to Phase 2 (Design) with provided information
|
|
67
|
+
|
|
68
|
+
If invoked directly (no instruction file):
|
|
69
|
+
1. Continue to Phase 1 (Discovery)
|
|
70
|
+
|
|
71
|
+
### Phase 1: Discovery
|
|
72
|
+
|
|
73
|
+
**Skip this phase if instructions were provided.**
|
|
74
|
+
|
|
75
|
+
Ask the user (open-ended, no presets):
|
|
76
|
+
|
|
77
|
+
1. What problem should this skill solve?
|
|
78
|
+
2. What domain knowledge is needed?
|
|
79
|
+
3. Which agent/harness will use it?
|
|
80
|
+
4. Existing patterns to extract from?
|
|
81
|
+
5. What does success look like?
|
|
82
|
+
6. Which tier? (default: Rigorous)
|
|
83
|
+
7. **What queries or situations SHOULD activate this skill?** (collect at least 3 examples → positive_triggers)
|
|
84
|
+
8. **What queries look similar but should NOT activate this skill?** (collect at least 2 examples → anti_triggers)
|
|
85
|
+
9. **Can you describe the skill's purpose in ONE sentence without using "and"?** (verifies atomic intent)
|
|
86
|
+
|
|
87
|
+
### Phase 2: Design
|
|
88
|
+
|
|
89
|
+
Determine:
|
|
90
|
+
|
|
91
|
+
1. **Scope**: Single atomic intent. One sentence, no "and". If compound → split into multiple skills.
|
|
92
|
+
2. **Fragility**: Mutation = strict, read-only = loose, creative = low — see `references/fragility-matching.md` (Level 2).
|
|
93
|
+
3. **Progressive disclosure**: SKILL.md vs references/ with context budget per tier:
|
|
94
|
+
- Tier 1 (frontmatter): <50 tokens
|
|
95
|
+
- Tier 2 (SKILL.md body): <1500 tokens
|
|
96
|
+
- Tier 3 (references/): on-demand only
|
|
97
|
+
4. **Components**: Persona / instructions / templates / data (not omnibus) — see `references/component-decomposition.md` (Level 2).
|
|
98
|
+
5. **Scripts**: Reusable logic to bundle (.mjs, cold/isolated, relative paths only)
|
|
99
|
+
6. **Eval strategy**: Test cases, assertions, near-miss negatives, baseline comparison
|
|
100
|
+
7. **Activation boundary**: Define what triggers and what does NOT trigger
|
|
101
|
+
8. **Runtime contract**: Required runtimes (bun/node/python), timeout_seconds, output_format (json)
|
|
102
|
+
9. **Output contract**: JSON schema for success and error responses
|
|
103
|
+
10. **Retrieval collision check**: Run hybrid retrieval over installed and project skills before scaffolding: `qmd query "<intent>" --json -n 10` (hybrid) per `project/AGENTS.md:180`, fetch hits via `qmd multi-get`, and evaluate description overlap. If semantic overlap is detected, flag the collision and suggest updating the existing skill instead of creating a new one.
|
|
104
|
+
|
|
105
|
+
### Phase 3: Authoring
|
|
106
|
+
|
|
107
|
+
1. **Scaffold**: `scripts/scaffold-skill.mjs <name>` → creates directory + SKILL.md skeleton
|
|
108
|
+
2. **Frontmatter**: name, description (imperative, specific, "Use when... Do NOT use when..."), positive_triggers (min 3), anti_triggers (min 2), allowed-tools, compatibility, runtime
|
|
109
|
+
3. **SKILL.md body sections** (in order):
|
|
110
|
+
- Activation Boundary: explicit trigger/exclusion lists
|
|
111
|
+
- Pre-Flight Checks: environment probes before execution
|
|
112
|
+
- Output Contract: JSON schema for success/error
|
|
113
|
+
- Core instructions (<500 lines, <1500 tokens)
|
|
114
|
+
- Contrast: `| Before (old) | After (new) | Why different |` table when proposal carries `Contrast:` hint (e.g., `C# null → Rust Option`); keep distinct from routing
|
|
115
|
+
- Anti-examples: `Do NOT: <before>` → `Do: <after>` with Why, when proposal carries `Anti-example:` hint from report's Concrete Gotcha; body content, NOT frontmatter `anti_triggers`
|
|
116
|
+
- Tiered depth: Level 1 basics inline, Level 2 advanced behind `references/<topic>.md` (cap one file per skill)
|
|
117
|
+
4. **Scripts**: Generate if clearly reusable (.mjs, self-contained, relative paths via import.meta.url, JSON output only)
|
|
118
|
+
5. **References**: Domain-specific docs, loaded on-demand — **Tiered depth:** Level 1 basics inline in SKILL.md, Level 2 advanced behind `references/<topic>.md`; cap at one `references/` file per skill; link explicitly from SKILL.md
|
|
119
|
+
6. **Templates**: Output shapes, examples
|
|
120
|
+
7. **Single source**: Define each rule ONCE, reference everywhere else
|
|
121
|
+
8. **Portability**: No absolute paths (/home/, /root/, C:\), no harness-specific dirs (.pi/, .agents/) in executable code
|
|
122
|
+
|
|
123
|
+
### Phase 4: Validation (mandatory)
|
|
124
|
+
|
|
125
|
+
**Output contract**: all three validators emit one shared JSON envelope — `{target, pass, checks:[{id, status: PASS|FAIL|WARN|SKIP, detail}], summary}` — so downstream tooling branches deterministically on `pass` and `checks[].status` regardless of which script produced it.
|
|
126
|
+
|
|
127
|
+
**Step 1: Structural validation**
|
|
128
|
+
```bash
|
|
129
|
+
scripts/validate-structure.mjs <skill-dir>
|
|
130
|
+
```
|
|
131
|
+
Checks: name format, description format, directory structure, file references, positive_triggers (min 3), anti_triggers (min 2), "Use when" phrasing, "Do NOT use when" phrasing, no compound intent, no hardcoded paths, runtime declared when scripts exist.
|
|
132
|
+
|
|
133
|
+
**Evidence requirement**: record each validator's output (stdout or pass summary) in the creation session's tasks or summary before proceeding to the next step. An unrecorded validation counts as not performed.
|
|
134
|
+
|
|
135
|
+
**Step 2: Semantic routing validation**
|
|
136
|
+
```bash
|
|
137
|
+
scripts/validate-routing.mjs <skill-dir>
|
|
138
|
+
```
|
|
139
|
+
Checks: positive_triggers coverage, anti_triggers coverage, description-body alignment, single-responsibility verification.
|
|
140
|
+
|
|
141
|
+
**Step 3: Content review**
|
|
142
|
+
Agent assesses:
|
|
143
|
+
- Description specificity (imperative, intent-driven, single atomic intent)
|
|
144
|
+
- Instruction clarity (actionable, not vague)
|
|
145
|
+
- Progressive disclosure (SKILL.md <500 lines, <1500 tokens)
|
|
146
|
+
- Gotchas present (environment-specific facts)
|
|
147
|
+
- Fragility matching (strict for mutation, loose for read-only)
|
|
148
|
+
- Activation boundary completeness (all triggers documented)
|
|
149
|
+
|
|
150
|
+
**Step 4: Antipattern self-audit**
|
|
151
|
+
```bash
|
|
152
|
+
scripts/audit-antipatterns.mjs <skill-dir>
|
|
153
|
+
```
|
|
154
|
+
Checks: phantom tools, duplicated invariants, passive-voice triggers, prose bloat, single-file omnibus, vague success bars, multi-domain descriptions (A17), missing activation boundary (A18), hardcoded paths (A19), context budget violation (A20).
|
|
155
|
+
|
|
156
|
+
Self-correct any issues before proceeding.
|
|
157
|
+
|
|
158
|
+
### Phase 5: Evaluation (mandatory)
|
|
159
|
+
|
|
160
|
+
**Step 1-5: Eval loop** — 2-3 cases + near-miss negatives; measure baseline (no skill) then with-skill (grade PASS/FAIL, record timing); near-miss 100% for mutation.
|
|
161
|
+
|
|
162
|
+
**Step 6: Compute benchmarks**
|
|
163
|
+
```bash
|
|
164
|
+
scripts/compute-benchmark.mjs <eval-dir>
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
**Step 6b: Cold-agent behavioral proof (mandatory)**
|
|
168
|
+
```bash
|
|
169
|
+
scripts/run-cold-eval.mjs <skill-dir>
|
|
170
|
+
```
|
|
171
|
+
Runs `evals/evals.json` cold A/B (without vs with skill), computes `d = sign(with - baseline)`, `m = |with - baseline|`, emits unified envelope `{target, pass, checks, summary}` + `behavioral: {at, baseline, with_skill, d, m, ship}` (30s timeout, JSON only). Writes `evals/benchmark.json` with `stage: behavioral` block (`{at, baseline, with_skill, d, m, ship}`) replacing `pending_cold_agent_run`; record outputs in creation session. Reuses `compute-benchmark.mjs` logic; no network, deterministic.
|
|
172
|
+
|
|
173
|
+
**Step 7: Calculate quality score**
|
|
174
|
+
```
|
|
175
|
+
d = direction (+1 if with_skill > baseline, -1 if lower, 0 if equal)
|
|
176
|
+
m = magnitude = |with_skill_pass_rate - baseline_pass_rate|
|
|
177
|
+
quality_score = d × m
|
|
178
|
+
```
|
|
179
|
+
Ship gate: d must be +1 AND m must be >= 0.2 (20% improvement over baseline) — now proven by `run-cold-eval.mjs` behavioral block.
|
|
180
|
+
|
|
181
|
+
**Step 8: Iterate**
|
|
182
|
+
If quality insufficient:
|
|
183
|
+
1. Analyze failures
|
|
184
|
+
2. Fix instructions
|
|
185
|
+
3. Re-run evals
|
|
186
|
+
4. Repeat until plateau
|
|
187
|
+
|
|
188
|
+
### Phase 6: Optimization (mandatory)
|
|
189
|
+
|
|
190
|
+
**Step 1: Create trigger test queries**
|
|
191
|
+
- 10-20 queries (mix should/shouldn't trigger + near-misses)
|
|
192
|
+
- Include queries matching each positive_trigger
|
|
193
|
+
- Include queries matching each anti_trigger
|
|
194
|
+
|
|
195
|
+
**Step 2: Test description triggering**
|
|
196
|
+
Run each query, check if skill activates correctly.
|
|
197
|
+
|
|
198
|
+
**Step 3: Iterate on description**
|
|
199
|
+
If trigger rate insufficient:
|
|
200
|
+
1. Revise description (imperative, intent-driven)
|
|
201
|
+
2. Re-test
|
|
202
|
+
3. Repeat until acceptable
|
|
203
|
+
|
|
204
|
+
**Step 4: Iterate on anti-triggers**
|
|
205
|
+
If false positives occur:
|
|
206
|
+
1. Revise anti_triggers to cover the missed exclusion
|
|
207
|
+
2. Update "Do NOT use when" in description
|
|
208
|
+
3. Re-test
|
|
209
|
+
|
|
210
|
+
**Step 5: Final validation**
|
|
211
|
+
Run `scripts/validate-structure.mjs` and `scripts/validate-routing.mjs` again after changes.
|
|
212
|
+
|
|
213
|
+
### Phase 7: Ship
|
|
214
|
+
|
|
215
|
+
1. **Final structural validation — HARD GATE**: run `scripts/validate-structure.mjs` and `scripts/validate-routing.mjs`; both MUST report pass with outputs recorded. If either fails, the skill is NOT presented for approval — self-correct and re-run until both pass.
|
|
216
|
+
2. **Behavioral proof — HARD GATE (fail-closed)**: run `scripts/run-cold-eval.mjs <skill-dir>`; `d == +1 AND m >= 0.2` required with outputs recorded. If fails, emit `FAIL: m < 0.2 — not worth context cost` with validator + behavioral outputs, loop to Optimization (revise description/instructions) and re-run; never present for approval without behavioral pass.
|
|
217
|
+
3. **Portability certificate**: verify no hardcoded paths, runtime deps declared, timeout bounds set, output contract defined
|
|
218
|
+
4. **Present summary**: what skill does, tier achieved, eval results, trigger rate, quality score (d × m) + behavioral `d×m`
|
|
219
|
+
5. **Wait for user approval**
|
|
220
|
+
6. **Save to `./project/skills/<skill-name>/`**
|
|
221
|
+
|
|
222
|
+
## Frontmatter Schema Reference
|
|
223
|
+
|
|
224
|
+
Every generated skill MUST include these fields:
|
|
225
|
+
|
|
226
|
+
```yaml
|
|
227
|
+
---
|
|
228
|
+
name: skill-name
|
|
229
|
+
description: >
|
|
230
|
+
Single atomic intent description.
|
|
231
|
+
Use when [specific conditions].
|
|
232
|
+
Do NOT use when [specific exclusions].
|
|
233
|
+
allowed-tools: Bash(*)
|
|
234
|
+
license: MIT
|
|
235
|
+
compatibility: Requires bun >= 1.0.
|
|
236
|
+
metadata:
|
|
237
|
+
author: agentic
|
|
238
|
+
version: "1.0"
|
|
239
|
+
positive_triggers:
|
|
240
|
+
- "query that should activate this skill"
|
|
241
|
+
- "another activating query"
|
|
242
|
+
- "third activating query"
|
|
243
|
+
anti_triggers:
|
|
244
|
+
- "similar-looking query that needs different handling"
|
|
245
|
+
- "out-of-domain query sharing keywords"
|
|
246
|
+
runtime:
|
|
247
|
+
requires:
|
|
248
|
+
- bun >= 1.0
|
|
249
|
+
timeout_seconds: 30
|
|
250
|
+
output_format: json
|
|
251
|
+
---
|
|
252
|
+
```
|
|
253
|
+
|
|
254
|
+
## Gotchas
|
|
255
|
+
|
|
256
|
+
- **Description is king**: Specific, imperative, intent-driven — see `references/description-optimization.md`.
|
|
257
|
+
- **Anti-triggers +31.8% precision**: Frontmatter `anti_triggers` min 2, plus body `Anti-examples` distinct from routing.
|
|
258
|
+
- **Progressive disclosure**: Tier 1 <50, Tier 2 <1500, Tier 3 on-demand — see `references/component-decomposition.md`.
|
|
259
|
+
- **Single-responsibility**: One sentence without `and`, else split.
|
|
260
|
+
- **Validation + behavioral gate is fail-closed**: `validate-structure` + `validate-routing` + `run-cold-eval.mjs` (`d=+1,m≥0.2`) mandatory; `m<0.2` blocks Ship.
|
|
261
|
+
- **Fragility**: Mutation strict, read-only loose, creative low — see `references/fragility-matching.md`.
|
|
262
|
+
- **No hardcoded paths**: Relative via `import.meta.url`; no `.pi`/`.agents` in scripts.
|
|
263
|
+
|
|
264
|
+
## Examples
|
|
265
|
+
|
|
266
|
+
### Example: Create Skill from Problem Description
|
|
267
|
+
|
|
268
|
+
```bash
|
|
269
|
+
/skill-create csv-analyzer
|
|
270
|
+
# 1. Discovery: Q "analyze CSV" → positive_triggers 3, anti_triggers 2
|
|
271
|
+
# 2. Design: scope atomic, fragility read-only=loose (see fragility-matching.md)
|
|
272
|
+
# 3. Scaffold: scripts/scaffold-skill.mjs csv-analyzer
|
|
273
|
+
# 4. Author SKILL.md with Contrast/Anti-examples, Tiered depth
|
|
274
|
+
# 5. Validate routing+structure + run-cold-eval (d×m≥0.2)
|
|
275
|
+
# 6. Ship when both gates pass
|
|
276
|
+
```
|
|
277
|
+
|
|
278
|
+
## Error Handling
|
|
279
|
+
|
|
280
|
+
Validation/evals fail → auto-fix and retry; quality <0.2 → loop to Optimization; invalid name → ask for valid (lowercase, hyphens, 1-64); user rejects → discard.
|
|
281
|
+
|
|
282
|
+
## References
|
|
283
|
+
|
|
284
|
+
- [Specification](references/specification.md) — Agent Skills spec: directory structure, SKILL.md format, frontmatter, progressive disclosure
|
|
285
|
+
- [Content Quality](references/content-quality-criteria.md) — What makes good instructions: clarity, actionability, edge cases, examples, gotchas
|
|
286
|
+
- [Eval Methodology](references/eval-methodology.md) — Full eval framework: test cases, assertions, grading, benchmarks, near-miss negatives
|
|
287
|
+
- [Description Optimization](references/description-optimization.md) — Trigger testing: queries, train/validation split, optimization loop
|
|
288
|
+
- [Gotchas Patterns](references/gotchas-patterns.md) — Common pitfalls: name format, description issues, SKILL.md length, over-specification
|
|
289
|
+
- [Fragility Matching](references/fragility-matching.md) — Task classification: mutation=strict, read-only=loose, creative=low specificity
|
|
290
|
+
- [Component Decomposition](references/component-decomposition.md) — Gem-factory pattern: persona / instructions / templates / data
|
|
291
|
+
- [Antipatterns](references/antipatterns.md) — 16 audited failure modes: phantom tools, duplicated invariants, passive-voice triggers
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: {{skill-name}}
|
|
3
|
+
description: >
|
|
4
|
+
{{description}} Use when {{positive_condition}}.
|
|
5
|
+
Do NOT use when {{negative_condition}}.
|
|
6
|
+
allowed-tools: Bash(*)
|
|
7
|
+
license: MIT
|
|
8
|
+
compatibility: Requires {{runtime}}.
|
|
9
|
+
metadata:
|
|
10
|
+
author: agentic
|
|
11
|
+
version: "1.0"
|
|
12
|
+
positive_triggers:
|
|
13
|
+
- "{{trigger_query_1}}"
|
|
14
|
+
- "{{trigger_query_2}}"
|
|
15
|
+
- "{{trigger_query_3}}"
|
|
16
|
+
anti_triggers:
|
|
17
|
+
- "{{near_miss_query_1}}"
|
|
18
|
+
- "{{near_miss_query_2}}"
|
|
19
|
+
runtime:
|
|
20
|
+
requires:
|
|
21
|
+
- "{{runtime}} >= {{version}}"
|
|
22
|
+
timeout_seconds: {{timeout}}
|
|
23
|
+
output_format: json
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
# {{skill-name}}
|
|
27
|
+
|
|
28
|
+
## Activation Boundary
|
|
29
|
+
|
|
30
|
+
**Triggers on:**
|
|
31
|
+
- {{trigger_condition_1}}
|
|
32
|
+
- {{trigger_condition_2}}
|
|
33
|
+
|
|
34
|
+
**Does NOT trigger on:**
|
|
35
|
+
- {{exclusion_1}}
|
|
36
|
+
- {{exclusion_2}}
|
|
37
|
+
|
|
38
|
+
## Pre-Flight Checks
|
|
39
|
+
|
|
40
|
+
Before executing, verify:
|
|
41
|
+
|
|
42
|
+
1. {{runtime}} is available: `which {{runtime}}`
|
|
43
|
+
2. Required environment variables exist: {{env_vars}}
|
|
44
|
+
3. If any check fails → emit structured error and halt
|
|
45
|
+
|
|
46
|
+
## Output Contract
|
|
47
|
+
|
|
48
|
+
All outputs MUST be valid JSON:
|
|
49
|
+
|
|
50
|
+
```json
|
|
51
|
+
{
|
|
52
|
+
"status": "success",
|
|
53
|
+
"data": {}
|
|
54
|
+
}
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Error format:
|
|
58
|
+
```json
|
|
59
|
+
{
|
|
60
|
+
"status": "error",
|
|
61
|
+
"error_type": "{{error_type}}",
|
|
62
|
+
"message": "{{error_message}}"
|
|
63
|
+
}
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## When to Use
|
|
67
|
+
|
|
68
|
+
{{when_to_use}}
|
|
69
|
+
|
|
70
|
+
## Usage
|
|
71
|
+
|
|
72
|
+
{{usage_instructions}}
|
|
73
|
+
|
|
74
|
+
## Contrast
|
|
75
|
+
|
|
76
|
+
<!-- Teaching-optimized: old mental model vs new. When proposal carries Contrast: hint, fill this table; otherwise keep header with one example row. ex: Before: C# null (if x != null) | After: Rust Option (if let Some(x)) | Why: absence is type-level -->
|
|
77
|
+
|
|
78
|
+
| Before (old) | After (new) | Why different |
|
|
79
|
+
|--------------|-------------|----------------|
|
|
80
|
+
| {{contrast_before}} | {{contrast_after}} | {{contrast_why}} |
|
|
81
|
+
|
|
82
|
+
## Anti-examples
|
|
83
|
+
|
|
84
|
+
<!-- First-class negative knowledge from report's Concrete Gotcha. When proposal carries Anti-example: hint, fill this; keep distinct from frontmatter anti_triggers (routing) -->
|
|
85
|
+
|
|
86
|
+
Do NOT:
|
|
87
|
+
```{{anti_lang_before}}
|
|
88
|
+
{{anti_before}}
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Do:
|
|
92
|
+
```{{anti_lang_after}}
|
|
93
|
+
{{anti_after}}
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Why: {{anti_why}}
|
|
97
|
+
|
|
98
|
+
## Tiered depth
|
|
99
|
+
|
|
100
|
+
<!-- Level 1 (always shown, basics) inline above; Level 2 (advanced) behind references/<topic>.md on-demand. Cap at one references/ file per skill. ex: Level 1: Option/Result → Level 2: references/lifetimes.md -->
|
|
101
|
+
|
|
102
|
+
- **Level 1 (inline, basics):** {{tier1_content}}
|
|
103
|
+
- **Level 2 (on-demand, advanced):** See [{{tier2_topic}}](references/{{tier2_file}}.md) — {{tier2_desc}}
|
|
104
|
+
|
|
105
|
+
## References
|
|
106
|
+
|
|
107
|
+
<!-- Tiered depth Level 2 + guardrails deduplication -->
|
|
108
|
+
- `references/{{topic}}.md` — Level 2 advanced detail (cap one file per skill)
|
|
109
|
+
- When touching `deps/Docker/HTML/auth`: also see `../../guardrails/SKILL.md` — cross-cutting anti-examples (deduplicate, don't copy guardrails patterns)
|
|
110
|
+
|
|
111
|
+
## Gotchas
|
|
112
|
+
|
|
113
|
+
{{gotchas}}
|
|
114
|
+
|
|
115
|
+
## Author Self-Check (final step before declaring this skill complete)
|
|
116
|
+
|
|
117
|
+
Run `scripts/validate-structure.mjs <this-skill-dir>` and `scripts/validate-routing.mjs <this-skill-dir>`.
|
|
118
|
+
Both MUST report pass with outputs recorded in the creation record. Failing either = skill not complete.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill_name": "{{skill-name}}",
|
|
3
|
+
"baseline_pass_rate": {{baseline_pass_rate}},
|
|
4
|
+
"evals": [
|
|
5
|
+
{
|
|
6
|
+
"id": 1,
|
|
7
|
+
"prompt": "{{test_prompt}}",
|
|
8
|
+
"expected_output": "{{expected_output}}",
|
|
9
|
+
"files": [],
|
|
10
|
+
"assertions": [
|
|
11
|
+
"{{assertion_1}}",
|
|
12
|
+
"{{assertion_2}}"
|
|
13
|
+
]
|
|
14
|
+
}
|
|
15
|
+
],
|
|
16
|
+
"anti_trigger_queries": [
|
|
17
|
+
{
|
|
18
|
+
"id": "at-1",
|
|
19
|
+
"query": "{{near_miss_query_1}}",
|
|
20
|
+
"should_trigger": false,
|
|
21
|
+
"reason": "{{why_not}}"
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"id": "at-2",
|
|
25
|
+
"query": "{{near_miss_query_2}}",
|
|
26
|
+
"should_trigger": false,
|
|
27
|
+
"reason": "{{why_not}}"
|
|
28
|
+
}
|
|
29
|
+
],
|
|
30
|
+
"quality_score": {
|
|
31
|
+
"d": null,
|
|
32
|
+
"m": null,
|
|
33
|
+
"score": null,
|
|
34
|
+
"ship_gate_passed": false
|
|
35
|
+
}
|
|
36
|
+
}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
{
|
|
2
|
+
"assertion_results": [
|
|
3
|
+
{
|
|
4
|
+
"text": "{{assertion_text}}",
|
|
5
|
+
"passed": true,
|
|
6
|
+
"evidence": "{{evidence}}"
|
|
7
|
+
}
|
|
8
|
+
],
|
|
9
|
+
"summary": {
|
|
10
|
+
"passed": 0,
|
|
11
|
+
"failed": 0,
|
|
12
|
+
"total": 0,
|
|
13
|
+
"pass_rate": 0
|
|
14
|
+
},
|
|
15
|
+
"baseline_comparison": {
|
|
16
|
+
"baseline_pass_rate": {{baseline_pass_rate}},
|
|
17
|
+
"with_skill_pass_rate": {{with_skill_pass_rate}},
|
|
18
|
+
"delta": {{delta}}
|
|
19
|
+
},
|
|
20
|
+
"quality_score": {
|
|
21
|
+
"d": {{direction}},
|
|
22
|
+
"m": {{magnitude}},
|
|
23
|
+
"score": {{quality_score}},
|
|
24
|
+
"ship_gate_passed": {{ship_gate_passed}}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill": "create-skill",
|
|
3
|
+
"generated": {
|
|
4
|
+
"by": "process:compute-benchmark/1.0",
|
|
5
|
+
"at": "2026-08-23T12:42:46Z"
|
|
6
|
+
},
|
|
7
|
+
"stage": "behavioral",
|
|
8
|
+
"structural": {
|
|
9
|
+
"validate_structure": {
|
|
10
|
+
"pass": true,
|
|
11
|
+
"warnings": []
|
|
12
|
+
},
|
|
13
|
+
"validate_routing": {
|
|
14
|
+
"checks_total": 6,
|
|
15
|
+
"positive_triggers": 5,
|
|
16
|
+
"anti_triggers": 3,
|
|
17
|
+
"description_body_alignment": "8/10 description keywords found in body (80% alignment)",
|
|
18
|
+
"single_responsibility": true
|
|
19
|
+
},
|
|
20
|
+
"evals": {
|
|
21
|
+
"count": 4,
|
|
22
|
+
"assertions": 13,
|
|
23
|
+
"anti_trigger_coverage": true
|
|
24
|
+
}
|
|
25
|
+
},
|
|
26
|
+
"behavioral_dxm": "1×0.31",
|
|
27
|
+
"ship_gate": {
|
|
28
|
+
"criterion": "d = +1 and m >= 0.2",
|
|
29
|
+
"applies_to": "behavioral stage"
|
|
30
|
+
},
|
|
31
|
+
"behavioral": {
|
|
32
|
+
"at": "2026-09-02T08:56:18.533Z",
|
|
33
|
+
"evals": 4,
|
|
34
|
+
"assertions": 13,
|
|
35
|
+
"baseline": 0.5385,
|
|
36
|
+
"with_skill": 0.8462,
|
|
37
|
+
"d": 1,
|
|
38
|
+
"m": 0.3077,
|
|
39
|
+
"ship": "pass"
|
|
40
|
+
}
|
|
41
|
+
}
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill_name": "create-skill",
|
|
3
|
+
"evals": [
|
|
4
|
+
{
|
|
5
|
+
"id": 1,
|
|
6
|
+
"prompt": "Create a skill that analyzes CSV files and generates summary statistics.",
|
|
7
|
+
"expected_output": "create-skill generates a new skill directory at ./project/skills/csv-analyzer/ with SKILL.md, scripts/, references/, and assets/ directories.",
|
|
8
|
+
"files": [],
|
|
9
|
+
"assertions": [
|
|
10
|
+
"Skill directory created at ./project/skills/csv-analyzer/",
|
|
11
|
+
"SKILL.md exists with valid frontmatter",
|
|
12
|
+
"Description is imperative and specific",
|
|
13
|
+
"Scripts use .mjs extension and are self-contained"
|
|
14
|
+
]
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
"id": 2,
|
|
18
|
+
"prompt": "What skills are available in the project?",
|
|
19
|
+
"expected_output": "System lists all skills in ./project/skills/ with their descriptions.",
|
|
20
|
+
"files": [],
|
|
21
|
+
"assertions": [
|
|
22
|
+
"List of skills displayed",
|
|
23
|
+
"Each skill shows name and description"
|
|
24
|
+
]
|
|
25
|
+
},
|
|
26
|
+
{
|
|
27
|
+
"id": 3,
|
|
28
|
+
"prompt": "Validate the create-skill skill structure.",
|
|
29
|
+
"expected_output": "System runs validate-structure.sh and reports pass/fail with details.",
|
|
30
|
+
"files": [],
|
|
31
|
+
"assertions": [
|
|
32
|
+
"Validation script executed",
|
|
33
|
+
"Report shows pass/fail status",
|
|
34
|
+
"Errors and warnings listed if any"
|
|
35
|
+
]
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": 4,
|
|
39
|
+
"prompt": "Create a new skill called deploy-checklist that guides agents through pre-deploy verification steps.",
|
|
40
|
+
"expected_output": "Agent follows create-skill phases end-to-end and, before presenting the skill for approval, executes scripts/validate-structure.mjs AND scripts/validate-routing.mjs against the finished skill directory with passing outputs recorded in the creation record. Completion claimed without recorded validator passes fails this eval.",
|
|
41
|
+
"files": [],
|
|
42
|
+
"assertions": [
|
|
43
|
+
"validate-structure.mjs invoked on the finished skill directory",
|
|
44
|
+
"validate-routing.mjs invoked on the finished skill directory",
|
|
45
|
+
"Passing validator outputs recorded before completion/approval claim",
|
|
46
|
+
"Completion without recorded passes = eval failure"
|
|
47
|
+
]
|
|
48
|
+
}
|
|
49
|
+
]
|
|
50
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
{
|
|
2
|
+
"assertion_results": [
|
|
3
|
+
{
|
|
4
|
+
"text": "Skill directory is created at ./project/skills/<skill-name>/",
|
|
5
|
+
"passed": null,
|
|
6
|
+
"evidence": "Deferred to deployment"
|
|
7
|
+
},
|
|
8
|
+
{
|
|
9
|
+
"text": "SKILL.md exists with valid frontmatter (name, description)",
|
|
10
|
+
"passed": null,
|
|
11
|
+
"evidence": "Deferred to deployment"
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"text": "Description is imperative and intent-driven",
|
|
15
|
+
"passed": null,
|
|
16
|
+
"evidence": "Deferred to deployment"
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"text": "Scripts use .mjs extension and are self-contained",
|
|
20
|
+
"passed": null,
|
|
21
|
+
"evidence": "Deferred to deployment"
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"text": "References include domain-specific content",
|
|
25
|
+
"passed": null,
|
|
26
|
+
"evidence": "Deferred to deployment"
|
|
27
|
+
}
|
|
28
|
+
],
|
|
29
|
+
"summary": {
|
|
30
|
+
"passed": 0,
|
|
31
|
+
"failed": 0,
|
|
32
|
+
"total": 5,
|
|
33
|
+
"pass_rate": 0
|
|
34
|
+
},
|
|
35
|
+
"note": "Grading will be completed when skill is deployed and tested against real use cases"
|
|
36
|
+
}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill_name": "create-skill",
|
|
3
|
+
"near_misses": [
|
|
4
|
+
{
|
|
5
|
+
"id": 1,
|
|
6
|
+
"prompt": "Write a Python script that reads CSV files and generates reports.",
|
|
7
|
+
"should_trigger": false,
|
|
8
|
+
"reason": "Involves CSV but task is to write a script, not create a skill"
|
|
9
|
+
},
|
|
10
|
+
{
|
|
11
|
+
"id": 2,
|
|
12
|
+
"prompt": "Help me document my API endpoints in OpenAPI format.",
|
|
13
|
+
"should_trigger": false,
|
|
14
|
+
"reason": "Documentation task, but not creating a skill — just formatting docs"
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
"id": 3,
|
|
18
|
+
"prompt": "Review this code for antipatterns and suggest improvements.",
|
|
19
|
+
"should_trigger": false,
|
|
20
|
+
"reason": "Code review task, not skill creation"
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
"id": 4,
|
|
24
|
+
"prompt": "Create a bash script that automates testing.",
|
|
25
|
+
"should_trigger": false,
|
|
26
|
+
"reason": "Script creation, not skill creation"
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"id": 5,
|
|
30
|
+
"prompt": "Write documentation for my existing skill.",
|
|
31
|
+
"should_trigger": false,
|
|
32
|
+
"reason": "Documentation for existing skill, not creating a new one"
|
|
33
|
+
}
|
|
34
|
+
]
|
|
35
|
+
}
|