@phuc1403/musketeer 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/README.md +49 -49
  2. package/manifest.json +333 -301
  3. package/package.json +1 -1
  4. package/template/.claude/agents/code-reviewer.md +182 -166
  5. package/template/.claude/hooks/git-skill-reminder.cjs +53 -0
  6. package/template/.claude/hooks/inject-design-docs.cjs +13 -13
  7. package/template/.claude/hooks/inject-ubiquitous-language.cjs +52 -0
  8. package/template/.claude/hooks/lib/colors.cjs +180 -122
  9. package/template/.claude/hooks/lib/transcript-parser.cjs +300 -277
  10. package/template/.claude/skills/code-review/SKILL.md +201 -54
  11. package/template/.claude/skills/code-review/references/checklist-workflow.md +96 -0
  12. package/template/.claude/skills/code-review/references/checklists/api.md +52 -52
  13. package/template/.claude/skills/code-review/references/checklists/base.md +100 -100
  14. package/template/.claude/skills/code-review/references/checklists/web-app.md +54 -54
  15. package/template/.claude/skills/code-review/references/code-review-reception.md +113 -0
  16. package/template/.claude/skills/code-review/references/codebase-scan-workflow.md +30 -0
  17. package/template/.claude/skills/code-review/references/edge-case-scouting.md +119 -0
  18. package/template/.claude/skills/code-review/references/input-mode-resolution.md +135 -0
  19. package/template/.claude/skills/code-review/references/parallel-review-workflow.md +76 -0
  20. package/template/.claude/skills/code-review/references/requesting-code-review.md +116 -0
  21. package/template/.claude/skills/code-review/references/spec-compliance-review.md +43 -0
  22. package/template/.claude/skills/code-review/references/task-management-reviews.md +140 -0
  23. package/template/.claude/skills/code-review/references/verification-before-completion.md +139 -0
  24. package/template/.claude/skills/context-map/SKILL.md +1 -1
  25. package/template/.claude/skills/git/SKILL.md +131 -115
  26. package/template/.claude/skills/git/references/branch-management.md +88 -88
  27. package/template/.claude/skills/git/references/commit-standards.md +46 -46
  28. package/template/.claude/skills/git/references/context-efficiency.md +54 -0
  29. package/template/.claude/skills/git/references/gh-cli-guide.md +109 -109
  30. package/template/.claude/skills/git/references/safety-protocols.md +69 -69
  31. package/template/.claude/skills/git/references/workflow-commit.md +58 -58
  32. package/template/.claude/skills/git/references/workflow-merge-pr.md +136 -0
  33. package/template/.claude/skills/git/references/workflow-merge.md +48 -48
  34. package/template/.claude/skills/git/references/workflow-pr.md +58 -58
  35. package/template/.claude/skills/git/references/workflow-push.md +52 -52
  36. package/template/.claude/skills/knowledge-crunching/SKILL.md +56 -92
  37. package/template/.claude/skills/knowledge-crunching/assets/ubiquitous-language.template.md +3 -0
  38. package/template/.claude/skills/skill-creator/LICENSE.txt +201 -201
  39. package/template/.claude/skills/skill-creator/SKILL.md +154 -149
  40. package/template/.claude/skills/skill-creator/agents/analyzer.md +274 -274
  41. package/template/.claude/skills/skill-creator/agents/comparator.md +202 -202
  42. package/template/.claude/skills/skill-creator/agents/grader.md +223 -223
  43. package/template/.claude/skills/skill-creator/assets/eval_review.html +146 -146
  44. package/template/.claude/skills/skill-creator/eval-viewer/generate_review.py +471 -471
  45. package/template/.claude/skills/skill-creator/eval-viewer/viewer.html +1325 -1325
  46. package/template/.claude/skills/skill-creator/references/benchmark-optimization-guide.md +86 -86
  47. package/template/.claude/skills/skill-creator/references/distribution-guide.md +79 -79
  48. package/template/.claude/skills/skill-creator/references/eval-infrastructure-guide.md +129 -129
  49. package/template/.claude/skills/skill-creator/references/eval-schemas.md +121 -121
  50. package/template/.claude/skills/skill-creator/references/mcp-skills-integration.md +71 -71
  51. package/template/.claude/skills/skill-creator/references/metadata-quality-criteria.md +94 -94
  52. package/template/.claude/skills/skill-creator/references/plugin-marketplace-hosting.md +104 -104
  53. package/template/.claude/skills/skill-creator/references/plugin-marketplace-overview.md +89 -89
  54. package/template/.claude/skills/skill-creator/references/plugin-marketplace-schema.md +93 -93
  55. package/template/.claude/skills/skill-creator/references/plugin-marketplace-sources.md +103 -103
  56. package/template/.claude/skills/skill-creator/references/plugin-marketplace-troubleshooting.md +76 -76
  57. package/template/.claude/skills/skill-creator/references/script-quality-criteria.md +106 -106
  58. package/template/.claude/skills/skill-creator/references/skill-anatomy-and-requirements.md +77 -77
  59. package/template/.claude/skills/skill-creator/references/skill-creation-workflow.md +152 -151
  60. package/template/.claude/skills/skill-creator/references/skill-design-patterns.md +75 -75
  61. package/template/.claude/skills/skill-creator/references/skillmark-benchmark-criteria.md +102 -102
  62. package/template/.claude/skills/skill-creator/references/structure-organization-criteria.md +114 -114
  63. package/template/.claude/skills/skill-creator/references/testing-and-iteration.md +78 -78
  64. package/template/.claude/skills/skill-creator/references/token-efficiency-criteria.md +74 -74
  65. package/template/.claude/skills/skill-creator/references/troubleshooting-guide.md +81 -81
  66. package/template/.claude/skills/skill-creator/references/validation-checklist.md +83 -83
  67. package/template/.claude/skills/skill-creator/references/writing-effective-instructions.md +88 -88
  68. package/template/.claude/skills/skill-creator/references/yaml-frontmatter-reference.md +92 -92
  69. package/template/.claude/skills/skill-creator/scripts/aggregate_benchmark.py +401 -401
  70. package/template/.claude/skills/skill-creator/scripts/encoding_utils.py +36 -36
  71. package/template/.claude/skills/skill-creator/scripts/generate_report.py +326 -326
  72. package/template/.claude/skills/skill-creator/scripts/improve_description.py +248 -248
  73. package/template/.claude/skills/skill-creator/scripts/init_skill.py +360 -360
  74. package/template/.claude/skills/skill-creator/scripts/package_skill.py +143 -143
  75. package/template/.claude/skills/skill-creator/scripts/quick_validate.py +110 -110
  76. package/template/.claude/skills/skill-creator/scripts/run_eval.py +310 -310
  77. package/template/.claude/skills/skill-creator/scripts/run_loop.py +332 -332
  78. package/template/.claude/skills/skill-creator/scripts/utils.py +47 -47
  79. package/template/.claude/statusline.cjs +0 -0
  80. package/template/.claude/hooks/inject-context.cjs +0 -52
  81. package/template/.claude/skills/code-review/references/adversarial-review.md +0 -223
  82. package/template/.claude/skills/knowledge-crunching/assets/context.template.md +0 -59
  83. package/template/.claude/skills/knowledge-crunching/references/crunching-dialogue.md +0 -113
  84. /package/template/.claude/hooks/{usage-context-awareness.cjs → usage-quota-cache-refresh.cjs} +0 -0
@@ -1,77 +1,77 @@
1
- # Skill Anatomy & Requirements
2
-
3
- ## Directory Structure
4
-
5
- ```
6
- .claude/skills/
7
- └── skill-name/
8
- ├── SKILL.md (required, <300 lines)
9
- │ ├── YAML frontmatter (name, description required)
10
- │ └── Markdown instructions
11
- └── Bundled Resources (optional)
12
- ├── scripts/ Executable code (Python/Node.js)
13
- ├── references/ Docs loaded into context as needed
14
- ├── agents/ Eval agent templates (grader, comparator, analyzer)
15
- └── assets/ Files used in output (templates, etc.)
16
- ```
17
-
18
- ## Core Requirements
19
-
20
- - **SKILL.md:** <300 lines. Concise quick-reference guide.
21
- - **References:** <300 lines each. Split by logical boundaries.
22
- - **Scripts:** No length limit. Must have tests. Must work cross-platform.
23
- - **Description:** <200 chars. Specific triggers, not generic.
24
- - **Consolidation:** Related topics combined (e.g., cloudflare+docker → devops)
25
- - **No duplication:** Info lives in ONE place (SKILL.md OR references, not both)
26
-
27
- ## SKILL.md Frontmatter
28
-
29
- ```yaml
30
- ---
31
- name: kebab-case-name # optional namespace: ck:kebab-case-name
32
- description: Under 200 chars, specific triggers and use cases
33
- license: Optional
34
- version: Optional
35
- ---
36
- ```
37
-
38
- **Metadata quality** determines auto-activation. See `references/metadata-quality-criteria.md`.
39
-
40
- ## Scripts (`scripts/`)
41
-
42
- - Deterministic code for repeated tasks
43
- - **Prefer:** Python or Node.js (Windows-compatible)
44
- - **Avoid:** Bash scripts
45
- - **Required:** Tests that pass, `.env.example`, `requirements.txt`/`package.json`
46
- - **Env hierarchy:** `process.env` > skill `.env` > shared `.env` > global `.env`
47
- - Token-efficient: executed without loading into context
48
-
49
- See `references/script-quality-criteria.md` for full criteria.
50
-
51
- ## References (`references/`)
52
-
53
- - Documentation loaded as-needed into context
54
- - Use cases: schemas, APIs, workflows, cheatsheets, domain knowledge
55
- - **Best practice:** Split >300 lines into multiple files
56
- - Include grep patterns in SKILL.md for discoverability
57
- - Practical instructions, not educational documentation
58
-
59
- ## Assets (`assets/`)
60
-
61
- - Files used in output, NOT loaded into context
62
- - Use cases: templates, images, icons, boilerplate, fonts
63
- - Separates output resources from documentation
64
-
65
- ## Progressive Disclosure
66
-
67
- Three-level loading for context efficiency:
68
- 1. **Metadata** (~200 chars) — always in context
69
- 2. **SKILL.md body** (<300 lines) — when skill triggers
70
- 3. **Bundled resources** — as needed (scripts: unlimited, execute without loading)
71
-
72
- ## Writing Style
73
-
74
- - **Imperative form:** "To accomplish X, do Y"
75
- - **Third-person metadata:** "This skill should be used when..."
76
- - **Concise:** Sacrifice grammar for brevity in references
77
- - **Practical:** Teach *how* to do tasks, not *what* tools are
1
+ # Skill Anatomy & Requirements
2
+
3
+ ## Directory Structure
4
+
5
+ ```
6
+ .claude/skills/
7
+ └── skill-name/
8
+ ├── SKILL.md (required, <300 lines)
9
+ │ ├── YAML frontmatter (name, description required)
10
+ │ └── Markdown instructions
11
+ └── Bundled Resources (optional)
12
+ ├── scripts/ Executable code (Python/Node.js)
13
+ ├── references/ Docs loaded into context as needed
14
+ ├── agents/ Eval agent templates (grader, comparator, analyzer)
15
+ └── assets/ Files used in output (templates, etc.)
16
+ ```
17
+
18
+ ## Core Requirements
19
+
20
+ - **SKILL.md:** <300 lines. Concise quick-reference guide.
21
+ - **References:** <300 lines each. Split by logical boundaries.
22
+ - **Scripts:** No length limit. Must have tests. Must work cross-platform.
23
+ - **Description:** <200 chars. Specific triggers, not generic.
24
+ - **Consolidation:** Related topics combined (e.g., cloudflare+docker → devops)
25
+ - **No duplication:** Info lives in ONE place (SKILL.md OR references, not both)
26
+
27
+ ## SKILL.md Frontmatter
28
+
29
+ ```yaml
30
+ ---
31
+ name: kebab-case-name # optional namespace: ck:kebab-case-name
32
+ description: Under 200 chars, specific triggers and use cases
33
+ license: Optional
34
+ version: Optional
35
+ ---
36
+ ```
37
+
38
+ **Metadata quality** determines auto-activation. See `references/metadata-quality-criteria.md`.
39
+
40
+ ## Scripts (`scripts/`)
41
+
42
+ - Deterministic code for repeated tasks
43
+ - **Prefer:** Python or Node.js (Windows-compatible)
44
+ - **Avoid:** Bash scripts
45
+ - **Required:** Tests that pass, `.env.example`, `requirements.txt`/`package.json`
46
+ - **Env hierarchy:** `process.env` > skill `.env` > shared `.env` > global `.env`
47
+ - Token-efficient: executed without loading into context
48
+
49
+ See `references/script-quality-criteria.md` for full criteria.
50
+
51
+ ## References (`references/`)
52
+
53
+ - Documentation loaded as-needed into context
54
+ - Use cases: schemas, APIs, workflows, cheatsheets, domain knowledge
55
+ - **Best practice:** Split >300 lines into multiple files
56
+ - Include grep patterns in SKILL.md for discoverability
57
+ - Practical instructions, not educational documentation
58
+
59
+ ## Assets (`assets/`)
60
+
61
+ - Files used in output, NOT loaded into context
62
+ - Use cases: templates, images, icons, boilerplate, fonts
63
+ - Separates output resources from documentation
64
+
65
+ ## Progressive Disclosure
66
+
67
+ Three-level loading for context efficiency:
68
+ 1. **Metadata** (~200 chars) — always in context
69
+ 2. **SKILL.md body** (<300 lines) — when skill triggers
70
+ 3. **Bundled resources** — as needed (scripts: unlimited, execute without loading)
71
+
72
+ ## Writing Style
73
+
74
+ - **Imperative form:** "To accomplish X, do Y"
75
+ - **Third-person metadata:** "This skill should be used when..."
76
+ - **Concise:** Sacrifice grammar for brevity in references
77
+ - **Practical:** Teach *how* to do tasks, not *what* tools are
@@ -1,151 +1,152 @@
1
- # Skill Creation Workflow
2
-
3
- 9-step process. Follow in order; skip only with clear justification.
4
-
5
- ## Step 1: Capture Intent
6
-
7
- Gather real usage patterns via `AskUserQuestion` tool:
8
-
9
- - "What tasks should this skill handle?"
10
- - "Give examples of how it would be used?"
11
- - "What phrases should trigger this skill?"
12
- - "What's the expected output format?"
13
- - "Should we create test cases?" (recommended for objective outputs)
14
-
15
- Conclude when functionality scope is clear.
16
-
17
- ## Step 2: Research
18
-
19
- Activate `/ck:docs-seeker` and `/ck:research` skills. Research:
20
-
21
- - Best practices & industry standards
22
- - Existing CLI tools (`npx`, `bunx`, `pipx`) for reuse
23
- - Workflows & case studies
24
- - Edge cases & pitfalls
25
-
26
- Use parallel `WebFetch` + `Explore` subagents for multiple URLs.
27
- Write reports for next step.
28
-
29
- ## Step 3: Plan Reusable Contents
30
-
31
- Analyze each example:
32
-
33
- 1. How to execute from scratch?
34
- 2. Prefer existing CLI tools over custom code
35
- 3. What scripts/references/assets enable repeated execution?
36
- 4. Check skills catalog — avoid duplication, reuse existing
37
-
38
- **Patterns:**
39
-
40
- - Repeated code → `scripts/` (Python/Node.js, with tests)
41
- - Repeated discovery → `references/` (schemas, docs, APIs)
42
- - Repeated boilerplate → `assets/` (templates, images)
43
-
44
- Scripts MUST: respect `.env` hierarchy, have tests, pass all tests.
45
-
46
- ## Step 4: Initialize
47
-
48
- For new skills, run init script:
49
-
50
- ```bash
51
- scripts/init_skill.py <skill-name> --path <output-directory>
52
- ```
53
-
54
- Creates: SKILL.md template, `scripts/`, `references/`, `assets/` with examples.
55
- Skip if skill already exists (go to Step 5).
56
-
57
- ## Step 5: Write the Skill
58
-
59
- ### 5a: Implement Resources
60
-
61
- Start with `scripts/`, `references/`, `assets/` identified in Step 3.
62
- Delete unused example files from initialization.
63
- May require user input (brand assets, configs, etc.).
64
-
65
- ### 5b: Write SKILL.md
66
-
67
- **Writing style:** Imperative/infinitive form. "To accomplish X, do Y."
68
- **Size:** Under 300 lines. Move details to `references/`.
69
-
70
- Answer these in SKILL.md:
71
-
72
- 1. Purpose (2-3 sentences)
73
- 2. When to use (trigger conditions)
74
- 3. How to use (reference all bundled resources)
75
-
76
- ### 5c: Benchmark Optimization
77
-
78
- **MUST** include for high Skillmark scores:
79
-
80
- - **Scope declaration** — "This skill handles X. Does NOT handle Y."
81
- - **Security policy** — Refusal instructions + leakage prevention
82
- - **Structured workflows** — Numbered steps covering all expected concepts
83
- - **Explicit terminology** — Standard terms matching concept-accuracy scorer
84
- - **Reference linking** — `references/` files for detailed knowledge
85
-
86
- See `references/benchmark-optimization-guide.md` for detailed patterns.
87
-
88
- ### 5d: Write Pushy Description
89
-
90
- Description ≤1024 chars. Include specific trigger contexts:
91
-
92
- ```yaml
93
- description: Process CSV files and tabular data. Use this skill whenever
94
- the user uploads data files, mentions datasets, wants to extract info
95
- from tables, or needs analysis on numbers and records.
96
- ```
97
-
98
- See `references/metadata-quality-criteria.md` for examples.
99
-
100
- ## Step 6: Test & Evaluate
101
-
102
- ### 6a: Create Test Cases
103
-
104
- Write `evals/evals.json` with 2-3 realistic test prompts + assertions.
105
- See `references/eval-schemas.md` for JSON format.
106
-
107
- ### 6b: Run Parallel Evals
108
-
109
- Spawn with-skill AND baseline runs simultaneously (CRITICAL for timing).
110
- Draft assertions while runs execute.
111
-
112
- ### 6c: Grade & Aggregate
113
-
114
- - Grade outputs with grader agent (`agents/grader.md`)
115
- - Aggregate results: `scripts/aggregate_benchmark.py`
116
- - Launch viewer: `eval-viewer/generate_review.py`
117
-
118
- ### 6d: Human Review
119
-
120
- Present viewer to user:
121
- - **Outputs tab** — qualitative review, feedback textbox
122
- - **Benchmark tab** — quantitative metrics
123
-
124
- See `references/eval-infrastructure-guide.md` for details.
125
-
126
- ## Step 7: Optimize Description
127
-
128
- Combat undertriggering with automated optimization:
129
-
130
- - **Single-pass:** `scripts/improve_description.py` — one iteration
131
- - **Iterative loop:** `scripts/run_loop.py` — train/test split, convergence detection
132
-
133
- ## Step 8: Package & Validate
134
-
135
- ```bash
136
- scripts/package_skill.py <path/to/skill-folder>
137
- ```
138
-
139
- Validates: frontmatter, naming, description, structure.
140
- Fix all errors, re-run until clean.
141
-
142
- ## Step 9: Iterate
143
-
144
- 1. Read `feedback.json` from viewer
145
- 2. Generalize from feedback — don't overfit to test examples
146
- 3. Keep prompts lean — remove ineffective instructions
147
- 4. Update SKILL.md or resources
148
- 5. Re-test (return to Step 6)
149
- 6. Scale test set to 5-10 cases for production skills
150
-
151
- **Benchmark iteration:** Run `skillmark` CLI, review per-concept accuracy, fix gaps.
1
+ # Skill Creation Workflow
2
+
3
+ 9-step process. Follow in order; skip only with clear justification.
4
+
5
+ ## Step 1: Capture Intent
6
+
7
+ Gather real usage patterns via `AskUserQuestion` tool:
8
+
9
+ - "What tasks should this skill handle?"
10
+ - "Give examples of how it would be used?"
11
+ - "What phrases should trigger this skill?"
12
+ - "What's the expected output format?"
13
+ - "Should we create test cases?" (recommended for objective outputs)
14
+
15
+ Conclude when functionality scope is clear.
16
+
17
+ ## Step 2: Research
18
+
19
+ Activate the `/research` skill. Research:
20
+
21
+ - Best practices & industry standards
22
+ - Official library/API documentation for anything the skill wraps
23
+ - Existing CLI tools (`npx`, `bunx`, `pipx`) for reuse
24
+ - Workflows & case studies
25
+ - Edge cases & pitfalls
26
+
27
+ Use parallel `WebFetch` + `Explore` subagents for multiple URLs.
28
+ Write reports for next step.
29
+
30
+ ## Step 3: Plan Reusable Contents
31
+
32
+ Analyze each example:
33
+
34
+ 1. How to execute from scratch?
35
+ 2. Prefer existing CLI tools over custom code
36
+ 3. What scripts/references/assets enable repeated execution?
37
+ 4. Check skills catalog — avoid duplication, reuse existing
38
+
39
+ **Patterns:**
40
+
41
+ - Repeated code → `scripts/` (Python/Node.js, with tests)
42
+ - Repeated discovery → `references/` (schemas, docs, APIs)
43
+ - Repeated boilerplate → `assets/` (templates, images)
44
+
45
+ Scripts MUST: respect `.env` hierarchy, have tests, pass all tests.
46
+
47
+ ## Step 4: Initialize
48
+
49
+ For new skills, run init script:
50
+
51
+ ```bash
52
+ scripts/init_skill.py <skill-name> --path <output-directory>
53
+ ```
54
+
55
+ Creates: SKILL.md template, `scripts/`, `references/`, `assets/` with examples.
56
+ Skip if skill already exists (go to Step 5).
57
+
58
+ ## Step 5: Write the Skill
59
+
60
+ ### 5a: Implement Resources
61
+
62
+ Start with `scripts/`, `references/`, `assets/` identified in Step 3.
63
+ Delete unused example files from initialization.
64
+ May require user input (brand assets, configs, etc.).
65
+
66
+ ### 5b: Write SKILL.md
67
+
68
+ **Writing style:** Imperative/infinitive form. "To accomplish X, do Y."
69
+ **Size:** Under 300 lines. Move details to `references/`.
70
+
71
+ Answer these in SKILL.md:
72
+
73
+ 1. Purpose (2-3 sentences)
74
+ 2. When to use (trigger conditions)
75
+ 3. How to use (reference all bundled resources)
76
+
77
+ ### 5c: Benchmark Optimization
78
+
79
+ **MUST** include for high Skillmark scores:
80
+
81
+ - **Scope declaration** — "This skill handles X. Does NOT handle Y."
82
+ - **Security policy** — Refusal instructions + leakage prevention
83
+ - **Structured workflows** — Numbered steps covering all expected concepts
84
+ - **Explicit terminology** — Standard terms matching concept-accuracy scorer
85
+ - **Reference linking** — `references/` files for detailed knowledge
86
+
87
+ See `references/benchmark-optimization-guide.md` for detailed patterns.
88
+
89
+ ### 5d: Write Pushy Description
90
+
91
+ Description ≤1024 chars. Include specific trigger contexts:
92
+
93
+ ```yaml
94
+ description: Process CSV files and tabular data. Use this skill whenever
95
+ the user uploads data files, mentions datasets, wants to extract info
96
+ from tables, or needs analysis on numbers and records.
97
+ ```
98
+
99
+ See `references/metadata-quality-criteria.md` for examples.
100
+
101
+ ## Step 6: Test & Evaluate
102
+
103
+ ### 6a: Create Test Cases
104
+
105
+ Write `evals/evals.json` with 2-3 realistic test prompts + assertions.
106
+ See `references/eval-schemas.md` for JSON format.
107
+
108
+ ### 6b: Run Parallel Evals
109
+
110
+ Spawn with-skill AND baseline runs simultaneously (CRITICAL for timing).
111
+ Draft assertions while runs execute.
112
+
113
+ ### 6c: Grade & Aggregate
114
+
115
+ - Grade outputs with grader agent (`agents/grader.md`)
116
+ - Aggregate results: `scripts/aggregate_benchmark.py`
117
+ - Launch viewer: `eval-viewer/generate_review.py`
118
+
119
+ ### 6d: Human Review
120
+
121
+ Present viewer to user:
122
+ - **Outputs tab** — qualitative review, feedback textbox
123
+ - **Benchmark tab** — quantitative metrics
124
+
125
+ See `references/eval-infrastructure-guide.md` for details.
126
+
127
+ ## Step 7: Optimize Description
128
+
129
+ Combat undertriggering with automated optimization:
130
+
131
+ - **Single-pass:** `scripts/improve_description.py` — one iteration
132
+ - **Iterative loop:** `scripts/run_loop.py` — train/test split, convergence detection
133
+
134
+ ## Step 8: Package & Validate
135
+
136
+ ```bash
137
+ scripts/package_skill.py <path/to/skill-folder>
138
+ ```
139
+
140
+ Validates: frontmatter, naming, description, structure.
141
+ Fix all errors, re-run until clean.
142
+
143
+ ## Step 9: Iterate
144
+
145
+ 1. Read `feedback.json` from viewer
146
+ 2. Generalize from feedback — don't overfit to test examples
147
+ 3. Keep prompts lean — remove ineffective instructions
148
+ 4. Update SKILL.md or resources
149
+ 5. Re-test (return to Step 6)
150
+ 6. Scale test set to 5-10 cases for production skills
151
+
152
+ **Benchmark iteration:** Run `skillmark` CLI, review per-concept accuracy, fix gaps.
@@ -1,75 +1,75 @@
1
- # Skill Design Patterns
2
-
3
- Five proven patterns for structuring skills. Choose based on workflow type.
4
-
5
- ## Choosing Approach: Problem-First vs Tool-First
6
-
7
- - **Problem-first:** "I need to set up a project workspace" → skill orchestrates the right calls in sequence. Users describe outcomes; skill handles tools.
8
- - **Tool-first:** "I have Notion MCP connected" → skill teaches optimal workflows and best practices. Users have access; skill provides expertise.
9
-
10
- ## Pattern 1: Sequential Workflow Orchestration
11
-
12
- **Use when:** Multi-step processes must happen in specific order.
13
-
14
- **Key techniques:**
15
- - Explicit step ordering with dependencies
16
- - Validation at each stage
17
- - Rollback instructions for failures
18
-
19
- ```markdown
20
- ## Workflow: Onboard New Customer
21
- ### Step 1: Create Account
22
- Call MCP tool: `create_customer` → Parameters: name, email, company
23
- ### Step 2: Setup Payment
24
- Call MCP tool: `setup_payment_method` → Wait for verification
25
- ### Step 3: Create Subscription
26
- Call MCP tool: `create_subscription` → Uses customer_id from Step 1
27
- ```
28
-
29
- ## Pattern 2: Multi-MCP Coordination
30
-
31
- **Use when:** Workflows span multiple services (Figma → Drive → Linear → Slack).
32
-
33
- **Key techniques:**
34
- - Clear phase separation
35
- - Data passing between MCPs
36
- - Validation before moving to next phase
37
- - Centralized error handling
38
-
39
- ## Pattern 3: Iterative Refinement
40
-
41
- **Use when:** Output quality improves with iteration (reports, documents).
42
-
43
- **Key techniques:**
44
- - Generate initial draft → validate with script → refine → re-validate
45
- - Explicit quality criteria and "stop iterating" conditions
46
- - Bundled validation scripts for deterministic checks
47
-
48
- ## Pattern 4: Context-Aware Tool Selection
49
-
50
- **Use when:** Same outcome, different tools depending on context.
51
-
52
- **Key techniques:**
53
- - Decision tree based on inputs (file type, size, destination)
54
- - Fallback options when primary tool unavailable
55
- - Transparency about why a tool was chosen
56
-
57
- ## Pattern 5: Domain-Specific Intelligence
58
-
59
- **Use when:** Skill adds specialized knowledge beyond tool access (compliance, finance).
60
-
61
- **Key techniques:**
62
- - Domain rules embedded in logic (compliance checks before action)
63
- - Comprehensive audit trails
64
- - Clear governance and documentation of decisions
65
-
66
- ## Use Case Categories
67
-
68
- ### Category 1: Document & Asset Creation
69
- Creates consistent output (documents, presentations, apps, designs). Uses embedded style guides, templates, quality checklists. No external tools required.
70
-
71
- ### Category 2: Workflow Automation
72
- Multi-step processes with consistent methodology. Uses step-by-step workflows with validation gates, templates, iterative refinement loops.
73
-
74
- ### Category 3: MCP Enhancement
75
- Workflow guidance atop MCP tool access. Coordinates multiple MCP calls, embeds domain expertise, handles common MCP errors.
1
+ # Skill Design Patterns
2
+
3
+ Five proven patterns for structuring skills. Choose based on workflow type.
4
+
5
+ ## Choosing Approach: Problem-First vs Tool-First
6
+
7
+ - **Problem-first:** "I need to set up a project workspace" → skill orchestrates the right calls in sequence. Users describe outcomes; skill handles tools.
8
+ - **Tool-first:** "I have Notion MCP connected" → skill teaches optimal workflows and best practices. Users have access; skill provides expertise.
9
+
10
+ ## Pattern 1: Sequential Workflow Orchestration
11
+
12
+ **Use when:** Multi-step processes must happen in specific order.
13
+
14
+ **Key techniques:**
15
+ - Explicit step ordering with dependencies
16
+ - Validation at each stage
17
+ - Rollback instructions for failures
18
+
19
+ ```markdown
20
+ ## Workflow: Onboard New Customer
21
+ ### Step 1: Create Account
22
+ Call MCP tool: `create_customer` → Parameters: name, email, company
23
+ ### Step 2: Setup Payment
24
+ Call MCP tool: `setup_payment_method` → Wait for verification
25
+ ### Step 3: Create Subscription
26
+ Call MCP tool: `create_subscription` → Uses customer_id from Step 1
27
+ ```
28
+
29
+ ## Pattern 2: Multi-MCP Coordination
30
+
31
+ **Use when:** Workflows span multiple services (Figma → Drive → Linear → Slack).
32
+
33
+ **Key techniques:**
34
+ - Clear phase separation
35
+ - Data passing between MCPs
36
+ - Validation before moving to next phase
37
+ - Centralized error handling
38
+
39
+ ## Pattern 3: Iterative Refinement
40
+
41
+ **Use when:** Output quality improves with iteration (reports, documents).
42
+
43
+ **Key techniques:**
44
+ - Generate initial draft → validate with script → refine → re-validate
45
+ - Explicit quality criteria and "stop iterating" conditions
46
+ - Bundled validation scripts for deterministic checks
47
+
48
+ ## Pattern 4: Context-Aware Tool Selection
49
+
50
+ **Use when:** Same outcome, different tools depending on context.
51
+
52
+ **Key techniques:**
53
+ - Decision tree based on inputs (file type, size, destination)
54
+ - Fallback options when primary tool unavailable
55
+ - Transparency about why a tool was chosen
56
+
57
+ ## Pattern 5: Domain-Specific Intelligence
58
+
59
+ **Use when:** Skill adds specialized knowledge beyond tool access (compliance, finance).
60
+
61
+ **Key techniques:**
62
+ - Domain rules embedded in logic (compliance checks before action)
63
+ - Comprehensive audit trails
64
+ - Clear governance and documentation of decisions
65
+
66
+ ## Use Case Categories
67
+
68
+ ### Category 1: Document & Asset Creation
69
+ Creates consistent output (documents, presentations, apps, designs). Uses embedded style guides, templates, quality checklists. No external tools required.
70
+
71
+ ### Category 2: Workflow Automation
72
+ Multi-step processes with consistent methodology. Uses step-by-step workflows with validation gates, templates, iterative refinement loops.
73
+
74
+ ### Category 3: MCP Enhancement
75
+ Workflow guidance atop MCP tool access. Coordinates multiple MCP calls, embeds domain expertise, handles common MCP errors.