@meyverick/agentic 5.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +234 -0
- package/CHANGELOG.md +236 -0
- package/README.md +50 -0
- package/install.ts +349 -0
- package/package.json +37 -0
- package/scripts/check-deps.mjs +587 -0
- package/scripts/git-dl.mjs +100 -0
- package/skills/check/SKILL.md +108 -0
- package/skills/check/evals/benchmark.json +40 -0
- package/skills/check/evals/evals.json +38 -0
- package/skills/check/references/diagnostic-matrix.md +170 -0
- package/skills/check/references/script-anatomy.md +154 -0
- package/skills/create-skill/SKILL.md +291 -0
- package/skills/create-skill/assets/templates/SKILL.md.template +118 -0
- package/skills/create-skill/assets/templates/evals.json.template +36 -0
- package/skills/create-skill/assets/templates/grading.json.template +26 -0
- package/skills/create-skill/evals/benchmark.json +41 -0
- package/skills/create-skill/evals/evals.json +50 -0
- package/skills/create-skill/evals/grading-template.json +36 -0
- package/skills/create-skill/evals/near-misses.json +35 -0
- package/skills/create-skill/evals/trigger-queries.json +80 -0
- package/skills/create-skill/references/antipatterns.md +123 -0
- package/skills/create-skill/references/component-decomposition.md +130 -0
- package/skills/create-skill/references/content-quality-criteria.md +61 -0
- package/skills/create-skill/references/description-optimization.md +90 -0
- package/skills/create-skill/references/eval-methodology.md +100 -0
- package/skills/create-skill/references/fragility-matching.md +88 -0
- package/skills/create-skill/references/gotchas-patterns.md +80 -0
- package/skills/create-skill/references/specification.md +77 -0
- package/skills/create-skill/scripts/audit-antipatterns.mjs +164 -0
- package/skills/create-skill/scripts/compute-benchmark.mjs +111 -0
- package/skills/create-skill/scripts/run-cold-eval.mjs +118 -0
- package/skills/create-skill/scripts/scaffold-skill.mjs +86 -0
- package/skills/create-skill/scripts/validate-routing.mjs +137 -0
- package/skills/create-skill/scripts/validate-structure.mjs +223 -0
- package/skills/design-craft/SKILL.md +134 -0
- package/skills/design-craft/evals/benchmark.json +41 -0
- package/skills/design-craft/evals/evals.json +81 -0
- package/skills/design-craft/references/anti-slop-patterns.md +49 -0
- package/skills/design-craft/references/art-direction.md +89 -0
- package/skills/design-craft/references/design-engineering.md +122 -0
- package/skills/design-craft/references/motion-craft.md +124 -0
- package/skills/design-craft/references/process.md +47 -0
- package/skills/design-craft/references/review-checklist.md +121 -0
- package/skills/guardrails/SKILL.md +118 -0
- package/skills/guardrails/evals/benchmark.json +40 -0
- package/skills/guardrails/evals/evals.json +49 -0
- package/skills/guardrails/references/guardrails-patterns.md +43 -0
- package/skills/okf-docs/SKILL.md +79 -0
- package/skills/okf-docs/evals/benchmark.json +21 -0
- package/skills/okf-docs/evals/evals.json +37 -0
- package/skills/okf-docs/references/okf-spec.md +56 -0
- package/skills/okf-docs/scripts/validate-frontmatter.mjs +130 -0
- package/skills/openspec-harden/SKILL.md +138 -0
- package/skills/openspec-harden/evals/benchmark.json +40 -0
- package/skills/openspec-harden/evals/evals.json +38 -0
- package/skills/openspec-learn/SKILL.md +216 -0
- package/skills/openspec-learn/evals/benchmark.json +44 -0
- package/skills/openspec-learn/evals/evals.json +48 -0
- package/skills/openspec-learn/evals/retrieval-bench.json +27 -0
- package/skills/openspec-learn/references/conflict-handling.md +20 -0
- package/skills/openspec-learn/references/evaluation-methodology.md +126 -0
- package/skills/openspec-learn/references/examples.md +37 -0
- package/skills/openspec-learn/references/improvement-patterns.md +155 -0
- package/skills/openspec-learn/references/report-analysis.md +104 -0
- package/skills/openspec-learn/references/skill-quality.md +103 -0
- package/skills/openspec-learn/references/tool-type-detection.md +30 -0
- package/skills/openspec-report/SKILL.md +104 -0
- package/skills/openspec-report/assets/templates/assessment.md.template +84 -0
- package/skills/openspec-report/assets/templates/report.md.template +92 -0
- package/skills/openspec-report/evals/benchmark.json +44 -0
- package/skills/openspec-report/evals/evals.json +46 -0
- package/skills/qmd-research/SKILL.md +89 -0
- package/skills/qmd-research/evals/benchmark.json +40 -0
- package/skills/qmd-research/evals/evals.json +38 -0
- package/skills/qmd-research/references/index-management.md +69 -0
- package/skills/qmd-research/references/query-craft.md +82 -0
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill_name": "create-skill",
|
|
3
|
+
"queries": [
|
|
4
|
+
{
|
|
5
|
+
"id": 1,
|
|
6
|
+
"query": "Build me a skill that processes PDFs and extracts text.",
|
|
7
|
+
"should_trigger": true
|
|
8
|
+
},
|
|
9
|
+
{
|
|
10
|
+
"id": 2,
|
|
11
|
+
"query": "I need a new agent skill for database queries.",
|
|
12
|
+
"should_trigger": true
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
"id": 3,
|
|
16
|
+
"query": "Create a skill that helps build other skills.",
|
|
17
|
+
"should_trigger": true
|
|
18
|
+
},
|
|
19
|
+
{
|
|
20
|
+
"id": 4,
|
|
21
|
+
"query": "Help me extract a reusable pattern from this workflow into a skill.",
|
|
22
|
+
"should_trigger": true
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
"id": 5,
|
|
26
|
+
"query": "Set up evaluation for my existing skill.",
|
|
27
|
+
"should_trigger": true
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": 6,
|
|
31
|
+
"query": "Write a Python script to analyze data.",
|
|
32
|
+
"should_trigger": false
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"id": 7,
|
|
36
|
+
"query": "Review this code for bugs.",
|
|
37
|
+
"should_trigger": false
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
"id": 8,
|
|
41
|
+
"query": "Create a bash script that automates deployment.",
|
|
42
|
+
"should_trigger": false
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"id": 9,
|
|
46
|
+
"query": "Help me write documentation for my API.",
|
|
47
|
+
"should_trigger": false
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
"id": 10,
|
|
51
|
+
"query": "Fix the bug in my authentication middleware.",
|
|
52
|
+
"should_trigger": false
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
"id": 11,
|
|
56
|
+
"query": "I want to build something that helps with file processing.",
|
|
57
|
+
"should_trigger": true
|
|
58
|
+
},
|
|
59
|
+
{
|
|
60
|
+
"id": 12,
|
|
61
|
+
"query": "Make a tool for converting JSON to YAML.",
|
|
62
|
+
"should_trigger": false
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"id": 13,
|
|
66
|
+
"query": "Design a skill for handling webhook integrations.",
|
|
67
|
+
"should_trigger": true
|
|
68
|
+
},
|
|
69
|
+
{
|
|
70
|
+
"id": 14,
|
|
71
|
+
"query": "Write unit tests for my function.",
|
|
72
|
+
"should_trigger": false
|
|
73
|
+
},
|
|
74
|
+
{
|
|
75
|
+
"id": 15,
|
|
76
|
+
"query": "Create a reusable skill from this task I just did.",
|
|
77
|
+
"should_trigger": true
|
|
78
|
+
}
|
|
79
|
+
]
|
|
80
|
+
}
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
# Antipatterns
|
|
2
|
+
|
|
3
|
+
16 audited failure modes from agent builds. Self-audit checklist.
|
|
4
|
+
|
|
5
|
+
## A1: Phantom Tool Reference
|
|
6
|
+
|
|
7
|
+
**Pattern**: Prompt mentions tool agent can't call.
|
|
8
|
+
|
|
9
|
+
**BAD**: `call action_user_consumeFood` while agent tags exclude that gate.
|
|
10
|
+
**GOOD**: Prompt names only tools agent actually has; cross-role requests go via delegation.
|
|
11
|
+
|
|
12
|
+
## A2: Duplicated Invariant
|
|
13
|
+
|
|
14
|
+
**Pattern**: Same rule copy-pasted across files AND injected at runtime.
|
|
15
|
+
|
|
16
|
+
**BAD**: 24-char ID rule in 3 prompts + auto-injected.
|
|
17
|
+
**GOOD**: Single source — global rules injected once; role-specific rules in one file only.
|
|
18
|
+
|
|
19
|
+
## A3: Passive-Voice Triggers
|
|
20
|
+
|
|
21
|
+
**Pattern**: Role-job description instead of imperative trigger.
|
|
22
|
+
|
|
23
|
+
**BAD**: "You are the Mercenary. Evaluate auctions."
|
|
24
|
+
**GOOD**: "When `activeAuction != null AND payout_per_damage > market_avg` → call `bid_contract`."
|
|
25
|
+
|
|
26
|
+
## A4: Copy-Pasted Cheat-Sheet
|
|
27
|
+
|
|
28
|
+
**Pattern**: Shared lookup list duplicated per file, each copy drifting.
|
|
29
|
+
|
|
30
|
+
**BAD**: Orchestrator lists 6 procedures, analyst lists 1 — same access, different coverage.
|
|
31
|
+
**GOOD**: One canonical `procedures.md`, linked/injected by both.
|
|
32
|
+
|
|
33
|
+
## A5: Prose Bloat in Trigger Prompts
|
|
34
|
+
|
|
35
|
+
**Pattern**: Injecting prose-style rules into a tool-gate trigger prompt.
|
|
36
|
+
|
|
37
|
+
**BAD**: 21-rule style pack pasted into role file (role emits no human prose).
|
|
38
|
+
**GOOD**: Style pack injected only for prose-emitting roles.
|
|
39
|
+
|
|
40
|
+
## A6: Mismatched Delegation in Prompt
|
|
41
|
+
|
|
42
|
+
**Pattern**: Prompt asserts delegation path runtime forbids.
|
|
43
|
+
|
|
44
|
+
**BAD**: "assigned by Orchestrator OR Economist" when only orchestrator delegates.
|
|
45
|
+
**GOOD**: "I receive tasks from the Orchestrator"; runtime enforces constraints.
|
|
46
|
+
|
|
47
|
+
## A7: Stale-Data Trade
|
|
48
|
+
|
|
49
|
+
**Pattern**: Agent acts on memory past its `stale_after` without refetching.
|
|
50
|
+
|
|
51
|
+
**BAD**: Trades on a market note fetched 3h ago.
|
|
52
|
+
**GOOD**: Stale flag → refetch upstream before directive.
|
|
53
|
+
|
|
54
|
+
## A8: Trust-Without-Freshness
|
|
55
|
+
|
|
56
|
+
**Pattern**: Verified fact trusted regardless of age.
|
|
57
|
+
|
|
58
|
+
**BAD**: "It's verified, so trust the 6h-old price."
|
|
59
|
+
**GOOD**: Stale overrides verified; un-staled-but-unverified still needs confirmation.
|
|
60
|
+
|
|
61
|
+
## A9: Naive Context Truncation
|
|
62
|
+
|
|
63
|
+
**Pattern**: `slice(last N)` discards oldest messages, including relevant tool results.
|
|
64
|
+
|
|
65
|
+
**BAD**: 50-msg context → last 10.
|
|
66
|
+
**GOOD**: Summarize-older + preserve recents; or move settled facts to bundles.
|
|
67
|
+
|
|
68
|
+
## A10: Silent Upstream Failure
|
|
69
|
+
|
|
70
|
+
**Pattern**: Upstream errors extract `data ?? body` blindly; agent acts on `null`.
|
|
71
|
+
|
|
72
|
+
**BAD**: Gateway returns `{}` on 500 → bot reports success.
|
|
73
|
+
**GOOD**: Schema validate; circuit breaker; skip + log + escalate.
|
|
74
|
+
|
|
75
|
+
## A11: Hardcoded Fallback Credential
|
|
76
|
+
|
|
77
|
+
**Pattern**: When live auth fails, code substitutes leaked session id.
|
|
78
|
+
|
|
79
|
+
**BAD**: `vid` falls back to captured session token in source.
|
|
80
|
+
**GOOD**: Surface auth failure; never silently substitute; never commit credentials.
|
|
81
|
+
|
|
82
|
+
## A12: Infinite Delegation
|
|
83
|
+
|
|
84
|
+
**Pattern**: Two agents ping-pong without depth cap.
|
|
85
|
+
|
|
86
|
+
**BAD**: No recursion bound → tokens → runaway.
|
|
87
|
+
**GOOD**: Depth cap on handoff (e.g., ≤3); force back to hub.
|
|
88
|
+
|
|
89
|
+
## A13: One-Size-Fits-All Fragility
|
|
90
|
+
|
|
91
|
+
**Pattern**: Same gating strictness for creative writer and money transfer.
|
|
92
|
+
|
|
93
|
+
**BAD**: Article publisher gated like market sale.
|
|
94
|
+
**GOOD**: Tier guardrails by fragility class.
|
|
95
|
+
|
|
96
|
+
## A14: Single File Omnibus
|
|
97
|
+
|
|
98
|
+
**Pattern**: 4000-line system prompt mixing everything.
|
|
99
|
+
|
|
100
|
+
**BAD**: One omnibus per agent, copy-paste everywhere.
|
|
101
|
+
**GOOD**: Persona / instructions / templates / data as separate files, loaded on demand.
|
|
102
|
+
|
|
103
|
+
## A15: Vague "Professional" Success Bar
|
|
104
|
+
|
|
105
|
+
**Pattern**: Quality judged by adjectives, not scores.
|
|
106
|
+
|
|
107
|
+
**BAD**: Ship when prompts "look right."
|
|
108
|
+
**GOOD**: Trigger pass-rate + near-miss rate + output-lint violations.
|
|
109
|
+
|
|
110
|
+
## A16: Promises vs Runtime Lint Missing
|
|
111
|
+
|
|
112
|
+
**Pattern**: No build-time check that prompt-named tools exist.
|
|
113
|
+
|
|
114
|
+
**BAD**: Phantom tool ships to production.
|
|
115
|
+
**GOOD**: Lint: every tool name in prompt ∈ runtime toolset for that agent.
|
|
116
|
+
|
|
117
|
+
## Self-Audit Protocol
|
|
118
|
+
|
|
119
|
+
Before declaring skill complete:
|
|
120
|
+
1. Read every file
|
|
121
|
+
2. For each antipattern, confirm no file exhibits it
|
|
122
|
+
3. Confirm: single-source invariants, no vendor installs, no phantom tools
|
|
123
|
+
4. If any violation found, fix before shipping
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# Component Decomposition
|
|
2
|
+
|
|
3
|
+
Gem-factory pattern for skill structure.
|
|
4
|
+
|
|
5
|
+
## The Problem
|
|
6
|
+
|
|
7
|
+
Single omnibus files (4000+ lines) mix persona, instructions, templates, and data. This creates:
|
|
8
|
+
- High churn surface (everything changes together)
|
|
9
|
+
- Copy-paste drift (same rule in multiple places)
|
|
10
|
+
- Context bloat (agent loads everything, uses little)
|
|
11
|
+
|
|
12
|
+
## The Solution
|
|
13
|
+
|
|
14
|
+
Decompose into discrete components:
|
|
15
|
+
|
|
16
|
+
| Component | Holds | File Shape | Churn Rate |
|
|
17
|
+
|-----------|-------|------------|------------|
|
|
18
|
+
| **Persona** | Identity, guiding principles | Short, stable | Low |
|
|
19
|
+
| **Instructions** | Workflows, decision trees | Markdown | Medium |
|
|
20
|
+
| **Templates** | Output shapes | Markdown with placeholders | High |
|
|
21
|
+
| **Data** | Factual reference, config | CSV/structured | High |
|
|
22
|
+
|
|
23
|
+
## Component Types
|
|
24
|
+
|
|
25
|
+
### Persona (Identity)
|
|
26
|
+
|
|
27
|
+
Who the skill is. Stable, rarely changes.
|
|
28
|
+
|
|
29
|
+
```markdown
|
|
30
|
+
# persona.md
|
|
31
|
+
|
|
32
|
+
You are a data analyst specializing in CSV processing.
|
|
33
|
+
Your goal is to help users understand their tabular data.
|
|
34
|
+
You prefer clarity over cleverness.
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
### Instructions (Workflows)
|
|
38
|
+
|
|
39
|
+
How the skill works. Changes when process changes.
|
|
40
|
+
|
|
41
|
+
```markdown
|
|
42
|
+
# instructions.md
|
|
43
|
+
|
|
44
|
+
## Workflow
|
|
45
|
+
|
|
46
|
+
1. Load data file
|
|
47
|
+
2. Validate schema
|
|
48
|
+
3. Compute statistics
|
|
49
|
+
4. Generate output
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
### Templates (Output Shapes)
|
|
53
|
+
|
|
54
|
+
What the skill produces. Changes when format changes.
|
|
55
|
+
|
|
56
|
+
```markdown
|
|
57
|
+
# templates/report.md
|
|
58
|
+
|
|
59
|
+
# Analysis Report
|
|
60
|
+
|
|
61
|
+
## Summary
|
|
62
|
+
{{summary}}
|
|
63
|
+
|
|
64
|
+
## Key Findings
|
|
65
|
+
{{findings}}
|
|
66
|
+
|
|
67
|
+
## Recommendations
|
|
68
|
+
{{recommendations}}
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
### Data (Reference Material)
|
|
72
|
+
|
|
73
|
+
Factual reference. Changes when facts change.
|
|
74
|
+
|
|
75
|
+
```csv
|
|
76
|
+
# data/schema.csv
|
|
77
|
+
column,type,required
|
|
78
|
+
id,integer,yes
|
|
79
|
+
name,string,yes
|
|
80
|
+
email,string,no
|
|
81
|
+
created_at,timestamp,yes
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
## File Structure
|
|
85
|
+
|
|
86
|
+
```
|
|
87
|
+
skill-name/
|
|
88
|
+
├── SKILL.md # Entry point, links to components
|
|
89
|
+
├── persona.md # Identity (optional, can be in SKILL.md)
|
|
90
|
+
├── instructions/ # Workflows
|
|
91
|
+
│ ├── workflow.md
|
|
92
|
+
│ └── decision-tree.md
|
|
93
|
+
├── templates/ # Output shapes
|
|
94
|
+
│ ├── report.md
|
|
95
|
+
│ └── email.md
|
|
96
|
+
├── data/ # Reference material
|
|
97
|
+
│ ├── schema.csv
|
|
98
|
+
│ └── examples.json
|
|
99
|
+
├── scripts/ # Executable logic
|
|
100
|
+
│ └── process.mjs
|
|
101
|
+
└── references/ # Detailed docs
|
|
102
|
+
└── api-reference.md
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
## Single Source Per Invariant
|
|
106
|
+
|
|
107
|
+
**Rule**: Define each rule ONCE, reference it everywhere else.
|
|
108
|
+
|
|
109
|
+
Bad:
|
|
110
|
+
- `workflow.md`: "User IDs are 24-char hex"
|
|
111
|
+
- `validation.md`: "User IDs are 24-char hex"
|
|
112
|
+
- `api-rules.md`: "User IDs are 24-char hex"
|
|
113
|
+
|
|
114
|
+
Good:
|
|
115
|
+
- `api-rules.md`: "User IDs are 24-char hex"
|
|
116
|
+
- `workflow.md`: "See api-rules.md for ID format"
|
|
117
|
+
- `validation.md`: "See api-rules.md for ID format"
|
|
118
|
+
|
|
119
|
+
## When to Decompose
|
|
120
|
+
|
|
121
|
+
Decompose when:
|
|
122
|
+
- File exceeds 500 lines
|
|
123
|
+
- Multiple concerns in one file
|
|
124
|
+
- Same content appears in multiple places
|
|
125
|
+
- Different parts change at different rates
|
|
126
|
+
|
|
127
|
+
Don't decompose when:
|
|
128
|
+
- Skill is simple (<100 lines total)
|
|
129
|
+
- All content changes together
|
|
130
|
+
- Decomposition adds more overhead than it saves
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
# Content Quality Criteria
|
|
2
|
+
|
|
3
|
+
What makes good skill instructions.
|
|
4
|
+
|
|
5
|
+
## Clarity
|
|
6
|
+
|
|
7
|
+
- **One interpretation**: Instructions should have exactly one meaning
|
|
8
|
+
- **Concrete, not abstract**: "Run `./scripts/validate.sh`" not "validate the output"
|
|
9
|
+
- **Sequential steps**: Number steps when order matters
|
|
10
|
+
- **Active voice**: "Check X" not "X should be checked"
|
|
11
|
+
|
|
12
|
+
## Actionability
|
|
13
|
+
|
|
14
|
+
- **Agent can follow without guessing**: Every step is executable
|
|
15
|
+
- **Tools specified**: Name the exact tool/command to use
|
|
16
|
+
- **Inputs/outputs defined**: What goes in, what comes out
|
|
17
|
+
- **Error handling**: What to do when things fail
|
|
18
|
+
|
|
19
|
+
## Edge Cases
|
|
20
|
+
|
|
21
|
+
- **Document non-obvious behaviors**: Things the agent would get wrong without being told
|
|
22
|
+
- **Input validation**: What happens with malformed input
|
|
23
|
+
- **Boundary conditions**: Empty inputs, large inputs, special characters
|
|
24
|
+
- **Failure modes**: Network errors, permission denied, missing files
|
|
25
|
+
|
|
26
|
+
## Examples
|
|
27
|
+
|
|
28
|
+
- **One canonical example**: Show the happy path clearly
|
|
29
|
+
- **BAD/GOOD pairs**: Show what to avoid and what to do
|
|
30
|
+
- **Real-world context**: Use realistic inputs, not "foo" and "bar"
|
|
31
|
+
- **Input → Output**: Show the transformation
|
|
32
|
+
|
|
33
|
+
## Gotchas
|
|
34
|
+
|
|
35
|
+
- **Environment-specific facts**: Things that defy reasonable assumptions
|
|
36
|
+
- **Naming inconsistencies**: Same concept, different names across systems
|
|
37
|
+
- **Hidden preconditions**: What must be true before running
|
|
38
|
+
- **Non-obvious side effects**: What happens beyond the obvious
|
|
39
|
+
|
|
40
|
+
## Progressive Disclosure
|
|
41
|
+
|
|
42
|
+
- **SKILL.md**: Core instructions, always loaded (<500 lines)
|
|
43
|
+
- **references/**: Detailed docs, loaded on-demand
|
|
44
|
+
- **scripts/**: Executable logic, invoked when needed
|
|
45
|
+
- **assets/**: Templates and static resources
|
|
46
|
+
|
|
47
|
+
## Fragility Matching
|
|
48
|
+
|
|
49
|
+
| Task Type | Specificity Level | Example |
|
|
50
|
+
|-----------|-------------------|---------|
|
|
51
|
+
| Mutation (create/delete/modify) | Strict, step-by-step | "Run exactly this command. Do not add flags." |
|
|
52
|
+
| Read-only (query/analyze) | Loose, latitude | "Query the database. Format as you see fit." |
|
|
53
|
+
| Creative (prose/design) | Low specificity | "Write a summary. Follow the style guide." |
|
|
54
|
+
|
|
55
|
+
## Antipatterns to Avoid
|
|
56
|
+
|
|
57
|
+
- **Vague instructions**: "Handle errors appropriately"
|
|
58
|
+
- **Over-specification**: "Use exactly 3 spaces for indentation"
|
|
59
|
+
- **Missing context**: "Run the script" (which script?)
|
|
60
|
+
- **Assumed knowledge**: "Use the standard approach" (what standard?)
|
|
61
|
+
- **Copy-paste**: Same rule in multiple places (drift risk)
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
# Description Optimization
|
|
2
|
+
|
|
3
|
+
How to improve skill triggering through description tuning.
|
|
4
|
+
|
|
5
|
+
## How Triggering Works
|
|
6
|
+
|
|
7
|
+
1. **Discovery**: Agent loads `name` and `description` of all skills
|
|
8
|
+
2. **Activation**: When task matches description, agent loads full SKILL.md
|
|
9
|
+
3. **Execution**: Agent follows instructions
|
|
10
|
+
|
|
11
|
+
Description carries the entire burden of triggering.
|
|
12
|
+
|
|
13
|
+
## Writing Effective Descriptions
|
|
14
|
+
|
|
15
|
+
### Principles
|
|
16
|
+
|
|
17
|
+
- **Imperative phrasing**: "Use this skill when..." not "This skill does..."
|
|
18
|
+
- **Focus on user intent**: What the user is trying to achieve
|
|
19
|
+
- **Err on pushy**: Explicitly list contexts where skill applies
|
|
20
|
+
- **Concise**: A few sentences to short paragraph
|
|
21
|
+
|
|
22
|
+
### Good vs Bad
|
|
23
|
+
|
|
24
|
+
Good:
|
|
25
|
+
```yaml
|
|
26
|
+
description: >
|
|
27
|
+
Analyze CSV and tabular data files — compute summary statistics,
|
|
28
|
+
add derived columns, generate charts, and clean messy data. Use this
|
|
29
|
+
skill when the user has a CSV, TSV, or Excel file and wants to
|
|
30
|
+
explore, transform, or visualize the data, even if they don't
|
|
31
|
+
explicitly mention "CSV" or "analysis."
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Bad:
|
|
35
|
+
```yaml
|
|
36
|
+
description: Process CSV files.
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Trigger Test Queries
|
|
40
|
+
|
|
41
|
+
Create 10-20 queries: mix of should-trigger and shouldn't-trigger.
|
|
42
|
+
|
|
43
|
+
### Should-Trigger Queries
|
|
44
|
+
|
|
45
|
+
- Vary phrasing (formal, casual, terse)
|
|
46
|
+
- Vary explicitness (direct naming vs describing need)
|
|
47
|
+
- Vary detail (terse vs context-heavy)
|
|
48
|
+
- Include cases where skill helps but connection isn't obvious
|
|
49
|
+
|
|
50
|
+
### Should-Not-Trigger Queries (Near-Misses)
|
|
51
|
+
|
|
52
|
+
Strong negative examples share keywords but need different handling:
|
|
53
|
+
- "Write a Python script that reads a CSV and uploads to Postgres" (involves CSV, but task is database ETL, not analysis)
|
|
54
|
+
- "Update the formulas in my Excel budget spreadsheet" (shares spreadsheet concept, but needs Excel editing, not CSV analysis)
|
|
55
|
+
|
|
56
|
+
Weak negatives (don't test anything):
|
|
57
|
+
- "Write a fibonacci function" (obviously irrelevant)
|
|
58
|
+
- "What's the weather today?" (no keyword overlap)
|
|
59
|
+
|
|
60
|
+
## Optimization Loop
|
|
61
|
+
|
|
62
|
+
1. **Evaluate** current description on train + validation sets
|
|
63
|
+
2. **Identify failures** in train set only
|
|
64
|
+
3. **Revise description**:
|
|
65
|
+
- Should-trigger failing → broaden scope
|
|
66
|
+
- Should-not-trigger failing → add specificity
|
|
67
|
+
- Avoid keyword overfitting → find general category
|
|
68
|
+
4. **Repeat** until train set passes or plateau
|
|
69
|
+
5. **Select best** by validation pass rate
|
|
70
|
+
|
|
71
|
+
## Train/Validation Split
|
|
72
|
+
|
|
73
|
+
- **Train set (~60%)**: Guides changes
|
|
74
|
+
- **Validation set (~40%)**: Checks generalization
|
|
75
|
+
|
|
76
|
+
Keep split fixed across iterations. Both sets need proportional mix of positive/negative queries.
|
|
77
|
+
|
|
78
|
+
## Avoiding Overfitting
|
|
79
|
+
|
|
80
|
+
- Don't add specific keywords from failed queries
|
|
81
|
+
- Find the general category those queries represent
|
|
82
|
+
- Try structurally different approaches when stuck
|
|
83
|
+
- Stay under 1024 chars
|
|
84
|
+
|
|
85
|
+
## Applying the Result
|
|
86
|
+
|
|
87
|
+
1. Update `description` field in SKILL.md
|
|
88
|
+
2. Verify under 1024-char limit
|
|
89
|
+
3. Manual sanity check with a few prompts
|
|
90
|
+
4. Optional: 5-10 fresh queries for rigorous verification
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
# Evaluation Methodology
|
|
2
|
+
|
|
3
|
+
Full eval framework for skill quality assurance.
|
|
4
|
+
|
|
5
|
+
## Two Eval Halves
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
┌─ TRIGGER evals ─────────────────────────────┐
|
|
9
|
+
│ Does the right skill fire at the right time?│
|
|
10
|
+
│ Positive cases + near-miss negatives │
|
|
11
|
+
└──────────────────────────────────────────────┘
|
|
12
|
+
┌─ OUTPUT evals ───────────────────────────────┐
|
|
13
|
+
│ Does the output satisfy quality rules? │
|
|
14
|
+
│ Deterministic + semantic judgment │
|
|
15
|
+
└──────────────────────────────────────────────┘
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
## Test Case Design
|
|
19
|
+
|
|
20
|
+
### Positive Cases
|
|
21
|
+
|
|
22
|
+
Prompts that SHOULD trigger the skill:
|
|
23
|
+
- Varied phrasing (formal, casual, terse)
|
|
24
|
+
- Different levels of detail
|
|
25
|
+
- Realistic context (file paths, names, specifics)
|
|
26
|
+
|
|
27
|
+
### Near-Miss Negatives (Critical)
|
|
28
|
+
|
|
29
|
+
Prompts that should NOT trigger the skill:
|
|
30
|
+
- Share keywords but need different handling
|
|
31
|
+
- Similar domain but different task
|
|
32
|
+
- Without these, over-firing passes silently
|
|
33
|
+
|
|
34
|
+
Example for CSV analysis skill:
|
|
35
|
+
- ✅ "Analyze my sales CSV and make a chart" (should trigger)
|
|
36
|
+
- ❌ "Write a Python script that reads a CSV and uploads to Postgres" (near-miss: involves CSV but different task)
|
|
37
|
+
|
|
38
|
+
### Assertion Writing
|
|
39
|
+
|
|
40
|
+
Good assertions:
|
|
41
|
+
- `"The output file is valid JSON"` — programmatically verifiable
|
|
42
|
+
- `"The chart has labeled axes"` — specific and observable
|
|
43
|
+
- `"The report includes at least 3 recommendations"` — countable
|
|
44
|
+
|
|
45
|
+
Bad assertions:
|
|
46
|
+
- `"The output is good"` — too vague
|
|
47
|
+
- `"Uses exactly the phrase 'Total Revenue: $X'"` — too brittle
|
|
48
|
+
|
|
49
|
+
## Grading Principles
|
|
50
|
+
|
|
51
|
+
- **Require concrete evidence for PASS**: Don't give benefit of the doubt
|
|
52
|
+
- **Review assertions themselves**: Are they too easy, too hard, or unverifiable?
|
|
53
|
+
- **Blind comparison**: Compare outputs without revealing which version
|
|
54
|
+
|
|
55
|
+
## Benchmark Computation
|
|
56
|
+
|
|
57
|
+
```json
|
|
58
|
+
{
|
|
59
|
+
"run_summary": {
|
|
60
|
+
"with_skill": {
|
|
61
|
+
"pass_rate": { "mean": 0.83, "stddev": 0.06 },
|
|
62
|
+
"time_seconds": { "mean": 45.0, "stddev": 12.0 },
|
|
63
|
+
"tokens": { "mean": 3800, "stddev": 400 }
|
|
64
|
+
},
|
|
65
|
+
"without_skill": {
|
|
66
|
+
"pass_rate": { "mean": 0.33, "stddev": 0.10 },
|
|
67
|
+
"time_seconds": { "mean": 32.0, "stddev": 8.0 },
|
|
68
|
+
"tokens": { "mean": 2100, "stddev": 300 }
|
|
69
|
+
},
|
|
70
|
+
"delta": {
|
|
71
|
+
"pass_rate": 0.50,
|
|
72
|
+
"time_seconds": 13.0,
|
|
73
|
+
"tokens": 1700
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
## Calibration Loop
|
|
80
|
+
|
|
81
|
+
```
|
|
82
|
+
write prompt → run evals → analyze failures → edit prompt → re-run
|
|
83
|
+
▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔
|
|
84
|
+
stop when pass-rate ≥ target AND near-miss rate = 0
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Analysis Patterns
|
|
88
|
+
|
|
89
|
+
- **Remove assertions that always pass**: They don't test anything
|
|
90
|
+
- **Investigate assertions that always fail**: Are they broken?
|
|
91
|
+
- **Study assertions that pass with skill but fail without**: This is where skill adds value
|
|
92
|
+
- **Check outliers**: If one eval takes 3x longer, find why
|
|
93
|
+
|
|
94
|
+
## Tiered Enforcement
|
|
95
|
+
|
|
96
|
+
| Tier | Action | When |
|
|
97
|
+
|------|--------|------|
|
|
98
|
+
| deny | Block outright | Zero expected false positives |
|
|
99
|
+
| warn | Flag, allow | Minor issues |
|
|
100
|
+
| ask | Pause for confirmation | Ambiguous cases |
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
# Fragility Matching
|
|
2
|
+
|
|
3
|
+
Match instruction specificity to task fragility.
|
|
4
|
+
|
|
5
|
+
## Fragility Classification
|
|
6
|
+
|
|
7
|
+
| Class | Description | Example | Specificity |
|
|
8
|
+
|-------|-------------|---------|-------------|
|
|
9
|
+
| **Mutation** | Creates, deletes, or modifies data | Place order, delete account, update record | Strict, step-by-step, precondition gates |
|
|
10
|
+
| **Read-only** | Queries or analyzes data | Search database, generate report, analyze CSV | Loose triggers, latitude for formatting |
|
|
11
|
+
| **Creative** | Prose, design, creative output | Write summary, design layout, draft email | Low specificity, examples over rules |
|
|
12
|
+
|
|
13
|
+
## Mutation Tasks (Strict)
|
|
14
|
+
|
|
15
|
+
When skill involves creating, deleting, or modifying data:
|
|
16
|
+
|
|
17
|
+
- **Precondition gates**: Check before acting
|
|
18
|
+
- **Exact commands**: "Run exactly this command. Do not add flags."
|
|
19
|
+
- **Validation loops**: Verify before proceeding
|
|
20
|
+
- **Rollback paths**: What to do if something fails
|
|
21
|
+
- **Idempotency**: Safe to retry
|
|
22
|
+
|
|
23
|
+
Example:
|
|
24
|
+
```markdown
|
|
25
|
+
## Place Order
|
|
26
|
+
|
|
27
|
+
1. Verify inventory: `./scripts/check-inventory.sh <item-id>`
|
|
28
|
+
2. ONLY if in stock: `./scripts/place-order.sh <item-id> <quantity>`
|
|
29
|
+
3. Verify order: `./scripts/verify-order.sh <order-id>`
|
|
30
|
+
4. NEVER skip step 1. NEVER add flags to step 2.
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Read-Only Tasks (Loose)
|
|
34
|
+
|
|
35
|
+
When skill involves querying or analyzing data:
|
|
36
|
+
|
|
37
|
+
- **Flexible triggers**: "When user asks about X"
|
|
38
|
+
- **Format latitude**: "Format as you see fit"
|
|
39
|
+
- **Optional steps**: "Optionally include Y"
|
|
40
|
+
- **Error handling**: "If query fails, report error"
|
|
41
|
+
|
|
42
|
+
Example:
|
|
43
|
+
```markdown
|
|
44
|
+
## Analyze Data
|
|
45
|
+
|
|
46
|
+
1. Load the data file
|
|
47
|
+
2. Compute summary statistics
|
|
48
|
+
3. Generate visualizations as appropriate
|
|
49
|
+
4. Present findings in a clear format
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Creative Tasks (Low Specificity)
|
|
53
|
+
|
|
54
|
+
When skill involves prose, design, or creative output:
|
|
55
|
+
|
|
56
|
+
- **Style guidelines**: "Follow the style guide"
|
|
57
|
+
- **Examples**: Show BAD/GOOD pairs
|
|
58
|
+
- **Constraints**: "Don't include X"
|
|
59
|
+
- **Latitude**: Let agent decide approach
|
|
60
|
+
|
|
61
|
+
Example:
|
|
62
|
+
```markdown
|
|
63
|
+
## Write Summary
|
|
64
|
+
|
|
65
|
+
- One paragraph overview
|
|
66
|
+
- Include key findings
|
|
67
|
+
- Use clear, concise language
|
|
68
|
+
- See references/style-guide.md for tone
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## Fragility Matrix
|
|
72
|
+
|
|
73
|
+
```
|
|
74
|
+
mutation (create/delete/modify) ─▶ strict preconditions, gating, deterrents
|
|
75
|
+
read-only (query/analyze) ─▶ loose triggers, free text
|
|
76
|
+
creative (prose/design) ─▶ low specificity, examples
|
|
77
|
+
|
|
78
|
+
⚠️ Same skill should never treat a buy-button with the looseness of a weather check
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Decision Guide
|
|
82
|
+
|
|
83
|
+
Ask:
|
|
84
|
+
1. Does this skill change state? → Mutation → Strict
|
|
85
|
+
2. Does this skill only read state? → Read-only → Loose
|
|
86
|
+
3. Does this skill produce creative output? → Creative → Low specificity
|
|
87
|
+
|
|
88
|
+
When uncertain, default to stricter. Can always loosen later.
|