@meyverick/agentic 5.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/AGENTS.md +234 -0
  2. package/CHANGELOG.md +236 -0
  3. package/README.md +50 -0
  4. package/install.ts +349 -0
  5. package/package.json +37 -0
  6. package/scripts/check-deps.mjs +587 -0
  7. package/scripts/git-dl.mjs +100 -0
  8. package/skills/check/SKILL.md +108 -0
  9. package/skills/check/evals/benchmark.json +40 -0
  10. package/skills/check/evals/evals.json +38 -0
  11. package/skills/check/references/diagnostic-matrix.md +170 -0
  12. package/skills/check/references/script-anatomy.md +154 -0
  13. package/skills/create-skill/SKILL.md +291 -0
  14. package/skills/create-skill/assets/templates/SKILL.md.template +118 -0
  15. package/skills/create-skill/assets/templates/evals.json.template +36 -0
  16. package/skills/create-skill/assets/templates/grading.json.template +26 -0
  17. package/skills/create-skill/evals/benchmark.json +41 -0
  18. package/skills/create-skill/evals/evals.json +50 -0
  19. package/skills/create-skill/evals/grading-template.json +36 -0
  20. package/skills/create-skill/evals/near-misses.json +35 -0
  21. package/skills/create-skill/evals/trigger-queries.json +80 -0
  22. package/skills/create-skill/references/antipatterns.md +123 -0
  23. package/skills/create-skill/references/component-decomposition.md +130 -0
  24. package/skills/create-skill/references/content-quality-criteria.md +61 -0
  25. package/skills/create-skill/references/description-optimization.md +90 -0
  26. package/skills/create-skill/references/eval-methodology.md +100 -0
  27. package/skills/create-skill/references/fragility-matching.md +88 -0
  28. package/skills/create-skill/references/gotchas-patterns.md +80 -0
  29. package/skills/create-skill/references/specification.md +77 -0
  30. package/skills/create-skill/scripts/audit-antipatterns.mjs +164 -0
  31. package/skills/create-skill/scripts/compute-benchmark.mjs +111 -0
  32. package/skills/create-skill/scripts/run-cold-eval.mjs +118 -0
  33. package/skills/create-skill/scripts/scaffold-skill.mjs +86 -0
  34. package/skills/create-skill/scripts/validate-routing.mjs +137 -0
  35. package/skills/create-skill/scripts/validate-structure.mjs +223 -0
  36. package/skills/design-craft/SKILL.md +134 -0
  37. package/skills/design-craft/evals/benchmark.json +41 -0
  38. package/skills/design-craft/evals/evals.json +81 -0
  39. package/skills/design-craft/references/anti-slop-patterns.md +49 -0
  40. package/skills/design-craft/references/art-direction.md +89 -0
  41. package/skills/design-craft/references/design-engineering.md +122 -0
  42. package/skills/design-craft/references/motion-craft.md +124 -0
  43. package/skills/design-craft/references/process.md +47 -0
  44. package/skills/design-craft/references/review-checklist.md +121 -0
  45. package/skills/guardrails/SKILL.md +118 -0
  46. package/skills/guardrails/evals/benchmark.json +40 -0
  47. package/skills/guardrails/evals/evals.json +49 -0
  48. package/skills/guardrails/references/guardrails-patterns.md +43 -0
  49. package/skills/okf-docs/SKILL.md +79 -0
  50. package/skills/okf-docs/evals/benchmark.json +21 -0
  51. package/skills/okf-docs/evals/evals.json +37 -0
  52. package/skills/okf-docs/references/okf-spec.md +56 -0
  53. package/skills/okf-docs/scripts/validate-frontmatter.mjs +130 -0
  54. package/skills/openspec-harden/SKILL.md +138 -0
  55. package/skills/openspec-harden/evals/benchmark.json +40 -0
  56. package/skills/openspec-harden/evals/evals.json +38 -0
  57. package/skills/openspec-learn/SKILL.md +216 -0
  58. package/skills/openspec-learn/evals/benchmark.json +44 -0
  59. package/skills/openspec-learn/evals/evals.json +48 -0
  60. package/skills/openspec-learn/evals/retrieval-bench.json +27 -0
  61. package/skills/openspec-learn/references/conflict-handling.md +20 -0
  62. package/skills/openspec-learn/references/evaluation-methodology.md +126 -0
  63. package/skills/openspec-learn/references/examples.md +37 -0
  64. package/skills/openspec-learn/references/improvement-patterns.md +155 -0
  65. package/skills/openspec-learn/references/report-analysis.md +104 -0
  66. package/skills/openspec-learn/references/skill-quality.md +103 -0
  67. package/skills/openspec-learn/references/tool-type-detection.md +30 -0
  68. package/skills/openspec-report/SKILL.md +104 -0
  69. package/skills/openspec-report/assets/templates/assessment.md.template +84 -0
  70. package/skills/openspec-report/assets/templates/report.md.template +92 -0
  71. package/skills/openspec-report/evals/benchmark.json +44 -0
  72. package/skills/openspec-report/evals/evals.json +46 -0
  73. package/skills/qmd-research/SKILL.md +89 -0
  74. package/skills/qmd-research/evals/benchmark.json +40 -0
  75. package/skills/qmd-research/evals/evals.json +38 -0
  76. package/skills/qmd-research/references/index-management.md +69 -0
  77. package/skills/qmd-research/references/query-craft.md +82 -0
@@ -0,0 +1,80 @@
1
+ {
2
+ "skill_name": "create-skill",
3
+ "queries": [
4
+ {
5
+ "id": 1,
6
+ "query": "Build me a skill that processes PDFs and extracts text.",
7
+ "should_trigger": true
8
+ },
9
+ {
10
+ "id": 2,
11
+ "query": "I need a new agent skill for database queries.",
12
+ "should_trigger": true
13
+ },
14
+ {
15
+ "id": 3,
16
+ "query": "Create a skill that helps build other skills.",
17
+ "should_trigger": true
18
+ },
19
+ {
20
+ "id": 4,
21
+ "query": "Help me extract a reusable pattern from this workflow into a skill.",
22
+ "should_trigger": true
23
+ },
24
+ {
25
+ "id": 5,
26
+ "query": "Set up evaluation for my existing skill.",
27
+ "should_trigger": true
28
+ },
29
+ {
30
+ "id": 6,
31
+ "query": "Write a Python script to analyze data.",
32
+ "should_trigger": false
33
+ },
34
+ {
35
+ "id": 7,
36
+ "query": "Review this code for bugs.",
37
+ "should_trigger": false
38
+ },
39
+ {
40
+ "id": 8,
41
+ "query": "Create a bash script that automates deployment.",
42
+ "should_trigger": false
43
+ },
44
+ {
45
+ "id": 9,
46
+ "query": "Help me write documentation for my API.",
47
+ "should_trigger": false
48
+ },
49
+ {
50
+ "id": 10,
51
+ "query": "Fix the bug in my authentication middleware.",
52
+ "should_trigger": false
53
+ },
54
+ {
55
+ "id": 11,
56
+ "query": "I want to build something that helps with file processing.",
57
+ "should_trigger": true
58
+ },
59
+ {
60
+ "id": 12,
61
+ "query": "Make a tool for converting JSON to YAML.",
62
+ "should_trigger": false
63
+ },
64
+ {
65
+ "id": 13,
66
+ "query": "Design a skill for handling webhook integrations.",
67
+ "should_trigger": true
68
+ },
69
+ {
70
+ "id": 14,
71
+ "query": "Write unit tests for my function.",
72
+ "should_trigger": false
73
+ },
74
+ {
75
+ "id": 15,
76
+ "query": "Create a reusable skill from this task I just did.",
77
+ "should_trigger": true
78
+ }
79
+ ]
80
+ }
@@ -0,0 +1,123 @@
1
+ # Antipatterns
2
+
3
+ 16 audited failure modes from agent builds. Self-audit checklist.
4
+
5
+ ## A1: Phantom Tool Reference
6
+
7
+ **Pattern**: Prompt mentions tool agent can't call.
8
+
9
+ **BAD**: `call action_user_consumeFood` while agent tags exclude that gate.
10
+ **GOOD**: Prompt names only tools agent actually has; cross-role requests go via delegation.
11
+
12
+ ## A2: Duplicated Invariant
13
+
14
+ **Pattern**: Same rule copy-pasted across files AND injected at runtime.
15
+
16
+ **BAD**: 24-char ID rule in 3 prompts + auto-injected.
17
+ **GOOD**: Single source — global rules injected once; role-specific rules in one file only.
18
+
19
+ ## A3: Passive-Voice Triggers
20
+
21
+ **Pattern**: Role-job description instead of imperative trigger.
22
+
23
+ **BAD**: "You are the Mercenary. Evaluate auctions."
24
+ **GOOD**: "When `activeAuction != null AND payout_per_damage > market_avg` → call `bid_contract`."
25
+
26
+ ## A4: Copy-Pasted Cheat-Sheet
27
+
28
+ **Pattern**: Shared lookup list duplicated per file, each copy drifting.
29
+
30
+ **BAD**: Orchestrator lists 6 procedures, analyst lists 1 — same access, different coverage.
31
+ **GOOD**: One canonical `procedures.md`, linked/injected by both.
32
+
33
+ ## A5: Prose Bloat in Trigger Prompts
34
+
35
+ **Pattern**: Injecting prose-style rules into a tool-gate trigger prompt.
36
+
37
+ **BAD**: 21-rule style pack pasted into role file (role emits no human prose).
38
+ **GOOD**: Style pack injected only for prose-emitting roles.
39
+
40
+ ## A6: Mismatched Delegation in Prompt
41
+
42
+ **Pattern**: Prompt asserts delegation path runtime forbids.
43
+
44
+ **BAD**: "assigned by Orchestrator OR Economist" when only orchestrator delegates.
45
+ **GOOD**: "I receive tasks from the Orchestrator"; runtime enforces constraints.
46
+
47
+ ## A7: Stale-Data Trade
48
+
49
+ **Pattern**: Agent acts on memory past its `stale_after` without refetching.
50
+
51
+ **BAD**: Trades on a market note fetched 3h ago.
52
+ **GOOD**: Stale flag → refetch upstream before directive.
53
+
54
+ ## A8: Trust-Without-Freshness
55
+
56
+ **Pattern**: Verified fact trusted regardless of age.
57
+
58
+ **BAD**: "It's verified, so trust the 6h-old price."
59
+ **GOOD**: Stale overrides verified; un-staled-but-unverified still needs confirmation.
60
+
61
+ ## A9: Naive Context Truncation
62
+
63
+ **Pattern**: `slice(last N)` discards oldest messages, including relevant tool results.
64
+
65
+ **BAD**: 50-msg context → last 10.
66
+ **GOOD**: Summarize-older + preserve recents; or move settled facts to bundles.
67
+
68
+ ## A10: Silent Upstream Failure
69
+
70
+ **Pattern**: Upstream errors extract `data ?? body` blindly; agent acts on `null`.
71
+
72
+ **BAD**: Gateway returns `{}` on 500 → bot reports success.
73
+ **GOOD**: Schema validate; circuit breaker; skip + log + escalate.
74
+
75
+ ## A11: Hardcoded Fallback Credential
76
+
77
+ **Pattern**: When live auth fails, code substitutes leaked session id.
78
+
79
+ **BAD**: `vid` falls back to captured session token in source.
80
+ **GOOD**: Surface auth failure; never silently substitute; never commit credentials.
81
+
82
+ ## A12: Infinite Delegation
83
+
84
+ **Pattern**: Two agents ping-pong without depth cap.
85
+
86
+ **BAD**: No recursion bound → tokens → runaway.
87
+ **GOOD**: Depth cap on handoff (e.g., ≤3); force back to hub.
88
+
89
+ ## A13: One-Size-Fits-All Fragility
90
+
91
+ **Pattern**: Same gating strictness for creative writer and money transfer.
92
+
93
+ **BAD**: Article publisher gated like market sale.
94
+ **GOOD**: Tier guardrails by fragility class.
95
+
96
+ ## A14: Single File Omnibus
97
+
98
+ **Pattern**: 4000-line system prompt mixing everything.
99
+
100
+ **BAD**: One omnibus per agent, copy-paste everywhere.
101
+ **GOOD**: Persona / instructions / templates / data as separate files, loaded on demand.
102
+
103
+ ## A15: Vague "Professional" Success Bar
104
+
105
+ **Pattern**: Quality judged by adjectives, not scores.
106
+
107
+ **BAD**: Ship when prompts "look right."
108
+ **GOOD**: Trigger pass-rate + near-miss rate + output-lint violations.
109
+
110
+ ## A16: Promises vs Runtime Lint Missing
111
+
112
+ **Pattern**: No build-time check that prompt-named tools exist.
113
+
114
+ **BAD**: Phantom tool ships to production.
115
+ **GOOD**: Lint: every tool name in prompt ∈ runtime toolset for that agent.
116
+
117
+ ## Self-Audit Protocol
118
+
119
+ Before declaring skill complete:
120
+ 1. Read every file
121
+ 2. For each antipattern, confirm no file exhibits it
122
+ 3. Confirm: single-source invariants, no vendor installs, no phantom tools
123
+ 4. If any violation found, fix before shipping
@@ -0,0 +1,130 @@
1
+ # Component Decomposition
2
+
3
+ Gem-factory pattern for skill structure.
4
+
5
+ ## The Problem
6
+
7
+ Single omnibus files (4000+ lines) mix persona, instructions, templates, and data. This creates:
8
+ - High churn surface (everything changes together)
9
+ - Copy-paste drift (same rule in multiple places)
10
+ - Context bloat (agent loads everything, uses little)
11
+
12
+ ## The Solution
13
+
14
+ Decompose into discrete components:
15
+
16
+ | Component | Holds | File Shape | Churn Rate |
17
+ |-----------|-------|------------|------------|
18
+ | **Persona** | Identity, guiding principles | Short, stable | Low |
19
+ | **Instructions** | Workflows, decision trees | Markdown | Medium |
20
+ | **Templates** | Output shapes | Markdown with placeholders | High |
21
+ | **Data** | Factual reference, config | CSV/structured | High |
22
+
23
+ ## Component Types
24
+
25
+ ### Persona (Identity)
26
+
27
+ Who the skill is. Stable, rarely changes.
28
+
29
+ ```markdown
30
+ # persona.md
31
+
32
+ You are a data analyst specializing in CSV processing.
33
+ Your goal is to help users understand their tabular data.
34
+ You prefer clarity over cleverness.
35
+ ```
36
+
37
+ ### Instructions (Workflows)
38
+
39
+ How the skill works. Changes when process changes.
40
+
41
+ ```markdown
42
+ # instructions.md
43
+
44
+ ## Workflow
45
+
46
+ 1. Load data file
47
+ 2. Validate schema
48
+ 3. Compute statistics
49
+ 4. Generate output
50
+ ```
51
+
52
+ ### Templates (Output Shapes)
53
+
54
+ What the skill produces. Changes when format changes.
55
+
56
+ ```markdown
57
+ # templates/report.md
58
+
59
+ # Analysis Report
60
+
61
+ ## Summary
62
+ {{summary}}
63
+
64
+ ## Key Findings
65
+ {{findings}}
66
+
67
+ ## Recommendations
68
+ {{recommendations}}
69
+ ```
70
+
71
+ ### Data (Reference Material)
72
+
73
+ Factual reference. Changes when facts change.
74
+
75
+ ```csv
76
+ # data/schema.csv
77
+ column,type,required
78
+ id,integer,yes
79
+ name,string,yes
80
+ email,string,no
81
+ created_at,timestamp,yes
82
+ ```
83
+
84
+ ## File Structure
85
+
86
+ ```
87
+ skill-name/
88
+ ├── SKILL.md # Entry point, links to components
89
+ ├── persona.md # Identity (optional, can be in SKILL.md)
90
+ ├── instructions/ # Workflows
91
+ │ ├── workflow.md
92
+ │ └── decision-tree.md
93
+ ├── templates/ # Output shapes
94
+ │ ├── report.md
95
+ │ └── email.md
96
+ ├── data/ # Reference material
97
+ │ ├── schema.csv
98
+ │ └── examples.json
99
+ ├── scripts/ # Executable logic
100
+ │ └── process.mjs
101
+ └── references/ # Detailed docs
102
+ └── api-reference.md
103
+ ```
104
+
105
+ ## Single Source Per Invariant
106
+
107
+ **Rule**: Define each rule ONCE, reference it everywhere else.
108
+
109
+ Bad:
110
+ - `workflow.md`: "User IDs are 24-char hex"
111
+ - `validation.md`: "User IDs are 24-char hex"
112
+ - `api-rules.md`: "User IDs are 24-char hex"
113
+
114
+ Good:
115
+ - `api-rules.md`: "User IDs are 24-char hex"
116
+ - `workflow.md`: "See api-rules.md for ID format"
117
+ - `validation.md`: "See api-rules.md for ID format"
118
+
119
+ ## When to Decompose
120
+
121
+ Decompose when:
122
+ - File exceeds 500 lines
123
+ - Multiple concerns in one file
124
+ - Same content appears in multiple places
125
+ - Different parts change at different rates
126
+
127
+ Don't decompose when:
128
+ - Skill is simple (<100 lines total)
129
+ - All content changes together
130
+ - Decomposition adds more overhead than it saves
@@ -0,0 +1,61 @@
1
+ # Content Quality Criteria
2
+
3
+ What makes good skill instructions.
4
+
5
+ ## Clarity
6
+
7
+ - **One interpretation**: Instructions should have exactly one meaning
8
+ - **Concrete, not abstract**: "Run `./scripts/validate.sh`" not "validate the output"
9
+ - **Sequential steps**: Number steps when order matters
10
+ - **Active voice**: "Check X" not "X should be checked"
11
+
12
+ ## Actionability
13
+
14
+ - **Agent can follow without guessing**: Every step is executable
15
+ - **Tools specified**: Name the exact tool/command to use
16
+ - **Inputs/outputs defined**: What goes in, what comes out
17
+ - **Error handling**: What to do when things fail
18
+
19
+ ## Edge Cases
20
+
21
+ - **Document non-obvious behaviors**: Things the agent would get wrong without being told
22
+ - **Input validation**: What happens with malformed input
23
+ - **Boundary conditions**: Empty inputs, large inputs, special characters
24
+ - **Failure modes**: Network errors, permission denied, missing files
25
+
26
+ ## Examples
27
+
28
+ - **One canonical example**: Show the happy path clearly
29
+ - **BAD/GOOD pairs**: Show what to avoid and what to do
30
+ - **Real-world context**: Use realistic inputs, not "foo" and "bar"
31
+ - **Input → Output**: Show the transformation
32
+
33
+ ## Gotchas
34
+
35
+ - **Environment-specific facts**: Things that defy reasonable assumptions
36
+ - **Naming inconsistencies**: Same concept, different names across systems
37
+ - **Hidden preconditions**: What must be true before running
38
+ - **Non-obvious side effects**: What happens beyond the obvious
39
+
40
+ ## Progressive Disclosure
41
+
42
+ - **SKILL.md**: Core instructions, always loaded (<500 lines)
43
+ - **references/**: Detailed docs, loaded on-demand
44
+ - **scripts/**: Executable logic, invoked when needed
45
+ - **assets/**: Templates and static resources
46
+
47
+ ## Fragility Matching
48
+
49
+ | Task Type | Specificity Level | Example |
50
+ |-----------|-------------------|---------|
51
+ | Mutation (create/delete/modify) | Strict, step-by-step | "Run exactly this command. Do not add flags." |
52
+ | Read-only (query/analyze) | Loose, latitude | "Query the database. Format as you see fit." |
53
+ | Creative (prose/design) | Low specificity | "Write a summary. Follow the style guide." |
54
+
55
+ ## Antipatterns to Avoid
56
+
57
+ - **Vague instructions**: "Handle errors appropriately"
58
+ - **Over-specification**: "Use exactly 3 spaces for indentation"
59
+ - **Missing context**: "Run the script" (which script?)
60
+ - **Assumed knowledge**: "Use the standard approach" (what standard?)
61
+ - **Copy-paste**: Same rule in multiple places (drift risk)
@@ -0,0 +1,90 @@
1
+ # Description Optimization
2
+
3
+ How to improve skill triggering through description tuning.
4
+
5
+ ## How Triggering Works
6
+
7
+ 1. **Discovery**: Agent loads `name` and `description` of all skills
8
+ 2. **Activation**: When task matches description, agent loads full SKILL.md
9
+ 3. **Execution**: Agent follows instructions
10
+
11
+ Description carries the entire burden of triggering.
12
+
13
+ ## Writing Effective Descriptions
14
+
15
+ ### Principles
16
+
17
+ - **Imperative phrasing**: "Use this skill when..." not "This skill does..."
18
+ - **Focus on user intent**: What the user is trying to achieve
19
+ - **Err on pushy**: Explicitly list contexts where skill applies
20
+ - **Concise**: A few sentences to short paragraph
21
+
22
+ ### Good vs Bad
23
+
24
+ Good:
25
+ ```yaml
26
+ description: >
27
+ Analyze CSV and tabular data files — compute summary statistics,
28
+ add derived columns, generate charts, and clean messy data. Use this
29
+ skill when the user has a CSV, TSV, or Excel file and wants to
30
+ explore, transform, or visualize the data, even if they don't
31
+ explicitly mention "CSV" or "analysis."
32
+ ```
33
+
34
+ Bad:
35
+ ```yaml
36
+ description: Process CSV files.
37
+ ```
38
+
39
+ ## Trigger Test Queries
40
+
41
+ Create 10-20 queries: mix of should-trigger and shouldn't-trigger.
42
+
43
+ ### Should-Trigger Queries
44
+
45
+ - Vary phrasing (formal, casual, terse)
46
+ - Vary explicitness (direct naming vs describing need)
47
+ - Vary detail (terse vs context-heavy)
48
+ - Include cases where skill helps but connection isn't obvious
49
+
50
+ ### Should-Not-Trigger Queries (Near-Misses)
51
+
52
+ Strong negative examples share keywords but need different handling:
53
+ - "Write a Python script that reads a CSV and uploads to Postgres" (involves CSV, but task is database ETL, not analysis)
54
+ - "Update the formulas in my Excel budget spreadsheet" (shares spreadsheet concept, but needs Excel editing, not CSV analysis)
55
+
56
+ Weak negatives (don't test anything):
57
+ - "Write a fibonacci function" (obviously irrelevant)
58
+ - "What's the weather today?" (no keyword overlap)
59
+
60
+ ## Optimization Loop
61
+
62
+ 1. **Evaluate** current description on train + validation sets
63
+ 2. **Identify failures** in train set only
64
+ 3. **Revise description**:
65
+ - Should-trigger failing → broaden scope
66
+ - Should-not-trigger failing → add specificity
67
+ - Avoid keyword overfitting → find general category
68
+ 4. **Repeat** until train set passes or plateau
69
+ 5. **Select best** by validation pass rate
70
+
71
+ ## Train/Validation Split
72
+
73
+ - **Train set (~60%)**: Guides changes
74
+ - **Validation set (~40%)**: Checks generalization
75
+
76
+ Keep split fixed across iterations. Both sets need proportional mix of positive/negative queries.
77
+
78
+ ## Avoiding Overfitting
79
+
80
+ - Don't add specific keywords from failed queries
81
+ - Find the general category those queries represent
82
+ - Try structurally different approaches when stuck
83
+ - Stay under 1024 chars
84
+
85
+ ## Applying the Result
86
+
87
+ 1. Update `description` field in SKILL.md
88
+ 2. Verify under 1024-char limit
89
+ 3. Manual sanity check with a few prompts
90
+ 4. Optional: 5-10 fresh queries for rigorous verification
@@ -0,0 +1,100 @@
1
+ # Evaluation Methodology
2
+
3
+ Full eval framework for skill quality assurance.
4
+
5
+ ## Two Eval Halves
6
+
7
+ ```
8
+ ┌─ TRIGGER evals ─────────────────────────────┐
9
+ │ Does the right skill fire at the right time?│
10
+ │ Positive cases + near-miss negatives │
11
+ └──────────────────────────────────────────────┘
12
+ ┌─ OUTPUT evals ───────────────────────────────┐
13
+ │ Does the output satisfy quality rules? │
14
+ │ Deterministic + semantic judgment │
15
+ └──────────────────────────────────────────────┘
16
+ ```
17
+
18
+ ## Test Case Design
19
+
20
+ ### Positive Cases
21
+
22
+ Prompts that SHOULD trigger the skill:
23
+ - Varied phrasing (formal, casual, terse)
24
+ - Different levels of detail
25
+ - Realistic context (file paths, names, specifics)
26
+
27
+ ### Near-Miss Negatives (Critical)
28
+
29
+ Prompts that should NOT trigger the skill:
30
+ - Share keywords but need different handling
31
+ - Similar domain but different task
32
+ - Without these, over-firing passes silently
33
+
34
+ Example for CSV analysis skill:
35
+ - ✅ "Analyze my sales CSV and make a chart" (should trigger)
36
+ - ❌ "Write a Python script that reads a CSV and uploads to Postgres" (near-miss: involves CSV but different task)
37
+
38
+ ### Assertion Writing
39
+
40
+ Good assertions:
41
+ - `"The output file is valid JSON"` — programmatically verifiable
42
+ - `"The chart has labeled axes"` — specific and observable
43
+ - `"The report includes at least 3 recommendations"` — countable
44
+
45
+ Bad assertions:
46
+ - `"The output is good"` — too vague
47
+ - `"Uses exactly the phrase 'Total Revenue: $X'"` — too brittle
48
+
49
+ ## Grading Principles
50
+
51
+ - **Require concrete evidence for PASS**: Don't give benefit of the doubt
52
+ - **Review assertions themselves**: Are they too easy, too hard, or unverifiable?
53
+ - **Blind comparison**: Compare outputs without revealing which version
54
+
55
+ ## Benchmark Computation
56
+
57
+ ```json
58
+ {
59
+ "run_summary": {
60
+ "with_skill": {
61
+ "pass_rate": { "mean": 0.83, "stddev": 0.06 },
62
+ "time_seconds": { "mean": 45.0, "stddev": 12.0 },
63
+ "tokens": { "mean": 3800, "stddev": 400 }
64
+ },
65
+ "without_skill": {
66
+ "pass_rate": { "mean": 0.33, "stddev": 0.10 },
67
+ "time_seconds": { "mean": 32.0, "stddev": 8.0 },
68
+ "tokens": { "mean": 2100, "stddev": 300 }
69
+ },
70
+ "delta": {
71
+ "pass_rate": 0.50,
72
+ "time_seconds": 13.0,
73
+ "tokens": 1700
74
+ }
75
+ }
76
+ }
77
+ ```
78
+
79
+ ## Calibration Loop
80
+
81
+ ```
82
+ write prompt → run evals → analyze failures → edit prompt → re-run
83
+ ▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔▔
84
+ stop when pass-rate ≥ target AND near-miss rate = 0
85
+ ```
86
+
87
+ ## Analysis Patterns
88
+
89
+ - **Remove assertions that always pass**: They don't test anything
90
+ - **Investigate assertions that always fail**: Are they broken?
91
+ - **Study assertions that pass with skill but fail without**: This is where skill adds value
92
+ - **Check outliers**: If one eval takes 3x longer, find why
93
+
94
+ ## Tiered Enforcement
95
+
96
+ | Tier | Action | When |
97
+ |------|--------|------|
98
+ | deny | Block outright | Zero expected false positives |
99
+ | warn | Flag, allow | Minor issues |
100
+ | ask | Pause for confirmation | Ambiguous cases |
@@ -0,0 +1,88 @@
1
+ # Fragility Matching
2
+
3
+ Match instruction specificity to task fragility.
4
+
5
+ ## Fragility Classification
6
+
7
+ | Class | Description | Example | Specificity |
8
+ |-------|-------------|---------|-------------|
9
+ | **Mutation** | Creates, deletes, or modifies data | Place order, delete account, update record | Strict, step-by-step, precondition gates |
10
+ | **Read-only** | Queries or analyzes data | Search database, generate report, analyze CSV | Loose triggers, latitude for formatting |
11
+ | **Creative** | Prose, design, creative output | Write summary, design layout, draft email | Low specificity, examples over rules |
12
+
13
+ ## Mutation Tasks (Strict)
14
+
15
+ When skill involves creating, deleting, or modifying data:
16
+
17
+ - **Precondition gates**: Check before acting
18
+ - **Exact commands**: "Run exactly this command. Do not add flags."
19
+ - **Validation loops**: Verify before proceeding
20
+ - **Rollback paths**: What to do if something fails
21
+ - **Idempotency**: Safe to retry
22
+
23
+ Example:
24
+ ```markdown
25
+ ## Place Order
26
+
27
+ 1. Verify inventory: `./scripts/check-inventory.sh <item-id>`
28
+ 2. ONLY if in stock: `./scripts/place-order.sh <item-id> <quantity>`
29
+ 3. Verify order: `./scripts/verify-order.sh <order-id>`
30
+ 4. NEVER skip step 1. NEVER add flags to step 2.
31
+ ```
32
+
33
+ ## Read-Only Tasks (Loose)
34
+
35
+ When skill involves querying or analyzing data:
36
+
37
+ - **Flexible triggers**: "When user asks about X"
38
+ - **Format latitude**: "Format as you see fit"
39
+ - **Optional steps**: "Optionally include Y"
40
+ - **Error handling**: "If query fails, report error"
41
+
42
+ Example:
43
+ ```markdown
44
+ ## Analyze Data
45
+
46
+ 1. Load the data file
47
+ 2. Compute summary statistics
48
+ 3. Generate visualizations as appropriate
49
+ 4. Present findings in a clear format
50
+ ```
51
+
52
+ ## Creative Tasks (Low Specificity)
53
+
54
+ When skill involves prose, design, or creative output:
55
+
56
+ - **Style guidelines**: "Follow the style guide"
57
+ - **Examples**: Show BAD/GOOD pairs
58
+ - **Constraints**: "Don't include X"
59
+ - **Latitude**: Let agent decide approach
60
+
61
+ Example:
62
+ ```markdown
63
+ ## Write Summary
64
+
65
+ - One paragraph overview
66
+ - Include key findings
67
+ - Use clear, concise language
68
+ - See references/style-guide.md for tone
69
+ ```
70
+
71
+ ## Fragility Matrix
72
+
73
+ ```
74
+ mutation (create/delete/modify) ─▶ strict preconditions, gating, deterrents
75
+ read-only (query/analyze) ─▶ loose triggers, free text
76
+ creative (prose/design) ─▶ low specificity, examples
77
+
78
+ ⚠️ Same skill should never treat a buy-button with the looseness of a weather check
79
+ ```
80
+
81
+ ## Decision Guide
82
+
83
+ Ask:
84
+ 1. Does this skill change state? → Mutation → Strict
85
+ 2. Does this skill only read state? → Read-only → Loose
86
+ 3. Does this skill produce creative output? → Creative → Low specificity
87
+
88
+ When uncertain, default to stricter. Can always loosen later.