@meyverick/agentic 5.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/AGENTS.md +234 -0
  2. package/CHANGELOG.md +236 -0
  3. package/README.md +50 -0
  4. package/install.ts +349 -0
  5. package/package.json +37 -0
  6. package/scripts/check-deps.mjs +587 -0
  7. package/scripts/git-dl.mjs +100 -0
  8. package/skills/check/SKILL.md +108 -0
  9. package/skills/check/evals/benchmark.json +40 -0
  10. package/skills/check/evals/evals.json +38 -0
  11. package/skills/check/references/diagnostic-matrix.md +170 -0
  12. package/skills/check/references/script-anatomy.md +154 -0
  13. package/skills/create-skill/SKILL.md +291 -0
  14. package/skills/create-skill/assets/templates/SKILL.md.template +118 -0
  15. package/skills/create-skill/assets/templates/evals.json.template +36 -0
  16. package/skills/create-skill/assets/templates/grading.json.template +26 -0
  17. package/skills/create-skill/evals/benchmark.json +41 -0
  18. package/skills/create-skill/evals/evals.json +50 -0
  19. package/skills/create-skill/evals/grading-template.json +36 -0
  20. package/skills/create-skill/evals/near-misses.json +35 -0
  21. package/skills/create-skill/evals/trigger-queries.json +80 -0
  22. package/skills/create-skill/references/antipatterns.md +123 -0
  23. package/skills/create-skill/references/component-decomposition.md +130 -0
  24. package/skills/create-skill/references/content-quality-criteria.md +61 -0
  25. package/skills/create-skill/references/description-optimization.md +90 -0
  26. package/skills/create-skill/references/eval-methodology.md +100 -0
  27. package/skills/create-skill/references/fragility-matching.md +88 -0
  28. package/skills/create-skill/references/gotchas-patterns.md +80 -0
  29. package/skills/create-skill/references/specification.md +77 -0
  30. package/skills/create-skill/scripts/audit-antipatterns.mjs +164 -0
  31. package/skills/create-skill/scripts/compute-benchmark.mjs +111 -0
  32. package/skills/create-skill/scripts/run-cold-eval.mjs +118 -0
  33. package/skills/create-skill/scripts/scaffold-skill.mjs +86 -0
  34. package/skills/create-skill/scripts/validate-routing.mjs +137 -0
  35. package/skills/create-skill/scripts/validate-structure.mjs +223 -0
  36. package/skills/design-craft/SKILL.md +134 -0
  37. package/skills/design-craft/evals/benchmark.json +41 -0
  38. package/skills/design-craft/evals/evals.json +81 -0
  39. package/skills/design-craft/references/anti-slop-patterns.md +49 -0
  40. package/skills/design-craft/references/art-direction.md +89 -0
  41. package/skills/design-craft/references/design-engineering.md +122 -0
  42. package/skills/design-craft/references/motion-craft.md +124 -0
  43. package/skills/design-craft/references/process.md +47 -0
  44. package/skills/design-craft/references/review-checklist.md +121 -0
  45. package/skills/guardrails/SKILL.md +118 -0
  46. package/skills/guardrails/evals/benchmark.json +40 -0
  47. package/skills/guardrails/evals/evals.json +49 -0
  48. package/skills/guardrails/references/guardrails-patterns.md +43 -0
  49. package/skills/okf-docs/SKILL.md +79 -0
  50. package/skills/okf-docs/evals/benchmark.json +21 -0
  51. package/skills/okf-docs/evals/evals.json +37 -0
  52. package/skills/okf-docs/references/okf-spec.md +56 -0
  53. package/skills/okf-docs/scripts/validate-frontmatter.mjs +130 -0
  54. package/skills/openspec-harden/SKILL.md +138 -0
  55. package/skills/openspec-harden/evals/benchmark.json +40 -0
  56. package/skills/openspec-harden/evals/evals.json +38 -0
  57. package/skills/openspec-learn/SKILL.md +216 -0
  58. package/skills/openspec-learn/evals/benchmark.json +44 -0
  59. package/skills/openspec-learn/evals/evals.json +48 -0
  60. package/skills/openspec-learn/evals/retrieval-bench.json +27 -0
  61. package/skills/openspec-learn/references/conflict-handling.md +20 -0
  62. package/skills/openspec-learn/references/evaluation-methodology.md +126 -0
  63. package/skills/openspec-learn/references/examples.md +37 -0
  64. package/skills/openspec-learn/references/improvement-patterns.md +155 -0
  65. package/skills/openspec-learn/references/report-analysis.md +104 -0
  66. package/skills/openspec-learn/references/skill-quality.md +103 -0
  67. package/skills/openspec-learn/references/tool-type-detection.md +30 -0
  68. package/skills/openspec-report/SKILL.md +104 -0
  69. package/skills/openspec-report/assets/templates/assessment.md.template +84 -0
  70. package/skills/openspec-report/assets/templates/report.md.template +92 -0
  71. package/skills/openspec-report/evals/benchmark.json +44 -0
  72. package/skills/openspec-report/evals/evals.json +46 -0
  73. package/skills/qmd-research/SKILL.md +89 -0
  74. package/skills/qmd-research/evals/benchmark.json +40 -0
  75. package/skills/qmd-research/evals/evals.json +38 -0
  76. package/skills/qmd-research/references/index-management.md +69 -0
  77. package/skills/qmd-research/references/query-craft.md +82 -0
@@ -0,0 +1,38 @@
1
+ {
2
+ "skill_name": "openspec-harden",
3
+ "evals": [
4
+ {
5
+ "id": 1,
6
+ "prompt": "Harden the proposal, specs, and tasks for the auth change with concrete file paths and test commands.",
7
+ "expected_output": "Agent loads openspec-harden skill and inspects proposal.md, design.md, specs/, and tasks.md to enrich them with grounded codebase context and exact paths.",
8
+ "files": [],
9
+ "assertions": [
10
+ "Skill activates on hardening request",
11
+ "Artifacts enriched with concrete paths",
12
+ "Source code is not edited"
13
+ ]
14
+ },
15
+ {
16
+ "id": 2,
17
+ "prompt": "Implement the tasks in tasks.md for the auth change and modify the codebase.",
18
+ "expected_output": "Agent does NOT activate openspec-harden. Implementation and code-writing are handled by openspec-apply-change; openspec-harden only enriches planning artifacts before apply.",
19
+ "files": [],
20
+ "assertions": [
21
+ "Anti-trigger fires: request routes to openspec-apply-change, not openspec-harden",
22
+ "No proposal hardening workflow executed",
23
+ "No planning artifacts modified"
24
+ ]
25
+ },
26
+ {
27
+ "id": 3,
28
+ "prompt": "Improve change for fresh agent by resolving either/or file paths into single targets.",
29
+ "expected_output": "Agent enriches tasks with single concrete paths and verify steps without implementing features.",
30
+ "files": [],
31
+ "assertions": [
32
+ "Single concrete paths assigned",
33
+ "Zero code mutations made to project files",
34
+ "Verify steps populated"
35
+ ]
36
+ }
37
+ ]
38
+ }
@@ -0,0 +1,216 @@
1
+ ---
2
+ name: openspec-learn
3
+ description: >
4
+ Analyze reports from `openspec-report` to generate OpenSpec proposals for
5
+ skill improvements. Use when analyzing reports to plan skill creation,
6
+ improvements, or lifecycle actions. Do NOT use when proposing a new change
7
+ from scratch (use openspec-propose), generating reports from archived changes,
8
+ or implementing proposals.
9
+ allowed-tools: Bash(openspec:*), Bash(node:*), Bash(mkdir:*), Bash(ls:*)
10
+ license: MIT
11
+ compatibility: Requires openspec CLI and bun.
12
+ metadata:
13
+ author: agentic
14
+ version: "1.3.1"
15
+ positive_triggers:
16
+ - "analyze reports and improve skills"
17
+ - "generate proposals from report insights"
18
+ - "process reports and create improvement plans"
19
+ anti_triggers:
20
+ - "propose a new change from scratch"
21
+ - "generate a report from an archived change"
22
+ - "implement a proposal or apply changes"
23
+ ---
24
+
25
+ # Openspec Learn
26
+
27
+ Analyze reports and generate OpenSpec proposals for improvements.
28
+
29
+ ## Quick Start
30
+
31
+ - **With argument**: Process single report: `./openspec/reports/<name>/`
32
+ - **Without argument**: Process ALL reports: `./openspec/reports/*/` (excluding `archives/`)
33
+ - **With store**: Forward selected `--store <id>` on applicable commands (`new change`, `status`, `instructions`, `list`, `show`, `validate`, `archive`, `doctor`, `context`, `schemas`, `view`); sticky for workflow; nearest-root if unselected
34
+
35
+ ## Workflow
36
+
37
+ ### Phase 1: Report Scanning
38
+
39
+ ```bash
40
+ ls ./openspec/reports/ | grep -v archives
41
+ ```
42
+
43
+ ### Phase 2: Report Analysis
44
+
45
+ **From report.md:** OKF frontmatter, problem statement, approach, specs, implementation, validation, trade-offs, follow-ups.
46
+
47
+ **From assessment.md:** knowledge gaps, difficulty ratings, tool improvements, what could have helped.
48
+
49
+ ### Phase 2a: Semantic Collision Detection
50
+
51
+ Scan existing skills for description overlap with proposed tool:
52
+
53
+ ```bash
54
+ grep -r "description:" .agents/skills/*/SKILL.md ./project/skills/*/SKILL.md 2>/dev/null
55
+ ```
56
+
57
+ Compare proposed tool's domain keywords against existing descriptions. If overlap detected → flag collision in proposal, suggest updating existing skill instead of creating new one.
58
+
59
+ ### Phase 2b: Trigger Metadata Extraction
60
+
61
+ Extract trigger candidates from report assessment:
62
+ - What queries/situations led to this change? → candidate positive_triggers
63
+ - What was confusing or out-of-scope? → candidate anti_triggers
64
+
65
+ Include as "Suggested Triggers" section in proposal.
66
+
67
+ ### Phase 2c: Ownership Pre-Check + Single-Responsibility Pre-Check
68
+
69
+ **Ownership pre-check (MANDATORY, runs first).** Classify every candidate target skill by ownership using this precedence chain — recorded facts before conventions, conventions before residual judgment:
70
+
71
+ 1. Inside the agentic repo itself → `project/skills/*` is owned (editable only by the project owner and the owner's trusted assistant session)
72
+ 2. Listed in `.agents/skills/.agentic-manifest.json` → agentic-distributed → **external** (read-only for consumers)
73
+ 3. Located in `~/.pi/agent/skills/` (not created by this project) → **external**
74
+ 4. Explicit verdict in `.agents/skills/.ownership.json` (`{"owned": [...], "external": [...]}`) → as declared
75
+ 5. Named `openspec-*` and authored by OpenSpec (upstream path `.agents/skills/openspec-*`, frontmatter `author: openspec`) → **external** (public namespace claim; does NOT apply to agentic-authored `project/skills/openspec-{learn,report,harden}`, which remain owned at source per item 1 and read-only once installed elsewhere)
76
+ 6. Any other skill in the project's `.agents/skills/` → project-created → editable
77
+ 7. Unresolved after all checks → unknown = **external**, ask the user
78
+
79
+ Rules:
80
+ - Proposals MUST NOT target external skills. Never edit installed/upstream files under `.agents/skills/*` in place — consumer agents treat installed skills as read-only; the next atomic-replace install silently wipes in-place edits. Reroute improvements upstream or to project-local placements.
81
+ - Domain-specific knowledge belongs in project-local homes (wiki, checklists, project-created skills). If no local home exists, ask the user where to place it.
82
+ - Genuinely generic improvements to external skills: record as an upstream recommendation for the user (issue/PR), never edit directly.
83
+ - When you ask the user about an unknown skill's ownership, record the verdict in `.ownership.json` so each question is asked once.
84
+
85
+ **Single-responsibility pre-check** (only for targets that passed the ownership check):
86
+
87
+ For proposed updates to existing skills:
88
+ - Does the improvement align with the skill's atomic intent?
89
+ - Or does it add a new operational domain?
90
+
91
+ If scope expansion detected → propose splitting into separate skill instead of updating.
92
+
93
+ ### Phase 2d: Value Justification
94
+
95
+ Estimate whether creating/updating a skill improves outcomes by >= 20%:
96
+ - Assessment difficulty >= 3/5 OR knowledge gaps identified → justified
97
+ - All difficulty <= 2/5 AND no gaps → flag as low value in proposal
98
+ - **Frequency × cost heuristic (new):** Read `Re-use Score` (high=3, medium=2, low=1) and `Time Cost` from report/assessment; prioritize `frequency × cost` — a `high` re-use gap that cost 60m and recurs ≥2 times outranks a singleton `low` gap even if its difficulty was 5/5
99
+
100
+ ### Phase 2e: Context Budget Impact
101
+
102
+ Estimate Tier 1 + Tier 2 token cost of proposed skill based on similar skills. Warn if >2000 tokens in proposal.
103
+
104
+ ### Phase 2f: Recurring Gap Clustering (compound learning)
105
+
106
+ Before proposing, cluster gaps and gotchas across ALL reports in `openspec/reports/*` (excluding `archives/`).
107
+
108
+ **Primary Clusterer (Hybrid Semantic via QMD):**
109
+ Query the project-local index (never a global or shared index) scoped to `openspec` and `references` collections:
110
+ ```bash
111
+ qmd query $'intent: cluster recurring knowledge gaps, skills gaps, and gotchas across reports\nlex: knowledge gaps skills gaps gotchas mental model shift\nvec: recurring difficulties, surprises, and lessons learned from past changes' --json -n 20 -c openspec -c references
112
+ ```
113
+ Retrieve evidence for clustered items via `qmd multi-get "<docids>" --json`. Clusters MUST carry `qmd://` document IDs as evidence in the proposal. Semantic matching groups paraphrased descriptions of the same underlying obstacle even when distinct keywords were used.
114
+
115
+ **Loud Fallback (Keyword Grep):**
116
+ If the QMD daemon is unreachable or the index is unhealthy (`qmd status` fails):
117
+ ```bash
118
+ # Loud fallback: note "qmd unavailable, grep fallback" in proposal
119
+ for f in openspec/reports/*/report.md openspec/reports/*/assessment.md; do [ -f "$f" ] && echo "== $f ==" && grep -E "Knowledge gaps|Skills gaps|Concrete Gotcha|Before \(old|After \(new|Re-use Score|Time Cost" "$f" | head -n 20; done
120
+ ```
121
+ Note `qmd unavailable, grep fallback` explicitly in the proposal's Analysis section.
122
+
123
+ Group recurring findings by topic/keyword and count occurrences. Prioritize a cluster that recurs in ≥2 reports over a singleton, even if the singleton's difficulty was higher. Record the cluster table (topic/keyword → count → representative gap → `qmd://` evidence) in the proposal's Analysis section; this drives `Deferred:` decisions in Phase 5.
124
+
125
+ **Dashboard generation (human curation view):** After clustering, write/overwrite `openspec/reports/dashboard.md` (never archived, never auto-loaded at agent startup):
126
+ ```bash
127
+ # Header: generated at $(date -u +"%Y-%m-%dT%H:%M:%SZ") from N reports
128
+ # Table: keyword | count | avg Time Cost | avg Re-use (high=3/med=2/low=1) | owning skill | m sorted by frequency×cost desc
129
+ # Source: grep Re-use Score + Time Cost + keyword + owning skill from clustered reports + benchmark.json m
130
+ # If no reports, write: No reports yet — run /opsx-report after a hard task
131
+ ```
132
+ Header `generated at <timestamp> from N reports`; table sorted `frequency × cost` desc (count × Re-use weight × avg Time Cost). Overwrite on each learn run; human reads it to curate manual reports.
133
+
134
+ **Lifecycle check (prune/merge/split):** After clustering, also evaluate:
135
+ - **Prune:** If a skill's usage count is 0 in last 10 reports OR its `benchmark.json` `behavioral.m < 0.2`, mark `Prune: <skill> — 0/10 or low m`
136
+ - **Merge:** If ≥3 shared gaps under same keyword across 2 skills, mark `Merge: <target> ← <a> + <b> — shared keyword`
137
+ - **Split:** If a skill's description would need `and`, mark `Split: <skill> → <a> + <b> — single-responsibility`
138
+ Each emits one-path `What Changes: Remove project/skills/<skill>/` or `Merge: ...` with `Deferred:` for rejected lifecycle candidates.
139
+
140
+ ### Phase 3: Tool Type Determination
141
+
142
+ See [references/tool-type-detection.md](references/tool-type-detection.md) for full detection matrix.
143
+
144
+ ### Phase 4: Conflict Resolution
145
+
146
+ See [references/conflict-handling.md](references/conflict-handling.md) for merging strategies.
147
+
148
+ ### Phase 5: OpenSpec Proposal Generation
149
+
150
+ Generate proposal including:
151
+ - What to build + why (from report + assessment)
152
+ - Suggested Triggers section (from Phase 2b) — MUST use the exact frontmatter field names `positive_triggers` and `anti_triggers` as list headers, with each value formatted for verbatim transfer into a skill's frontmatter (no prose labels like "Positive:" or "Negative:")
153
+ - **Contrast and Anti-example hints (from report's Mental Model Shift / Concrete Gotcha):** Carry `Contrast: Before X → After Y` (1–2 lines) and `Anti-example: Do NOT: <before code> → Do: <after code>` verbatim into What Changes so `create-skill` can generate contrast tables and anti-examples as first-class content
154
+ - **Layer-aware decision note:** State `Layer: 1 (always) vs 2 (on-demand skill) vs 3 (gate)` — for `Re-use Score: high` + `Time Cost >30m` + recurring cluster, suggest `Layer 1/3` promotion (e.g., `guardrails` skill or `AGENTS.md` Must-read pointer); otherwise `Layer 2` new/updated skill; no gate is implemented in this change, only the hint
155
+ - **Lifecycle What Changes (when applicable):** Emit `What Changes: Remove project/skills/<skill>/` for prune, `What Changes: Merge project/skills/<target>/ ← <a> + <b>` for merge, or `Split` with two one-path creates; each with `Impact: evals + manifest updated` and `Deferred:` for rejected lifecycle candidates; cap 8 enforced here (see Gotchas)
156
+ - Value Justification section (from Phase 2d, now including frequency × cost)
157
+ - Collision warnings (from Phase 2a)
158
+ - Context budget impact (from Phase 2e)
159
+ - Clustering summary (from Phase 2f) — keyword → count → representative gap
160
+ - **Dashboard citation (MANDATORY when lifecycle or create):** Each `create`/`prune`/`merge`/`split` in `What Changes` MUST cite `Source: dashboard.md#<keyword> — <count>× <Re-use> <time> (avg)` with verification path `openspec/reports/dashboard.md` — e.g., `Source: dashboard.md#lifetime — 4× high 45m avg`
161
+ - Evals impact statement (MANDATORY when the proposal modifies an existing skill): state whether `evals/evals.json` changes; if triggers are added or altered, include at least one matching eval entry in Impact. No trigger changes → state that existing evals remain valid.
162
+ - Deferred signals line (when the source report/assessment contains more improvement candidates than the proposal adopts): name each unadopted candidate with a one-line reason, so deferral is explicit rather than silent
163
+
164
+ **One-path rule**: every What Changes and Impact item names exactly ONE concrete target file path. Either/or targets ("X or Y") are prohibited — resolve the choice during design, before tasks are written. Task verify clauses must reference the same single path.
165
+
166
+ **OpenSpec CLI Contract & Store Forwarding:**
167
+ - Scaffold new changes with `openspec new change "<name>"` (forwarding `--store <id>` if selected). Never create change directories by hand.
168
+ - Forward selected `--store <id>` on applicable commands (`new change`, `status`, `instructions`, `list`, `show`, `validate`, `archive`, `doctor`, `context`, `schemas`, `view`); keep sticky; nearest-root if unselected. Other commands run unflagged.
169
+ - Failure envelope: Parse stdout as single JSON payload; stderr carries prose/spinners/store banner (never parse stderr as JSON). On exit 1, parse `status: [diagnostic]` array (`severity`, `code`, `message`, `fix`). Exit 130 = prompt cancelled.
170
+ - Payload casing: Workflow payloads use `camelCase`; store payloads use `snake_case` (`root.store_id` always `snake_case`).
171
+ - Archive delegation: Follow-on change archiving delegates to `openspec archive --json` (`archivedAs`, `specsUpdated`, `totals`, `warnings`). Never hand `mv` change directories or hand-merge delta specs into main specs.
172
+
173
+ The proposal instructs the AI agent to invoke create-skill during `/opsx-apply`.
174
+
175
+ ### Phase 6: Archive Processed Reports
176
+
177
+ ```bash
178
+ mkdir -p ./openspec/reports/archives
179
+ mv ./openspec/reports/<name> ./openspec/reports/archives/
180
+ # dashboard.md never moves — it stays at openspec/reports/dashboard.md (human curation view, not a report)
181
+ ```
182
+ Note: Only processed reports are moved via `mv`. OpenSpec changes MUST be archived via `openspec archive --json`, never by hand `mv`.
183
+
184
+ ### Phase 7: Display Summary
185
+
186
+ Report count, proposal count, archive count, next steps.
187
+
188
+ ## Gotchas
189
+
190
+ - **Collision detection prevents dilution**: Two skills with similar descriptions reduce routing confidence for both.
191
+ - **Trigger metadata saves Discovery time**: Pre-filling triggers from assessment data gives create-skill a head start.
192
+ - **Low-value skills waste context**: If difficulty <= 2/5 and no gaps, don't create a skill — the agent handles it fine already.
193
+ - **Context budget matters**: Every skill costs tokens on every activation. Estimate before creating.
194
+ - **Single-responsibility**: If improvement adds new domain to existing skill, split instead of updating.
195
+ - **Deferred is explicit (clustering-driven):** When clustering finds 4 candidates and proposal adopts 2, the remaining 2 MUST appear under `Deferred:` with one-line reasons — populated from Phase 2f cluster table, not silently dropped.
196
+ - **Lifecycle: prune/merge/split:** `Prune` when 0/10 or `m<0.2`, `Merge` when ≥3 shared gaps same keyword, `Split` when description needs `and` — emit one-path `What Changes` with `Deferred:` for rejected
197
+ - **Cap 8: create must pair with prune/merge at cap:** When 8 skills exist, any `create` proposal MUST also include a `prune` or `merge` in same proposal; never silently exceed cap
198
+
199
+ ## Error Handling
200
+
201
+ | Error | Action |
202
+ |-------|--------|
203
+ | No reports found | Suggest `/opsx-report` |
204
+ | Report missing assessment.md | Process report.md only |
205
+ | Invalid proposal generation | Log error, continue |
206
+ | Archive move fails | Log error, leave in place |
207
+
208
+ ## Reference Files
209
+
210
+ - `references/report-analysis.md` — How to parse reports
211
+ - `references/skill-quality.md` — Quality criteria
212
+ - `references/improvement-patterns.md` — Common improvement types
213
+ - `references/evaluation-methodology.md` — Eval framework
214
+ - `references/tool-type-detection.md` — Full detection matrix
215
+ - `references/conflict-handling.md` — Merging strategies
216
+ - `references/examples.md` — Worked examples
@@ -0,0 +1,44 @@
1
+ {
2
+ "skill": "opsx-learn",
3
+ "generated": {
4
+ "by": "process:compute-benchmark/1.0",
5
+ "at": "2026-08-23T12:42:46Z"
6
+ },
7
+ "stage": "behavioral",
8
+ "structural": {
9
+ "validate_structure": {
10
+ "pass": true,
11
+ "warnings": [
12
+ "Missing scripts/ directory",
13
+ "Missing assets/ directory"
14
+ ]
15
+ },
16
+ "validate_routing": {
17
+ "checks_total": 6,
18
+ "positive_triggers": 3,
19
+ "anti_triggers": 2,
20
+ "description_body_alignment": "10/10 description keywords found in body (100% alignment)",
21
+ "single_responsibility": true
22
+ },
23
+ "evals": {
24
+ "count": 3,
25
+ "assertions": 9,
26
+ "anti_trigger_coverage": true
27
+ }
28
+ },
29
+ "behavioral_dxm": "1×0.27",
30
+ "ship_gate": {
31
+ "criterion": "d = +1 and m >= 0.2",
32
+ "applies_to": "behavioral stage"
33
+ },
34
+ "behavioral": {
35
+ "at": "2026-09-12T09:38:54.002Z",
36
+ "evals": 4,
37
+ "assertions": 11,
38
+ "baseline": 0.5455,
39
+ "with_skill": 0.8182,
40
+ "d": 1,
41
+ "m": 0.2727,
42
+ "ship": "pass"
43
+ }
44
+ }
@@ -0,0 +1,48 @@
1
+ {
2
+ "skill_name": "opsx-learn",
3
+ "evals": [
4
+ {
5
+ "id": 1,
6
+ "prompt": "Analyze all reports in ./openspec/reports/ and generate OpenSpec proposals for improvements.",
7
+ "expected_output": "System scans ./openspec/reports/, processes each report, determines tool type (skill/prompt/combo), and generates OpenSpec proposals in ./openspec/changes/.",
8
+ "files": [],
9
+ "assertions": [
10
+ "Report directory exists at ./openspec/reports/",
11
+ "At least one proposal generated in ./openspec/changes/",
12
+ "Each proposal has proposal.md, specs/, design.md, tasks.md",
13
+ "Processed reports moved to ./openspec/reports/archives/"
14
+ ]
15
+ },
16
+ {
17
+ "id": 2,
18
+ "prompt": "Analyze the add-central-datastore report and generate a proposal.",
19
+ "expected_output": "System processes single report, determines it's a database skill, generates proposal for database-schema skill.",
20
+ "files": [],
21
+ "assertions": [
22
+ "Single report processed from ./openspec/reports/add-central-datastore/",
23
+ "Proposal generated for database-schema skill",
24
+ "Proposal contains relevant specs from report"
25
+ ]
26
+ },
27
+ {
28
+ "id": 3,
29
+ "prompt": "What reports are available for analysis?",
30
+ "expected_output": "System lists available reports in ./openspec/reports/ directory.",
31
+ "files": [],
32
+ "assertions": [
33
+ "List of available reports displayed",
34
+ "Reports in ./openspec/reports/archives/ not shown"
35
+ ]
36
+ },
37
+ {
38
+ "id": 4,
39
+ "prompt": "I want to propose a new change to add caching to our API service.",
40
+ "expected_output": "Agent does NOT activate openspec-learn. This is a new change proposal request handled by openspec-propose; openspec-learn only analyzes existing reports to plan skill improvements.",
41
+ "files": [],
42
+ "assertions": [
43
+ "Anti-trigger fires: request routes to openspec-propose, not openspec-learn",
44
+ "No report scanning or skill learning workflow executed"
45
+ ]
46
+ }
47
+ ]
48
+ }
@@ -0,0 +1,27 @@
1
+ {
2
+ "description": "Retrieval benchmark for openspec-learn clustering queries",
3
+ "version": 1,
4
+ "collection": "openspec",
5
+ "queries": [
6
+ {
7
+ "id": "cluster-baseline-exact",
8
+ "query": "submodule pointer sync",
9
+ "type": "exact",
10
+ "description": "Exact keyword match baseline for submodule pointer sync",
11
+ "expected_files": [
12
+ "specs/submodule-pointer-sync/spec.md"
13
+ ],
14
+ "expected_in_top_k": 1
15
+ },
16
+ {
17
+ "id": "cluster-paraphrase-semantic",
18
+ "query": "intent: find the spec that owns submodule pointer freshness\nlex: submodule pointer sync gitlink\nvec: which capability requires committing the updated submodule pointer",
19
+ "type": "semantic",
20
+ "description": "Paraphrased query document retrieving submodule pointer spec without exact title match",
21
+ "expected_files": [
22
+ "specs/submodule-pointer-sync/spec.md"
23
+ ],
24
+ "expected_in_top_k": 1
25
+ }
26
+ ]
27
+ }
@@ -0,0 +1,20 @@
1
+ # Conflict Handling
2
+
3
+ When multiple reports affect same tool:
4
+
5
+ 1. **Collect all improvements** from each report
6
+ 2. **Merge into single proposal** with combined requirements
7
+ 3. **Deduplicate** redundant improvements
8
+ 4. **Prioritize** by assessment difficulty ratings
9
+ 5. **Note provenance** which report contributed which improvement
10
+
11
+ ## Example
12
+
13
+ ```
14
+ Report A: "Improved csv-analyzer with better parsing"
15
+ Report B: "Fixed csv-analyzer edge cases"
16
+
17
+ System detects: Both affect csv-analyzer
18
+ System action: Merges into single proposal
19
+ Result: One proposal with combined improvements
20
+ ```
@@ -0,0 +1,126 @@
1
+ # Evaluation Methodology
2
+
3
+ How to measure quality before and after improvements.
4
+
5
+ ## Before/After Measurement
6
+
7
+ ### Step 1: Baseline Measurement
8
+
9
+ Before applying improvements:
10
+
11
+ 1. **Structural validation**:
12
+ ```bash
13
+ project/skills/create-skill/scripts/validate-structure.mjs <skill-dir>
14
+ ```
15
+ Extract: pass/fail, errors, warnings
16
+
17
+ 2. **Antipattern audit**:
18
+ ```bash
19
+ project/skills/create-skill/scripts/audit-antipatterns.mjs <skill-dir>
20
+ ```
21
+ Extract: pass/fail, violation count, violations
22
+
23
+ 3. **Content review** (manual):
24
+ - Description specificity: 0-10
25
+ - Instruction clarity: 0-10
26
+ - Progressive disclosure: 0-5
27
+ - Gotchas present: 0-5
28
+
29
+ 4. **Calculate baseline score**:
30
+ ```
31
+ Structural: 30 points (from validation)
32
+ Content: 30 points (from review)
33
+ Fragility: 15 points (from classification)
34
+ Antipatterns: 15 points (from audit)
35
+ Trigger: 10 points (from tests)
36
+ Total: 0-100
37
+ ```
38
+
39
+ ### Step 2: Apply Improvements
40
+
41
+ Apply the determined improvements (see improvement-patterns.md).
42
+
43
+ ### Step 3: Post-Improvement Measurement
44
+
45
+ After applying improvements:
46
+
47
+ 1. Re-run all checks from Step 1
48
+ 2. Calculate new score
49
+ 3. Calculate delta:
50
+ ```
51
+ delta = new_score - baseline_score
52
+ percentage = (delta / baseline_score) * 100
53
+ ```
54
+
55
+ ## Quality Delta Interpretation
56
+
57
+ | Delta | Interpretation | Action |
58
+ |-------|----------------|--------|
59
+ | > 10% | Significant improvement | Report success |
60
+ | 0-10% | Minor improvement | Report success, consider more |
61
+ | 0 | Plateau | Stop iterating |
62
+ | < 0 | Degradation | Rollback, report failure |
63
+
64
+ ## Rollback Procedure
65
+
66
+ If quality degraded:
67
+
68
+ 1. **Identify what changed**: Compare before/after file states
69
+ 2. **Revert changes**: Restore previous version
70
+ 3. **Update changelog**: Add entry noting rollback
71
+ 4. **Report failure**: Explain what was tried and why it failed
72
+
73
+ ## Iteration Logic
74
+
75
+ ```
76
+ iteration = 0
77
+ max_iterations = 3
78
+
79
+ WHILE iteration < max_iterations:
80
+ baseline = measure_quality()
81
+ apply_improvements()
82
+ post = measure_quality()
83
+ delta = post - baseline
84
+
85
+ IF delta > 0:
86
+ report_success()
87
+ BREAK
88
+ ELSE IF delta == 0:
89
+ report_plateau()
90
+ BREAK
91
+ ELSE:
92
+ rollback()
93
+ iteration++
94
+
95
+ IF iteration == max_iterations:
96
+ report_max_reached()
97
+ ```
98
+
99
+ ## Metrics to Track
100
+
101
+ | Metric | Before | After | Delta |
102
+ |--------|--------|-------|-------|
103
+ | Structural score | X | Y | Y-X |
104
+ | Antipattern count | X | Y | Y-X |
105
+ | Description length | X | Y | Y-X |
106
+ | SKILL.md lines | X | Y | Y-X |
107
+ | Reference count | X | Y | Y-X |
108
+ | Script count | X | Y | Y-X |
109
+
110
+ ## Report Template
111
+
112
+ ```markdown
113
+ ## Quality Evaluation
114
+
115
+ **Baseline score:** X/100
116
+ **Post-improvement score:** Y/100
117
+ **Delta:** +Z%
118
+
119
+ ### Changes
120
+ - Structural: X → Y
121
+ - Content: X → Y
122
+ - Antipatterns: X → Y
123
+
124
+ ### Verdict
125
+ [Success / Plateau / Failure]
126
+ ```
@@ -0,0 +1,37 @@
1
+ # Examples
2
+
3
+ ## Single Report Processing
4
+
5
+ ```bash
6
+ /opsx-learn add-central-datastore
7
+ ```
8
+
9
+ 1. System finds report at `./openspec/reports/add-central-datastore/`
10
+ 2. Analyzes report.md and assessment.md
11
+ 3. Determines: database-schema skill needs improvement
12
+ 4. Generates proposal at `./openspec/changes/add-database-schema-improvements/`
13
+ 5. Moves report to `./openspec/reports/archives/`
14
+ 6. User reviews proposal with `/opsx-explore` or implements with `/opsx-apply`
15
+
16
+ ## Batch Processing
17
+
18
+ ```bash
19
+ /opsx-learn
20
+ ```
21
+
22
+ 1. System scans `./openspec/reports/` for all reports (excluding archives)
23
+ 2. Processes each report sequentially
24
+ 3. Merges improvements for same tool
25
+ 4. Generates separate proposals for different tools
26
+ 5. Moves all processed reports to archives
27
+
28
+ ## Conflict Resolution
29
+
30
+ ```
31
+ Report A: "Improved csv-analyzer with better parsing"
32
+ Report B: "Fixed csv-analyzer edge cases"
33
+
34
+ System detects: Both affect csv-analyzer
35
+ System action: Merges into single proposal
36
+ Result: One proposal with combined improvements
37
+ ```