@meyverick/agentic 5.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +234 -0
- package/CHANGELOG.md +236 -0
- package/README.md +50 -0
- package/install.ts +349 -0
- package/package.json +37 -0
- package/scripts/check-deps.mjs +587 -0
- package/scripts/git-dl.mjs +100 -0
- package/skills/check/SKILL.md +108 -0
- package/skills/check/evals/benchmark.json +40 -0
- package/skills/check/evals/evals.json +38 -0
- package/skills/check/references/diagnostic-matrix.md +170 -0
- package/skills/check/references/script-anatomy.md +154 -0
- package/skills/create-skill/SKILL.md +291 -0
- package/skills/create-skill/assets/templates/SKILL.md.template +118 -0
- package/skills/create-skill/assets/templates/evals.json.template +36 -0
- package/skills/create-skill/assets/templates/grading.json.template +26 -0
- package/skills/create-skill/evals/benchmark.json +41 -0
- package/skills/create-skill/evals/evals.json +50 -0
- package/skills/create-skill/evals/grading-template.json +36 -0
- package/skills/create-skill/evals/near-misses.json +35 -0
- package/skills/create-skill/evals/trigger-queries.json +80 -0
- package/skills/create-skill/references/antipatterns.md +123 -0
- package/skills/create-skill/references/component-decomposition.md +130 -0
- package/skills/create-skill/references/content-quality-criteria.md +61 -0
- package/skills/create-skill/references/description-optimization.md +90 -0
- package/skills/create-skill/references/eval-methodology.md +100 -0
- package/skills/create-skill/references/fragility-matching.md +88 -0
- package/skills/create-skill/references/gotchas-patterns.md +80 -0
- package/skills/create-skill/references/specification.md +77 -0
- package/skills/create-skill/scripts/audit-antipatterns.mjs +164 -0
- package/skills/create-skill/scripts/compute-benchmark.mjs +111 -0
- package/skills/create-skill/scripts/run-cold-eval.mjs +118 -0
- package/skills/create-skill/scripts/scaffold-skill.mjs +86 -0
- package/skills/create-skill/scripts/validate-routing.mjs +137 -0
- package/skills/create-skill/scripts/validate-structure.mjs +223 -0
- package/skills/design-craft/SKILL.md +134 -0
- package/skills/design-craft/evals/benchmark.json +41 -0
- package/skills/design-craft/evals/evals.json +81 -0
- package/skills/design-craft/references/anti-slop-patterns.md +49 -0
- package/skills/design-craft/references/art-direction.md +89 -0
- package/skills/design-craft/references/design-engineering.md +122 -0
- package/skills/design-craft/references/motion-craft.md +124 -0
- package/skills/design-craft/references/process.md +47 -0
- package/skills/design-craft/references/review-checklist.md +121 -0
- package/skills/guardrails/SKILL.md +118 -0
- package/skills/guardrails/evals/benchmark.json +40 -0
- package/skills/guardrails/evals/evals.json +49 -0
- package/skills/guardrails/references/guardrails-patterns.md +43 -0
- package/skills/okf-docs/SKILL.md +79 -0
- package/skills/okf-docs/evals/benchmark.json +21 -0
- package/skills/okf-docs/evals/evals.json +37 -0
- package/skills/okf-docs/references/okf-spec.md +56 -0
- package/skills/okf-docs/scripts/validate-frontmatter.mjs +130 -0
- package/skills/openspec-harden/SKILL.md +138 -0
- package/skills/openspec-harden/evals/benchmark.json +40 -0
- package/skills/openspec-harden/evals/evals.json +38 -0
- package/skills/openspec-learn/SKILL.md +216 -0
- package/skills/openspec-learn/evals/benchmark.json +44 -0
- package/skills/openspec-learn/evals/evals.json +48 -0
- package/skills/openspec-learn/evals/retrieval-bench.json +27 -0
- package/skills/openspec-learn/references/conflict-handling.md +20 -0
- package/skills/openspec-learn/references/evaluation-methodology.md +126 -0
- package/skills/openspec-learn/references/examples.md +37 -0
- package/skills/openspec-learn/references/improvement-patterns.md +155 -0
- package/skills/openspec-learn/references/report-analysis.md +104 -0
- package/skills/openspec-learn/references/skill-quality.md +103 -0
- package/skills/openspec-learn/references/tool-type-detection.md +30 -0
- package/skills/openspec-report/SKILL.md +104 -0
- package/skills/openspec-report/assets/templates/assessment.md.template +84 -0
- package/skills/openspec-report/assets/templates/report.md.template +92 -0
- package/skills/openspec-report/evals/benchmark.json +44 -0
- package/skills/openspec-report/evals/evals.json +46 -0
- package/skills/qmd-research/SKILL.md +89 -0
- package/skills/qmd-research/evals/benchmark.json +40 -0
- package/skills/qmd-research/evals/evals.json +38 -0
- package/skills/qmd-research/references/index-management.md +69 -0
- package/skills/qmd-research/references/query-craft.md +82 -0
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill_name": "openspec-harden",
|
|
3
|
+
"evals": [
|
|
4
|
+
{
|
|
5
|
+
"id": 1,
|
|
6
|
+
"prompt": "Harden the proposal, specs, and tasks for the auth change with concrete file paths and test commands.",
|
|
7
|
+
"expected_output": "Agent loads openspec-harden skill and inspects proposal.md, design.md, specs/, and tasks.md to enrich them with grounded codebase context and exact paths.",
|
|
8
|
+
"files": [],
|
|
9
|
+
"assertions": [
|
|
10
|
+
"Skill activates on hardening request",
|
|
11
|
+
"Artifacts enriched with concrete paths",
|
|
12
|
+
"Source code is not edited"
|
|
13
|
+
]
|
|
14
|
+
},
|
|
15
|
+
{
|
|
16
|
+
"id": 2,
|
|
17
|
+
"prompt": "Implement the tasks in tasks.md for the auth change and modify the codebase.",
|
|
18
|
+
"expected_output": "Agent does NOT activate openspec-harden. Implementation and code-writing are handled by openspec-apply-change; openspec-harden only enriches planning artifacts before apply.",
|
|
19
|
+
"files": [],
|
|
20
|
+
"assertions": [
|
|
21
|
+
"Anti-trigger fires: request routes to openspec-apply-change, not openspec-harden",
|
|
22
|
+
"No proposal hardening workflow executed",
|
|
23
|
+
"No planning artifacts modified"
|
|
24
|
+
]
|
|
25
|
+
},
|
|
26
|
+
{
|
|
27
|
+
"id": 3,
|
|
28
|
+
"prompt": "Improve change for fresh agent by resolving either/or file paths into single targets.",
|
|
29
|
+
"expected_output": "Agent enriches tasks with single concrete paths and verify steps without implementing features.",
|
|
30
|
+
"files": [],
|
|
31
|
+
"assertions": [
|
|
32
|
+
"Single concrete paths assigned",
|
|
33
|
+
"Zero code mutations made to project files",
|
|
34
|
+
"Verify steps populated"
|
|
35
|
+
]
|
|
36
|
+
}
|
|
37
|
+
]
|
|
38
|
+
}
|
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: openspec-learn
|
|
3
|
+
description: >
|
|
4
|
+
Analyze reports from `openspec-report` to generate OpenSpec proposals for
|
|
5
|
+
skill improvements. Use when analyzing reports to plan skill creation,
|
|
6
|
+
improvements, or lifecycle actions. Do NOT use when proposing a new change
|
|
7
|
+
from scratch (use openspec-propose), generating reports from archived changes,
|
|
8
|
+
or implementing proposals.
|
|
9
|
+
allowed-tools: Bash(openspec:*), Bash(node:*), Bash(mkdir:*), Bash(ls:*)
|
|
10
|
+
license: MIT
|
|
11
|
+
compatibility: Requires openspec CLI and bun.
|
|
12
|
+
metadata:
|
|
13
|
+
author: agentic
|
|
14
|
+
version: "1.3.1"
|
|
15
|
+
positive_triggers:
|
|
16
|
+
- "analyze reports and improve skills"
|
|
17
|
+
- "generate proposals from report insights"
|
|
18
|
+
- "process reports and create improvement plans"
|
|
19
|
+
anti_triggers:
|
|
20
|
+
- "propose a new change from scratch"
|
|
21
|
+
- "generate a report from an archived change"
|
|
22
|
+
- "implement a proposal or apply changes"
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
# Openspec Learn
|
|
26
|
+
|
|
27
|
+
Analyze reports and generate OpenSpec proposals for improvements.
|
|
28
|
+
|
|
29
|
+
## Quick Start
|
|
30
|
+
|
|
31
|
+
- **With argument**: Process single report: `./openspec/reports/<name>/`
|
|
32
|
+
- **Without argument**: Process ALL reports: `./openspec/reports/*/` (excluding `archives/`)
|
|
33
|
+
- **With store**: Forward selected `--store <id>` on applicable commands (`new change`, `status`, `instructions`, `list`, `show`, `validate`, `archive`, `doctor`, `context`, `schemas`, `view`); sticky for workflow; nearest-root if unselected
|
|
34
|
+
|
|
35
|
+
## Workflow
|
|
36
|
+
|
|
37
|
+
### Phase 1: Report Scanning
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
ls ./openspec/reports/ | grep -v archives
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
### Phase 2: Report Analysis
|
|
44
|
+
|
|
45
|
+
**From report.md:** OKF frontmatter, problem statement, approach, specs, implementation, validation, trade-offs, follow-ups.
|
|
46
|
+
|
|
47
|
+
**From assessment.md:** knowledge gaps, difficulty ratings, tool improvements, what could have helped.
|
|
48
|
+
|
|
49
|
+
### Phase 2a: Semantic Collision Detection
|
|
50
|
+
|
|
51
|
+
Scan existing skills for description overlap with proposed tool:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
grep -r "description:" .agents/skills/*/SKILL.md ./project/skills/*/SKILL.md 2>/dev/null
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Compare proposed tool's domain keywords against existing descriptions. If overlap detected → flag collision in proposal, suggest updating existing skill instead of creating new one.
|
|
58
|
+
|
|
59
|
+
### Phase 2b: Trigger Metadata Extraction
|
|
60
|
+
|
|
61
|
+
Extract trigger candidates from report assessment:
|
|
62
|
+
- What queries/situations led to this change? → candidate positive_triggers
|
|
63
|
+
- What was confusing or out-of-scope? → candidate anti_triggers
|
|
64
|
+
|
|
65
|
+
Include as "Suggested Triggers" section in proposal.
|
|
66
|
+
|
|
67
|
+
### Phase 2c: Ownership Pre-Check + Single-Responsibility Pre-Check
|
|
68
|
+
|
|
69
|
+
**Ownership pre-check (MANDATORY, runs first).** Classify every candidate target skill by ownership using this precedence chain — recorded facts before conventions, conventions before residual judgment:
|
|
70
|
+
|
|
71
|
+
1. Inside the agentic repo itself → `project/skills/*` is owned (editable only by the project owner and the owner's trusted assistant session)
|
|
72
|
+
2. Listed in `.agents/skills/.agentic-manifest.json` → agentic-distributed → **external** (read-only for consumers)
|
|
73
|
+
3. Located in `~/.pi/agent/skills/` (not created by this project) → **external**
|
|
74
|
+
4. Explicit verdict in `.agents/skills/.ownership.json` (`{"owned": [...], "external": [...]}`) → as declared
|
|
75
|
+
5. Named `openspec-*` and authored by OpenSpec (upstream path `.agents/skills/openspec-*`, frontmatter `author: openspec`) → **external** (public namespace claim; does NOT apply to agentic-authored `project/skills/openspec-{learn,report,harden}`, which remain owned at source per item 1 and read-only once installed elsewhere)
|
|
76
|
+
6. Any other skill in the project's `.agents/skills/` → project-created → editable
|
|
77
|
+
7. Unresolved after all checks → unknown = **external**, ask the user
|
|
78
|
+
|
|
79
|
+
Rules:
|
|
80
|
+
- Proposals MUST NOT target external skills. Never edit installed/upstream files under `.agents/skills/*` in place — consumer agents treat installed skills as read-only; the next atomic-replace install silently wipes in-place edits. Reroute improvements upstream or to project-local placements.
|
|
81
|
+
- Domain-specific knowledge belongs in project-local homes (wiki, checklists, project-created skills). If no local home exists, ask the user where to place it.
|
|
82
|
+
- Genuinely generic improvements to external skills: record as an upstream recommendation for the user (issue/PR), never edit directly.
|
|
83
|
+
- When you ask the user about an unknown skill's ownership, record the verdict in `.ownership.json` so each question is asked once.
|
|
84
|
+
|
|
85
|
+
**Single-responsibility pre-check** (only for targets that passed the ownership check):
|
|
86
|
+
|
|
87
|
+
For proposed updates to existing skills:
|
|
88
|
+
- Does the improvement align with the skill's atomic intent?
|
|
89
|
+
- Or does it add a new operational domain?
|
|
90
|
+
|
|
91
|
+
If scope expansion detected → propose splitting into separate skill instead of updating.
|
|
92
|
+
|
|
93
|
+
### Phase 2d: Value Justification
|
|
94
|
+
|
|
95
|
+
Estimate whether creating/updating a skill improves outcomes by >= 20%:
|
|
96
|
+
- Assessment difficulty >= 3/5 OR knowledge gaps identified → justified
|
|
97
|
+
- All difficulty <= 2/5 AND no gaps → flag as low value in proposal
|
|
98
|
+
- **Frequency × cost heuristic (new):** Read `Re-use Score` (high=3, medium=2, low=1) and `Time Cost` from report/assessment; prioritize `frequency × cost` — a `high` re-use gap that cost 60m and recurs ≥2 times outranks a singleton `low` gap even if its difficulty was 5/5
|
|
99
|
+
|
|
100
|
+
### Phase 2e: Context Budget Impact
|
|
101
|
+
|
|
102
|
+
Estimate Tier 1 + Tier 2 token cost of proposed skill based on similar skills. Warn if >2000 tokens in proposal.
|
|
103
|
+
|
|
104
|
+
### Phase 2f: Recurring Gap Clustering (compound learning)
|
|
105
|
+
|
|
106
|
+
Before proposing, cluster gaps and gotchas across ALL reports in `openspec/reports/*` (excluding `archives/`).
|
|
107
|
+
|
|
108
|
+
**Primary Clusterer (Hybrid Semantic via QMD):**
|
|
109
|
+
Query the project-local index (never a global or shared index) scoped to `openspec` and `references` collections:
|
|
110
|
+
```bash
|
|
111
|
+
qmd query $'intent: cluster recurring knowledge gaps, skills gaps, and gotchas across reports\nlex: knowledge gaps skills gaps gotchas mental model shift\nvec: recurring difficulties, surprises, and lessons learned from past changes' --json -n 20 -c openspec -c references
|
|
112
|
+
```
|
|
113
|
+
Retrieve evidence for clustered items via `qmd multi-get "<docids>" --json`. Clusters MUST carry `qmd://` document IDs as evidence in the proposal. Semantic matching groups paraphrased descriptions of the same underlying obstacle even when distinct keywords were used.
|
|
114
|
+
|
|
115
|
+
**Loud Fallback (Keyword Grep):**
|
|
116
|
+
If the QMD daemon is unreachable or the index is unhealthy (`qmd status` fails):
|
|
117
|
+
```bash
|
|
118
|
+
# Loud fallback: note "qmd unavailable, grep fallback" in proposal
|
|
119
|
+
for f in openspec/reports/*/report.md openspec/reports/*/assessment.md; do [ -f "$f" ] && echo "== $f ==" && grep -E "Knowledge gaps|Skills gaps|Concrete Gotcha|Before \(old|After \(new|Re-use Score|Time Cost" "$f" | head -n 20; done
|
|
120
|
+
```
|
|
121
|
+
Note `qmd unavailable, grep fallback` explicitly in the proposal's Analysis section.
|
|
122
|
+
|
|
123
|
+
Group recurring findings by topic/keyword and count occurrences. Prioritize a cluster that recurs in ≥2 reports over a singleton, even if the singleton's difficulty was higher. Record the cluster table (topic/keyword → count → representative gap → `qmd://` evidence) in the proposal's Analysis section; this drives `Deferred:` decisions in Phase 5.
|
|
124
|
+
|
|
125
|
+
**Dashboard generation (human curation view):** After clustering, write/overwrite `openspec/reports/dashboard.md` (never archived, never auto-loaded at agent startup):
|
|
126
|
+
```bash
|
|
127
|
+
# Header: generated at $(date -u +"%Y-%m-%dT%H:%M:%SZ") from N reports
|
|
128
|
+
# Table: keyword | count | avg Time Cost | avg Re-use (high=3/med=2/low=1) | owning skill | m sorted by frequency×cost desc
|
|
129
|
+
# Source: grep Re-use Score + Time Cost + keyword + owning skill from clustered reports + benchmark.json m
|
|
130
|
+
# If no reports, write: No reports yet — run /opsx-report after a hard task
|
|
131
|
+
```
|
|
132
|
+
Header `generated at <timestamp> from N reports`; table sorted `frequency × cost` desc (count × Re-use weight × avg Time Cost). Overwrite on each learn run; human reads it to curate manual reports.
|
|
133
|
+
|
|
134
|
+
**Lifecycle check (prune/merge/split):** After clustering, also evaluate:
|
|
135
|
+
- **Prune:** If a skill's usage count is 0 in last 10 reports OR its `benchmark.json` `behavioral.m < 0.2`, mark `Prune: <skill> — 0/10 or low m`
|
|
136
|
+
- **Merge:** If ≥3 shared gaps under same keyword across 2 skills, mark `Merge: <target> ← <a> + <b> — shared keyword`
|
|
137
|
+
- **Split:** If a skill's description would need `and`, mark `Split: <skill> → <a> + <b> — single-responsibility`
|
|
138
|
+
Each emits one-path `What Changes: Remove project/skills/<skill>/` or `Merge: ...` with `Deferred:` for rejected lifecycle candidates.
|
|
139
|
+
|
|
140
|
+
### Phase 3: Tool Type Determination
|
|
141
|
+
|
|
142
|
+
See [references/tool-type-detection.md](references/tool-type-detection.md) for full detection matrix.
|
|
143
|
+
|
|
144
|
+
### Phase 4: Conflict Resolution
|
|
145
|
+
|
|
146
|
+
See [references/conflict-handling.md](references/conflict-handling.md) for merging strategies.
|
|
147
|
+
|
|
148
|
+
### Phase 5: OpenSpec Proposal Generation
|
|
149
|
+
|
|
150
|
+
Generate proposal including:
|
|
151
|
+
- What to build + why (from report + assessment)
|
|
152
|
+
- Suggested Triggers section (from Phase 2b) — MUST use the exact frontmatter field names `positive_triggers` and `anti_triggers` as list headers, with each value formatted for verbatim transfer into a skill's frontmatter (no prose labels like "Positive:" or "Negative:")
|
|
153
|
+
- **Contrast and Anti-example hints (from report's Mental Model Shift / Concrete Gotcha):** Carry `Contrast: Before X → After Y` (1–2 lines) and `Anti-example: Do NOT: <before code> → Do: <after code>` verbatim into What Changes so `create-skill` can generate contrast tables and anti-examples as first-class content
|
|
154
|
+
- **Layer-aware decision note:** State `Layer: 1 (always) vs 2 (on-demand skill) vs 3 (gate)` — for `Re-use Score: high` + `Time Cost >30m` + recurring cluster, suggest `Layer 1/3` promotion (e.g., `guardrails` skill or `AGENTS.md` Must-read pointer); otherwise `Layer 2` new/updated skill; no gate is implemented in this change, only the hint
|
|
155
|
+
- **Lifecycle What Changes (when applicable):** Emit `What Changes: Remove project/skills/<skill>/` for prune, `What Changes: Merge project/skills/<target>/ ← <a> + <b>` for merge, or `Split` with two one-path creates; each with `Impact: evals + manifest updated` and `Deferred:` for rejected lifecycle candidates; cap 8 enforced here (see Gotchas)
|
|
156
|
+
- Value Justification section (from Phase 2d, now including frequency × cost)
|
|
157
|
+
- Collision warnings (from Phase 2a)
|
|
158
|
+
- Context budget impact (from Phase 2e)
|
|
159
|
+
- Clustering summary (from Phase 2f) — keyword → count → representative gap
|
|
160
|
+
- **Dashboard citation (MANDATORY when lifecycle or create):** Each `create`/`prune`/`merge`/`split` in `What Changes` MUST cite `Source: dashboard.md#<keyword> — <count>× <Re-use> <time> (avg)` with verification path `openspec/reports/dashboard.md` — e.g., `Source: dashboard.md#lifetime — 4× high 45m avg`
|
|
161
|
+
- Evals impact statement (MANDATORY when the proposal modifies an existing skill): state whether `evals/evals.json` changes; if triggers are added or altered, include at least one matching eval entry in Impact. No trigger changes → state that existing evals remain valid.
|
|
162
|
+
- Deferred signals line (when the source report/assessment contains more improvement candidates than the proposal adopts): name each unadopted candidate with a one-line reason, so deferral is explicit rather than silent
|
|
163
|
+
|
|
164
|
+
**One-path rule**: every What Changes and Impact item names exactly ONE concrete target file path. Either/or targets ("X or Y") are prohibited — resolve the choice during design, before tasks are written. Task verify clauses must reference the same single path.
|
|
165
|
+
|
|
166
|
+
**OpenSpec CLI Contract & Store Forwarding:**
|
|
167
|
+
- Scaffold new changes with `openspec new change "<name>"` (forwarding `--store <id>` if selected). Never create change directories by hand.
|
|
168
|
+
- Forward selected `--store <id>` on applicable commands (`new change`, `status`, `instructions`, `list`, `show`, `validate`, `archive`, `doctor`, `context`, `schemas`, `view`); keep sticky; nearest-root if unselected. Other commands run unflagged.
|
|
169
|
+
- Failure envelope: Parse stdout as single JSON payload; stderr carries prose/spinners/store banner (never parse stderr as JSON). On exit 1, parse `status: [diagnostic]` array (`severity`, `code`, `message`, `fix`). Exit 130 = prompt cancelled.
|
|
170
|
+
- Payload casing: Workflow payloads use `camelCase`; store payloads use `snake_case` (`root.store_id` always `snake_case`).
|
|
171
|
+
- Archive delegation: Follow-on change archiving delegates to `openspec archive --json` (`archivedAs`, `specsUpdated`, `totals`, `warnings`). Never hand `mv` change directories or hand-merge delta specs into main specs.
|
|
172
|
+
|
|
173
|
+
The proposal instructs the AI agent to invoke create-skill during `/opsx-apply`.
|
|
174
|
+
|
|
175
|
+
### Phase 6: Archive Processed Reports
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
mkdir -p ./openspec/reports/archives
|
|
179
|
+
mv ./openspec/reports/<name> ./openspec/reports/archives/
|
|
180
|
+
# dashboard.md never moves — it stays at openspec/reports/dashboard.md (human curation view, not a report)
|
|
181
|
+
```
|
|
182
|
+
Note: Only processed reports are moved via `mv`. OpenSpec changes MUST be archived via `openspec archive --json`, never by hand `mv`.
|
|
183
|
+
|
|
184
|
+
### Phase 7: Display Summary
|
|
185
|
+
|
|
186
|
+
Report count, proposal count, archive count, next steps.
|
|
187
|
+
|
|
188
|
+
## Gotchas
|
|
189
|
+
|
|
190
|
+
- **Collision detection prevents dilution**: Two skills with similar descriptions reduce routing confidence for both.
|
|
191
|
+
- **Trigger metadata saves Discovery time**: Pre-filling triggers from assessment data gives create-skill a head start.
|
|
192
|
+
- **Low-value skills waste context**: If difficulty <= 2/5 and no gaps, don't create a skill — the agent handles it fine already.
|
|
193
|
+
- **Context budget matters**: Every skill costs tokens on every activation. Estimate before creating.
|
|
194
|
+
- **Single-responsibility**: If improvement adds new domain to existing skill, split instead of updating.
|
|
195
|
+
- **Deferred is explicit (clustering-driven):** When clustering finds 4 candidates and proposal adopts 2, the remaining 2 MUST appear under `Deferred:` with one-line reasons — populated from Phase 2f cluster table, not silently dropped.
|
|
196
|
+
- **Lifecycle: prune/merge/split:** `Prune` when 0/10 or `m<0.2`, `Merge` when ≥3 shared gaps same keyword, `Split` when description needs `and` — emit one-path `What Changes` with `Deferred:` for rejected
|
|
197
|
+
- **Cap 8: create must pair with prune/merge at cap:** When 8 skills exist, any `create` proposal MUST also include a `prune` or `merge` in same proposal; never silently exceed cap
|
|
198
|
+
|
|
199
|
+
## Error Handling
|
|
200
|
+
|
|
201
|
+
| Error | Action |
|
|
202
|
+
|-------|--------|
|
|
203
|
+
| No reports found | Suggest `/opsx-report` |
|
|
204
|
+
| Report missing assessment.md | Process report.md only |
|
|
205
|
+
| Invalid proposal generation | Log error, continue |
|
|
206
|
+
| Archive move fails | Log error, leave in place |
|
|
207
|
+
|
|
208
|
+
## Reference Files
|
|
209
|
+
|
|
210
|
+
- `references/report-analysis.md` — How to parse reports
|
|
211
|
+
- `references/skill-quality.md` — Quality criteria
|
|
212
|
+
- `references/improvement-patterns.md` — Common improvement types
|
|
213
|
+
- `references/evaluation-methodology.md` — Eval framework
|
|
214
|
+
- `references/tool-type-detection.md` — Full detection matrix
|
|
215
|
+
- `references/conflict-handling.md` — Merging strategies
|
|
216
|
+
- `references/examples.md` — Worked examples
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill": "opsx-learn",
|
|
3
|
+
"generated": {
|
|
4
|
+
"by": "process:compute-benchmark/1.0",
|
|
5
|
+
"at": "2026-08-23T12:42:46Z"
|
|
6
|
+
},
|
|
7
|
+
"stage": "behavioral",
|
|
8
|
+
"structural": {
|
|
9
|
+
"validate_structure": {
|
|
10
|
+
"pass": true,
|
|
11
|
+
"warnings": [
|
|
12
|
+
"Missing scripts/ directory",
|
|
13
|
+
"Missing assets/ directory"
|
|
14
|
+
]
|
|
15
|
+
},
|
|
16
|
+
"validate_routing": {
|
|
17
|
+
"checks_total": 6,
|
|
18
|
+
"positive_triggers": 3,
|
|
19
|
+
"anti_triggers": 2,
|
|
20
|
+
"description_body_alignment": "10/10 description keywords found in body (100% alignment)",
|
|
21
|
+
"single_responsibility": true
|
|
22
|
+
},
|
|
23
|
+
"evals": {
|
|
24
|
+
"count": 3,
|
|
25
|
+
"assertions": 9,
|
|
26
|
+
"anti_trigger_coverage": true
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
"behavioral_dxm": "1×0.27",
|
|
30
|
+
"ship_gate": {
|
|
31
|
+
"criterion": "d = +1 and m >= 0.2",
|
|
32
|
+
"applies_to": "behavioral stage"
|
|
33
|
+
},
|
|
34
|
+
"behavioral": {
|
|
35
|
+
"at": "2026-09-12T09:38:54.002Z",
|
|
36
|
+
"evals": 4,
|
|
37
|
+
"assertions": 11,
|
|
38
|
+
"baseline": 0.5455,
|
|
39
|
+
"with_skill": 0.8182,
|
|
40
|
+
"d": 1,
|
|
41
|
+
"m": 0.2727,
|
|
42
|
+
"ship": "pass"
|
|
43
|
+
}
|
|
44
|
+
}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill_name": "opsx-learn",
|
|
3
|
+
"evals": [
|
|
4
|
+
{
|
|
5
|
+
"id": 1,
|
|
6
|
+
"prompt": "Analyze all reports in ./openspec/reports/ and generate OpenSpec proposals for improvements.",
|
|
7
|
+
"expected_output": "System scans ./openspec/reports/, processes each report, determines tool type (skill/prompt/combo), and generates OpenSpec proposals in ./openspec/changes/.",
|
|
8
|
+
"files": [],
|
|
9
|
+
"assertions": [
|
|
10
|
+
"Report directory exists at ./openspec/reports/",
|
|
11
|
+
"At least one proposal generated in ./openspec/changes/",
|
|
12
|
+
"Each proposal has proposal.md, specs/, design.md, tasks.md",
|
|
13
|
+
"Processed reports moved to ./openspec/reports/archives/"
|
|
14
|
+
]
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
"id": 2,
|
|
18
|
+
"prompt": "Analyze the add-central-datastore report and generate a proposal.",
|
|
19
|
+
"expected_output": "System processes single report, determines it's a database skill, generates proposal for database-schema skill.",
|
|
20
|
+
"files": [],
|
|
21
|
+
"assertions": [
|
|
22
|
+
"Single report processed from ./openspec/reports/add-central-datastore/",
|
|
23
|
+
"Proposal generated for database-schema skill",
|
|
24
|
+
"Proposal contains relevant specs from report"
|
|
25
|
+
]
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
"id": 3,
|
|
29
|
+
"prompt": "What reports are available for analysis?",
|
|
30
|
+
"expected_output": "System lists available reports in ./openspec/reports/ directory.",
|
|
31
|
+
"files": [],
|
|
32
|
+
"assertions": [
|
|
33
|
+
"List of available reports displayed",
|
|
34
|
+
"Reports in ./openspec/reports/archives/ not shown"
|
|
35
|
+
]
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"id": 4,
|
|
39
|
+
"prompt": "I want to propose a new change to add caching to our API service.",
|
|
40
|
+
"expected_output": "Agent does NOT activate openspec-learn. This is a new change proposal request handled by openspec-propose; openspec-learn only analyzes existing reports to plan skill improvements.",
|
|
41
|
+
"files": [],
|
|
42
|
+
"assertions": [
|
|
43
|
+
"Anti-trigger fires: request routes to openspec-propose, not openspec-learn",
|
|
44
|
+
"No report scanning or skill learning workflow executed"
|
|
45
|
+
]
|
|
46
|
+
}
|
|
47
|
+
]
|
|
48
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
{
|
|
2
|
+
"description": "Retrieval benchmark for openspec-learn clustering queries",
|
|
3
|
+
"version": 1,
|
|
4
|
+
"collection": "openspec",
|
|
5
|
+
"queries": [
|
|
6
|
+
{
|
|
7
|
+
"id": "cluster-baseline-exact",
|
|
8
|
+
"query": "submodule pointer sync",
|
|
9
|
+
"type": "exact",
|
|
10
|
+
"description": "Exact keyword match baseline for submodule pointer sync",
|
|
11
|
+
"expected_files": [
|
|
12
|
+
"specs/submodule-pointer-sync/spec.md"
|
|
13
|
+
],
|
|
14
|
+
"expected_in_top_k": 1
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
"id": "cluster-paraphrase-semantic",
|
|
18
|
+
"query": "intent: find the spec that owns submodule pointer freshness\nlex: submodule pointer sync gitlink\nvec: which capability requires committing the updated submodule pointer",
|
|
19
|
+
"type": "semantic",
|
|
20
|
+
"description": "Paraphrased query document retrieving submodule pointer spec without exact title match",
|
|
21
|
+
"expected_files": [
|
|
22
|
+
"specs/submodule-pointer-sync/spec.md"
|
|
23
|
+
],
|
|
24
|
+
"expected_in_top_k": 1
|
|
25
|
+
}
|
|
26
|
+
]
|
|
27
|
+
}
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# Conflict Handling
|
|
2
|
+
|
|
3
|
+
When multiple reports affect same tool:
|
|
4
|
+
|
|
5
|
+
1. **Collect all improvements** from each report
|
|
6
|
+
2. **Merge into single proposal** with combined requirements
|
|
7
|
+
3. **Deduplicate** redundant improvements
|
|
8
|
+
4. **Prioritize** by assessment difficulty ratings
|
|
9
|
+
5. **Note provenance** which report contributed which improvement
|
|
10
|
+
|
|
11
|
+
## Example
|
|
12
|
+
|
|
13
|
+
```
|
|
14
|
+
Report A: "Improved csv-analyzer with better parsing"
|
|
15
|
+
Report B: "Fixed csv-analyzer edge cases"
|
|
16
|
+
|
|
17
|
+
System detects: Both affect csv-analyzer
|
|
18
|
+
System action: Merges into single proposal
|
|
19
|
+
Result: One proposal with combined improvements
|
|
20
|
+
```
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
# Evaluation Methodology
|
|
2
|
+
|
|
3
|
+
How to measure quality before and after improvements.
|
|
4
|
+
|
|
5
|
+
## Before/After Measurement
|
|
6
|
+
|
|
7
|
+
### Step 1: Baseline Measurement
|
|
8
|
+
|
|
9
|
+
Before applying improvements:
|
|
10
|
+
|
|
11
|
+
1. **Structural validation**:
|
|
12
|
+
```bash
|
|
13
|
+
project/skills/create-skill/scripts/validate-structure.mjs <skill-dir>
|
|
14
|
+
```
|
|
15
|
+
Extract: pass/fail, errors, warnings
|
|
16
|
+
|
|
17
|
+
2. **Antipattern audit**:
|
|
18
|
+
```bash
|
|
19
|
+
project/skills/create-skill/scripts/audit-antipatterns.mjs <skill-dir>
|
|
20
|
+
```
|
|
21
|
+
Extract: pass/fail, violation count, violations
|
|
22
|
+
|
|
23
|
+
3. **Content review** (manual):
|
|
24
|
+
- Description specificity: 0-10
|
|
25
|
+
- Instruction clarity: 0-10
|
|
26
|
+
- Progressive disclosure: 0-5
|
|
27
|
+
- Gotchas present: 0-5
|
|
28
|
+
|
|
29
|
+
4. **Calculate baseline score**:
|
|
30
|
+
```
|
|
31
|
+
Structural: 30 points (from validation)
|
|
32
|
+
Content: 30 points (from review)
|
|
33
|
+
Fragility: 15 points (from classification)
|
|
34
|
+
Antipatterns: 15 points (from audit)
|
|
35
|
+
Trigger: 10 points (from tests)
|
|
36
|
+
Total: 0-100
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
### Step 2: Apply Improvements
|
|
40
|
+
|
|
41
|
+
Apply the determined improvements (see improvement-patterns.md).
|
|
42
|
+
|
|
43
|
+
### Step 3: Post-Improvement Measurement
|
|
44
|
+
|
|
45
|
+
After applying improvements:
|
|
46
|
+
|
|
47
|
+
1. Re-run all checks from Step 1
|
|
48
|
+
2. Calculate new score
|
|
49
|
+
3. Calculate delta:
|
|
50
|
+
```
|
|
51
|
+
delta = new_score - baseline_score
|
|
52
|
+
percentage = (delta / baseline_score) * 100
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## Quality Delta Interpretation
|
|
56
|
+
|
|
57
|
+
| Delta | Interpretation | Action |
|
|
58
|
+
|-------|----------------|--------|
|
|
59
|
+
| > 10% | Significant improvement | Report success |
|
|
60
|
+
| 0-10% | Minor improvement | Report success, consider more |
|
|
61
|
+
| 0 | Plateau | Stop iterating |
|
|
62
|
+
| < 0 | Degradation | Rollback, report failure |
|
|
63
|
+
|
|
64
|
+
## Rollback Procedure
|
|
65
|
+
|
|
66
|
+
If quality degraded:
|
|
67
|
+
|
|
68
|
+
1. **Identify what changed**: Compare before/after file states
|
|
69
|
+
2. **Revert changes**: Restore previous version
|
|
70
|
+
3. **Update changelog**: Add entry noting rollback
|
|
71
|
+
4. **Report failure**: Explain what was tried and why it failed
|
|
72
|
+
|
|
73
|
+
## Iteration Logic
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
iteration = 0
|
|
77
|
+
max_iterations = 3
|
|
78
|
+
|
|
79
|
+
WHILE iteration < max_iterations:
|
|
80
|
+
baseline = measure_quality()
|
|
81
|
+
apply_improvements()
|
|
82
|
+
post = measure_quality()
|
|
83
|
+
delta = post - baseline
|
|
84
|
+
|
|
85
|
+
IF delta > 0:
|
|
86
|
+
report_success()
|
|
87
|
+
BREAK
|
|
88
|
+
ELSE IF delta == 0:
|
|
89
|
+
report_plateau()
|
|
90
|
+
BREAK
|
|
91
|
+
ELSE:
|
|
92
|
+
rollback()
|
|
93
|
+
iteration++
|
|
94
|
+
|
|
95
|
+
IF iteration == max_iterations:
|
|
96
|
+
report_max_reached()
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
## Metrics to Track
|
|
100
|
+
|
|
101
|
+
| Metric | Before | After | Delta |
|
|
102
|
+
|--------|--------|-------|-------|
|
|
103
|
+
| Structural score | X | Y | Y-X |
|
|
104
|
+
| Antipattern count | X | Y | Y-X |
|
|
105
|
+
| Description length | X | Y | Y-X |
|
|
106
|
+
| SKILL.md lines | X | Y | Y-X |
|
|
107
|
+
| Reference count | X | Y | Y-X |
|
|
108
|
+
| Script count | X | Y | Y-X |
|
|
109
|
+
|
|
110
|
+
## Report Template
|
|
111
|
+
|
|
112
|
+
```markdown
|
|
113
|
+
## Quality Evaluation
|
|
114
|
+
|
|
115
|
+
**Baseline score:** X/100
|
|
116
|
+
**Post-improvement score:** Y/100
|
|
117
|
+
**Delta:** +Z%
|
|
118
|
+
|
|
119
|
+
### Changes
|
|
120
|
+
- Structural: X → Y
|
|
121
|
+
- Content: X → Y
|
|
122
|
+
- Antipatterns: X → Y
|
|
123
|
+
|
|
124
|
+
### Verdict
|
|
125
|
+
[Success / Plateau / Failure]
|
|
126
|
+
```
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# Examples
|
|
2
|
+
|
|
3
|
+
## Single Report Processing
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
/opsx-learn add-central-datastore
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
1. System finds report at `./openspec/reports/add-central-datastore/`
|
|
10
|
+
2. Analyzes report.md and assessment.md
|
|
11
|
+
3. Determines: database-schema skill needs improvement
|
|
12
|
+
4. Generates proposal at `./openspec/changes/add-database-schema-improvements/`
|
|
13
|
+
5. Moves report to `./openspec/reports/archives/`
|
|
14
|
+
6. User reviews proposal with `/opsx-explore` or implements with `/opsx-apply`
|
|
15
|
+
|
|
16
|
+
## Batch Processing
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
/opsx-learn
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
1. System scans `./openspec/reports/` for all reports (excluding archives)
|
|
23
|
+
2. Processes each report sequentially
|
|
24
|
+
3. Merges improvements for same tool
|
|
25
|
+
4. Generates separate proposals for different tools
|
|
26
|
+
5. Moves all processed reports to archives
|
|
27
|
+
|
|
28
|
+
## Conflict Resolution
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
Report A: "Improved csv-analyzer with better parsing"
|
|
32
|
+
Report B: "Fixed csv-analyzer edge cases"
|
|
33
|
+
|
|
34
|
+
System detects: Both affect csv-analyzer
|
|
35
|
+
System action: Merges into single proposal
|
|
36
|
+
Result: One proposal with combined improvements
|
|
37
|
+
```
|