@supersuit/superskill 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/LICENSE +21 -0
- package/README.md +103 -0
- package/SPEC.md +239 -0
- package/bin/superskill.mjs +56 -0
- package/package.json +15 -0
- package/src/args.mjs +30 -0
- package/src/changed.mjs +52 -0
- package/src/collection.mjs +139 -0
- package/src/commands/approve.mjs +41 -0
- package/src/commands/collection.mjs +56 -0
- package/src/commands/common.mjs +19 -0
- package/src/commands/doctor.mjs +63 -0
- package/src/commands/fix.mjs +29 -0
- package/src/commands/import.mjs +42 -0
- package/src/commands/init.mjs +74 -0
- package/src/commands/miss.mjs +24 -0
- package/src/commands/snippet.mjs +14 -0
- package/src/context.mjs +60 -0
- package/src/doctor.mjs +65 -0
- package/src/evals.mjs +44 -0
- package/src/frontmatter.mjs +109 -0
- package/src/goldens.mjs +34 -0
- package/src/ledger.mjs +44 -0
- package/src/levels.mjs +38 -0
- package/src/misses.mjs +88 -0
- package/src/report.mjs +27 -0
- package/src/rules/define.mjs +9 -0
- package/src/rules/index.mjs +7 -0
- package/src/rules/interop.mjs +13 -0
- package/src/rules/skill.mjs +359 -0
- package/src/rules/superskill.mjs +113 -0
- package/src/rules/tested.mjs +52 -0
- package/src/run/claude.mjs +36 -0
- package/src/run/codex.mjs +32 -0
- package/src/run/fake.mjs +28 -0
- package/src/run/grade.mjs +56 -0
- package/src/run/index.mjs +130 -0
- package/src/run/workspace.mjs +25 -0
- package/src/session.mjs +34 -0
- package/src/snippet.md +16 -0
package/CHANGELOG.md
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0 (2026-09-28)
|
|
4
|
+
|
|
5
|
+
- `reference-says-when`: a link from SKILL.md to an instruction file must say when to read it. Long-skill fixes now point at step files (`steps/<step>.md`).
|
|
6
|
+
|
|
7
|
+
- Length is not a defect. The 500-line `body-lines` failure and the `body-tokens` warning are replaced by `body-size` (info), `rules-above-the-fold`, `navigable` and `no-repeated-paragraphs`, which check what goes wrong in a long skill rather than its length.
|
|
8
|
+
|
|
9
|
+
First release of the standard ([SPEC.md](SPEC.md) v0.1.0) and its checker.
|
|
10
|
+
|
|
11
|
+
- `doctor`: scores a skill, a folder of skills, or a plugin as skill / tested / superskill, with
|
|
12
|
+
the to-do list for the next level. `--level`, `--json`, `--changed`, `--base`,
|
|
13
|
+
`--baseline-json`.
|
|
14
|
+
- `doctor --run`: runs evals with and without the skill through Claude Code or Codex and writes
|
|
15
|
+
`evals/results/latest.json`. Prints an estimate and asks before spending calls.
|
|
16
|
+
- `init` (with `--from-session`), `miss`, `fix`, `approve`, `miss import --freedom-ledger`.
|
|
17
|
+
- `collection`: listing budget, cut-off descriptions, overlapping skills, the plugin line.
|
|
18
|
+
- `snippet`: agent-instructions block.
|
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 SupersuitUp
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# superskill
|
|
2
|
+
|
|
3
|
+
Score any agent skill folder as **skill**, **tested**, or **superskill**, and get the exact
|
|
4
|
+
to-do list for the next level. Works on skills for Claude Code, Codex, or any harness that
|
|
5
|
+
reads the [Agent Skills](https://agentskills.io) format. Zero dependencies, Node 20 or later.
|
|
6
|
+
|
|
7
|
+
A superskill runs on frontier intelligence, is checked against examples you approved, and is
|
|
8
|
+
fixed every time it gets something wrong. [What that means](https://supersuit.wiki/concepts/superskill);
|
|
9
|
+
[the standard](SPEC.md).
|
|
10
|
+
|
|
11
|
+
## 30 seconds
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
npx @supersuit/superskill doctor ./my-skill
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
```
|
|
18
|
+
my-skill level: skill
|
|
19
|
+
/path/to/my-skill
|
|
20
|
+
to reach tested:
|
|
21
|
+
- evals-present: 0 eval cases (need 3)
|
|
22
|
+
fix: Add real requests to evals/evals.json (`superskill init` writes an example).
|
|
23
|
+
- triggers-present: trigger set has 0 should-load and 0 should-not (need 10 total, at least 3 of each)
|
|
24
|
+
fix: Add realistic requests to evals/triggers.json, including near-misses that share words with the skill but need something else.
|
|
25
|
+
|
|
26
|
+
1 skill: 1 skill. target skill: met
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Point it at a folder of skills or a plugin and it scores every one. Exit code 0 means every
|
|
30
|
+
skill met the target level (`--level`, default `skill`), 1 means one did not, 2 means a usage or
|
|
31
|
+
file error. `--json` prints one JSON document and nothing else.
|
|
32
|
+
|
|
33
|
+
## The levels
|
|
34
|
+
|
|
35
|
+
1. **skill**: a valid `SKILL.md` (name matches the folder, description of 1024 characters or
|
|
36
|
+
fewer that says when to use it, hard rules above the compaction fold, headings in long bodies, nothing said twice), references one level deep, no
|
|
37
|
+
hard-coded machine paths, nothing that reads like a prompt injection.
|
|
38
|
+
2. **tested**: at least three task evals with checks a machine can verify, and a trigger set of
|
|
39
|
+
at least ten requests, some that should load the skill and some near-misses that should not.
|
|
40
|
+
3. **superskill**: a golden a person approved; no miss open longer than 14 days and every fixed
|
|
41
|
+
miss guarded by an eval; a recent `--run` on file where the skill beats the same task done
|
|
42
|
+
without it. "Recent" follows `metadata.cadence` (a weekly skill's proof lasts 30 days).
|
|
43
|
+
|
|
44
|
+
Every rule and threshold is in [SPEC.md](SPEC.md).
|
|
45
|
+
|
|
46
|
+
## Commands
|
|
47
|
+
|
|
48
|
+
| Command | What it does |
|
|
49
|
+
|---|---|
|
|
50
|
+
| `superskill doctor <path...>` | Score skills. `--level`, `--json`. |
|
|
51
|
+
| `superskill doctor <path> --changed [--base <ref>] [--baseline-json <file>]` | The update gate: score only skills a change touched, and refuse one whose level dropped against a previous `--json`. For CI or an agent revising a skill in a loop. |
|
|
52
|
+
| `superskill doctor <skill> --run [--harness claude\|codex] [--repeat 3] [--yes]` | Run the evals for real, with and without the skill, and write `evals/results/latest.json`. **The only command that spends model calls**; it prints an estimate and asks first. |
|
|
53
|
+
| `superskill init <skill>` | Add missing `evals/`, `goldens/`, `MISSES.md`. Never overwrites. |
|
|
54
|
+
| `superskill init <skill> --from-session <transcript>` | Turn the session where you did the job by hand into the first eval and a golden candidate (Claude Code `.jsonl`, or any text file as the request). |
|
|
55
|
+
| `superskill miss <skill> "<what happened>" [--expected "..."]` | Log a time the skill got it wrong. |
|
|
56
|
+
| `superskill fix <skill> <miss-id> --eval <id> [--commit <sha>]` | Close a miss. Refuses without an eval that exists. |
|
|
57
|
+
| `superskill approve <skill> <golden>` | A person signs off on a golden. Terminal only, asks for your name, so an agent cannot approve its own output. |
|
|
58
|
+
| `superskill collection <folder...> [--budget <chars>] [--overlap 0.5]` | Listing budget used, descriptions that get cut off, pairs of skills an agent could confuse (with near-miss triggers to add). |
|
|
59
|
+
| `superskill miss import <skill> --freedom-ledger [--ledger <file>]` | Import runs that needed correcting from Freedom's run ledger. |
|
|
60
|
+
| `superskill snippet` | Print a block for `AGENTS.md` / `CLAUDE.md` that teaches any agent these habits. |
|
|
61
|
+
|
|
62
|
+
Each command takes `--help`.
|
|
63
|
+
|
|
64
|
+
## The files
|
|
65
|
+
|
|
66
|
+
A superskill keeps its evidence in its own folder, so the proof moves with it:
|
|
67
|
+
|
|
68
|
+
```
|
|
69
|
+
my-skill/
|
|
70
|
+
SKILL.md
|
|
71
|
+
evals/evals.json task evals (Anthropic skill-creator format)
|
|
72
|
+
evals/triggers.json should / should-not load (skill-creator format)
|
|
73
|
+
goldens/<id>/ input.md, output.md, APPROVAL.json
|
|
74
|
+
MISSES.md every miss, open or fixed with its eval
|
|
75
|
+
evals/results/latest.json the last --run
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Harnesses ignore folders they do not know, so none of this changes how the skill loads.
|
|
79
|
+
|
|
80
|
+
## About `--run`
|
|
81
|
+
|
|
82
|
+
- Claude Code: each run happens in a fresh folder with the skill linked at
|
|
83
|
+
`.claude/skills/<name>`; the baseline runs with no skill linked and `--disable-slash-commands`.
|
|
84
|
+
- Codex: the skill is linked at `.agents/skills/<name>`. Codex cannot switch skills off, so a
|
|
85
|
+
copy installed in `~/.agents/skills` can leak into the baseline; move it aside while proving.
|
|
86
|
+
- Machine checks (`contains:`, `regex:`, `file_exists:`) are free. Each plain-language expectation
|
|
87
|
+
and each golden costs one grader call per run.
|
|
88
|
+
|
|
89
|
+
## Freedom
|
|
90
|
+
|
|
91
|
+
Nothing here needs [Freedom](https://getfreedom.wiki). If a skill has Freedom's `HDSOP.md`, the
|
|
92
|
+
doctor shows it as a bonus; `miss import --freedom-ledger` reads Freedom's run ledger as plain
|
|
93
|
+
files. `superskill snippet` gives any agent the same habits with no Freedom installed.
|
|
94
|
+
|
|
95
|
+
## Releasing
|
|
96
|
+
|
|
97
|
+
Bump `version` in `package.json`, add a `CHANGELOG.md` entry, commit, then
|
|
98
|
+
`git tag vX.Y.Z && git push origin vX.Y.Z`. `.github/workflows/publish.yml` publishes through npm
|
|
99
|
+
trusted publishing. Never `npm publish` from a laptop.
|
|
100
|
+
|
|
101
|
+
## License
|
|
102
|
+
|
|
103
|
+
MIT
|
package/SPEC.md
ADDED
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
# The superskill standard
|
|
2
|
+
|
|
3
|
+
**Version 0.1.0** (2026-09-28). The reference checker is `@supersuit/superskill`; where this
|
|
4
|
+
document and the checker disagree, the checker has a bug.
|
|
5
|
+
|
|
6
|
+
A **superskill** runs on frontier intelligence, is checked against examples a person approved,
|
|
7
|
+
and is fixed every time it gets something wrong
|
|
8
|
+
([definition](https://supersuit.wiki/concepts/superskill)). This standard turns each clause into
|
|
9
|
+
a file in the skill's own folder, so the evidence travels with the skill wherever it is copied.
|
|
10
|
+
|
|
11
|
+
## Contents
|
|
12
|
+
|
|
13
|
+
- [The clauses and their evidence](#the-clauses-and-their-evidence)
|
|
14
|
+
- [Compatibility](#compatibility)
|
|
15
|
+
- [Levels and rules](#levels-and-rules)
|
|
16
|
+
- [File formats](#file-formats)
|
|
17
|
+
- [Collections and plugins](#collections-and-plugins)
|
|
18
|
+
- [Freedom interop](#freedom-interop)
|
|
19
|
+
- [Versioning](#versioning)
|
|
20
|
+
|
|
21
|
+
## The clauses and their evidence
|
|
22
|
+
|
|
23
|
+
| The clause | What proves it | Where it lives |
|
|
24
|
+
|---|---|---|
|
|
25
|
+
| A skill at all | A valid `SKILL.md` under the Agent Skills spec, plus the hygiene rules below | `SKILL.md` |
|
|
26
|
+
| Checked against examples you approved | At least one golden: a real input, the output a person said was right, and a record of who approved it and when | `goldens/<id>/` |
|
|
27
|
+
| Fixed every time it gets something wrong | A miss log where every miss is open (recently) or fixed with a regression eval that would catch it again | `MISSES.md` + `evals/evals.json` |
|
|
28
|
+
| Runs on frontier intelligence | Its evals last passed on a current model, recently, and beat the same task run without the skill | `evals/results/latest.json` |
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
my-skill/
|
|
32
|
+
SKILL.md
|
|
33
|
+
references/ scripts/ ... (as the Agent Skills spec allows)
|
|
34
|
+
evals/evals.json task evals (level: tested)
|
|
35
|
+
evals/triggers.json trigger evals (level: tested)
|
|
36
|
+
goldens/<id>/input.md approved examples (level: superskill)
|
|
37
|
+
goldens/<id>/output.md
|
|
38
|
+
goldens/<id>/APPROVAL.json
|
|
39
|
+
MISSES.md miss log (level: superskill)
|
|
40
|
+
evals/results/latest.json last --run (level: superskill)
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Compatibility
|
|
44
|
+
|
|
45
|
+
- **A superset of the Agent Skills spec** ([agentskills.io](https://agentskills.io)). Everything
|
|
46
|
+
beyond `SKILL.md` is optional to that spec, and harnesses ignore folders they do not know, so a
|
|
47
|
+
superskill loads everywhere a plain skill does.
|
|
48
|
+
- **`evals/evals.json` and `evals/triggers.json` use the formats of Anthropic's skill-creator**
|
|
49
|
+
(`anthropics/skills`, `skills/skill-creator/references/schemas.md`, read 2026-09-28). Its tools
|
|
50
|
+
read these files and this checker reads theirs. The checker writes their field names
|
|
51
|
+
(`expectations`) and also accepts `assertions` on read.
|
|
52
|
+
- **Unknown frontmatter is ignored**, as the spec allows (for example Freedom's `spans_turns` and
|
|
53
|
+
`analytics`).
|
|
54
|
+
|
|
55
|
+
## Levels and rules
|
|
56
|
+
|
|
57
|
+
A skill is at the highest level whose rules, and every lower level's rules, report no `fail`.
|
|
58
|
+
Below `skill` it is `none`. `warn` and `info` never block a level. The doctor lists, for each
|
|
59
|
+
skill, exactly which `fail`s stand between it and the next level.
|
|
60
|
+
|
|
61
|
+
A line in a bundled file containing `superskill-ignore` is skipped by `no-absolute-paths` and
|
|
62
|
+
`injection-scan`, for a deliberate example.
|
|
63
|
+
|
|
64
|
+
### Level 1: skill
|
|
65
|
+
|
|
66
|
+
| Rule | Severity | Threshold |
|
|
67
|
+
|---|---|---|
|
|
68
|
+
| `frontmatter-valid` | fail | `SKILL.md` exists and opens with a `---` fenced YAML block |
|
|
69
|
+
| `name-format` | fail | `name` is 1-64 chars, `^[a-z0-9]+(-[a-z0-9]+)*$` (so no `--`) |
|
|
70
|
+
| `name-matches-folder` | fail | `name` equals the folder name |
|
|
71
|
+
| `name-reserved-words` | fail | `name` contains neither `anthropic` nor `claude` |
|
|
72
|
+
| `description-length` | fail | `description` is 1-1024 characters |
|
|
73
|
+
| `description-has-trigger` | warn | `description` says when to use it (`when`, `trigger`, `for requests`), or `when_to_use` is set |
|
|
74
|
+
| `description-no-xml` | fail | no `<tag>` in `description`, placeholders like `<slug>` included (Anthropic's skill guidance forbids XML tags in the description) |
|
|
75
|
+
| `compatibility-length` | fail | `compatibility`, if present, is at most 500 characters |
|
|
76
|
+
| `metadata-string-map` | fail | `metadata`, if present, maps string keys to string values |
|
|
77
|
+
| `body-size` | info | reports lines and estimated tokens (characters / 4) once the body passes about 5000 tokens. Length alone is never a defect. |
|
|
78
|
+
| `rules-above-the-fold` | warn | in a body past about 5000 tokens, every hard rule (a shouted NEVER, ALWAYS, MUST, DO NOT, REFUSE, or a bolded **Never ...** command, outside code fences) appears in the first 5000 tokens, either there or restated there. After compaction Claude Code keeps only that much of each invoked skill. |
|
|
79
|
+
| `reference-says-when` | warn | every link from SKILL.md to a markdown file sits on a line that says when to read it (before, when, if, for, read ...). Step files (`steps/<step>.md`) are the recommended way to keep a long skill's detail out of the always-loaded body: they are read fresh when the step comes up, so compaction does not lose them. |
|
|
80
|
+
| `navigable` | warn | a body over 300 lines has no run of more than 150 lines without a heading |
|
|
81
|
+
| `no-repeated-paragraphs` | warn | no paragraph of 100+ characters appears twice |
|
|
82
|
+
| `references-one-deep` | fail | a markdown file linked from `SKILL.md` links on to no further local file |
|
|
83
|
+
| `long-reference-toc` | warn | every markdown file over 100 lines (other than `SKILL.md` and Freedom's `HDSOP.md`) has a table of contents in its first 30 lines (a line matching `/contents/i`, or three or more `- [x](#anchor)` lines) |
|
|
84
|
+
| `no-absolute-paths` | fail | no bundled text file (outside `evals/`, `goldens/` and test files such as `tests/`, `test_*.py`, `*.test.mjs`) contains a path starting `/Users/<name>`, `/home/<name>` or `C:\<name>` |
|
|
85
|
+
| `injection-scan` | fail | no instruction file contains an override phrase ("ignore all previous instructions", "disregard the system prompt"), a download piped into a shell (`curl ... \| sh`), a base64-like run of 200+ characters, or an HTML comment that addresses the agent (`<!-- assistant: ...`) or pairs an action (send, read, upload, run...) with a secret or a URL |
|
|
86
|
+
| `workflow-map` | info | a Freedom `HDSOP.md` is present (bonus, never required) |
|
|
87
|
+
|
|
88
|
+
### Level 2: tested
|
|
89
|
+
|
|
90
|
+
| Rule | Severity | Threshold |
|
|
91
|
+
|---|---|---|
|
|
92
|
+
| `evals-present` | fail | `evals/evals.json` parses and there are at least 3 cases (each golden with an input and an output counts as one) |
|
|
93
|
+
| `evals-verifiable` | fail | every case in `evals.json` has a prompt and at least one expectation |
|
|
94
|
+
| `triggers-present` | fail | `evals/triggers.json` has at least 10 queries, at least 3 that should load the skill and at least 3 near-misses that should not |
|
|
95
|
+
|
|
96
|
+
### Level 3: superskill
|
|
97
|
+
|
|
98
|
+
| Rule | Severity | Threshold |
|
|
99
|
+
|---|---|---|
|
|
100
|
+
| `golden-approved` | fail | at least one golden has an `APPROVAL.json` with non-empty `approved_by` and a valid `approved_at`; info when it was approved against an earlier `SKILL.md` |
|
|
101
|
+
| `misses-log-present` | fail | `MISSES.md` exists (it may have no entries) |
|
|
102
|
+
| `no-stale-open-miss` | fail | no miss has been open more than 14 days |
|
|
103
|
+
| `fixed-miss-has-eval` | fail | every fixed miss names an eval id present in `evals.json` or `goldens/` |
|
|
104
|
+
| `run-evidence` | fail | `evals/results/latest.json` exists and `with_skill.pass_rate` > `without_skill.pass_rate` |
|
|
105
|
+
| `run-fresh` | fail / warn | the last run is younger than the cadence window: `daily` or `weekly` 30 days, `monthly` 60, `quarterly` 120, `yearly` 365, none declared 30 (info). A `yearly` skill always warns to `--run` before its next real use. An unknown cadence warns |
|
|
106
|
+
|
|
107
|
+
`metadata.cadence` in `SKILL.md` frontmatter declares how often the skill really runs:
|
|
108
|
+
|
|
109
|
+
```yaml
|
|
110
|
+
metadata:
|
|
111
|
+
cadence: weekly
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
## File formats
|
|
115
|
+
|
|
116
|
+
### `evals/evals.json` (skill-creator)
|
|
117
|
+
|
|
118
|
+
```json
|
|
119
|
+
{
|
|
120
|
+
"skill_name": "weekly-status",
|
|
121
|
+
"evals": [
|
|
122
|
+
{
|
|
123
|
+
"id": 1,
|
|
124
|
+
"prompt": "Here are my finished tasks: ... Write my weekly status.",
|
|
125
|
+
"expected_output": "A note grouped by project",
|
|
126
|
+
"files": ["evals/files/tasks.csv"],
|
|
127
|
+
"expectations": ["contains:Atlas", "regex:(?i)^## ", "file_exists:out.md", "Every line is in the past tense"]
|
|
128
|
+
}
|
|
129
|
+
]
|
|
130
|
+
}
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
`id` is an integer or a string (misses and `init --from-session` use strings like `m1`, `s1`).
|
|
134
|
+
`files` are paths relative to the skill, copied into the run folder. An expectation is either a
|
|
135
|
+
machine check or a plain-language statement:
|
|
136
|
+
|
|
137
|
+
| Form | Passes when |
|
|
138
|
+
|---|---|
|
|
139
|
+
| `contains:<text>` | the output contains `<text>` |
|
|
140
|
+
| `regex:<pattern>` | the output matches (multiline; a leading `(?i)` makes it case-insensitive) |
|
|
141
|
+
| `file_exists:<path>` | the run left `<path>` in its working folder |
|
|
142
|
+
| anything else | a grader call answers `{"pass": true, ...}` |
|
|
143
|
+
|
|
144
|
+
A bare array of cases is accepted on read, as is `assertions` for `expectations`.
|
|
145
|
+
|
|
146
|
+
### `evals/triggers.json` (skill-creator)
|
|
147
|
+
|
|
148
|
+
```json
|
|
149
|
+
[
|
|
150
|
+
{ "query": "write my weekly status from these tasks", "should_trigger": true },
|
|
151
|
+
{ "query": "write a status page for our API uptime", "should_trigger": false }
|
|
152
|
+
]
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
### `goldens/<id>/`
|
|
156
|
+
|
|
157
|
+
- `input.md`: the real request.
|
|
158
|
+
- `output.md` (or any other file that is not `input.*` or `APPROVAL.json`): the output a person
|
|
159
|
+
said was right.
|
|
160
|
+
- `APPROVAL.json`, written only by `superskill approve` at an interactive terminal:
|
|
161
|
+
|
|
162
|
+
```json
|
|
163
|
+
{ "approved_by": "Ann Example", "approved_at": "2026-09-10T15:00:00.000Z", "skill_sha": "<sha256 of SKILL.md>", "note": "Exactly the shape I send my manager." }
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
A golden is also an eval: `--run` judges the skill's output for `input.md` against the approved
|
|
167
|
+
output.
|
|
168
|
+
|
|
169
|
+
### `MISSES.md`
|
|
170
|
+
|
|
171
|
+
```markdown
|
|
172
|
+
# Misses
|
|
173
|
+
|
|
174
|
+
## m1 · 2026-09-20 · fixed
|
|
175
|
+
- What happened: A task listed twice showed up twice in the note.
|
|
176
|
+
- Should have: Merged duplicates into one line.
|
|
177
|
+
- Fix: a1b2c3d
|
|
178
|
+
- Eval: m1
|
|
179
|
+
- Source: freedom-ledger inv_2026-09-20T10-00-00Z_ab12 (optional)
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
Headings are `## <id> · <YYYY-MM-DD> · <open|fixed>`; `|` or `-` also separate. Ids are `m1`,
|
|
183
|
+
`m2`, ... Other headings are ignored.
|
|
184
|
+
|
|
185
|
+
### `evals/results/latest.json`
|
|
186
|
+
|
|
187
|
+
Written only by `superskill doctor --run`:
|
|
188
|
+
|
|
189
|
+
```json
|
|
190
|
+
{
|
|
191
|
+
"run_at": "2026-09-20T10:00:00.000Z",
|
|
192
|
+
"harness": "claude",
|
|
193
|
+
"model": "<model id the harness reported>",
|
|
194
|
+
"cases": 4,
|
|
195
|
+
"repeat": 3,
|
|
196
|
+
"with_skill": { "pass_rate": 1.0, "mean_ms": 21000, "mean_tokens": 4100 },
|
|
197
|
+
"without_skill": { "pass_rate": 0.33, "mean_ms": 18000, "mean_tokens": 3900 },
|
|
198
|
+
"per_case": [{ "id": "1", "with_skill": { "runs": 3, "passes": 3 }, "without_skill": { "runs": 3, "passes": 1 }, "failures": [] }]
|
|
199
|
+
}
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
A run passes when every expectation of its case passes; `pass_rate` is passing runs over runs.
|
|
203
|
+
|
|
204
|
+
## Collections and plugins
|
|
205
|
+
|
|
206
|
+
`superskill collection` measures a set of skills together:
|
|
207
|
+
|
|
208
|
+
- **Listing budget.** Each skill costs `name + ": " + description` (plus `when_to_use`) in the
|
|
209
|
+
listing a harness loads every turn, with the description part cut off at 1536 characters
|
|
210
|
+
(Claude Code's documented cap). The total is reported against a budget per harness: 8000
|
|
211
|
+
characters for Claude Code and Codex by default. Neither vendor published a whole-listing
|
|
212
|
+
figure as of 2026-09-28, so this default is deliberately conservative and `--budget` sets it.
|
|
213
|
+
- **Cut-off entries.** Every skill whose `description` plus `when_to_use` exceeds 1536 characters.
|
|
214
|
+
- **Overlap.** Pairs whose descriptions have a token Jaccard similarity of 0.5 or more after
|
|
215
|
+
stopwords. Each flagged pair yields a near-miss query for each skill's `triggers.json`, so the
|
|
216
|
+
fix is proven by the trigger evals rather than guessed.
|
|
217
|
+
|
|
218
|
+
A **plugin** is a folder with `.claude-plugin/plugin.json` and `skills/<name>/`. Its line reports:
|
|
219
|
+
a `version`; a `CHANGELOG.md` heading for that version; and files duplicated byte-for-byte across
|
|
220
|
+
skills (a helper to share once). A plugin is a **superplugin** when every skill in it is a
|
|
221
|
+
superskill and its own line has no fail.
|
|
222
|
+
|
|
223
|
+
## Freedom interop
|
|
224
|
+
|
|
225
|
+
Nothing here requires Freedom, and the checker never imports or runs it. Where Freedom's files
|
|
226
|
+
exist they are read as plain files:
|
|
227
|
+
|
|
228
|
+
- `HDSOP.md` beside `SKILL.md` (the workflow map) is shown as a bonus line.
|
|
229
|
+
- `superskill miss import <skill> --freedom-ledger` reads `<skill>/invocations.jsonl` and
|
|
230
|
+
`~/.freedom/ledger/skills/<plugin>/<skill>.jsonl`. Records are
|
|
231
|
+
`{id, skill, started, outcome, interventions: [{kind, note|what}], errors}`. A record with an
|
|
232
|
+
intervention of kind `redirect`, `correction` or `rescue`, or with `outcome: "failed"`, becomes
|
|
233
|
+
an open miss dated from `started`; `taste` interventions are skipped; the ledger id is kept on
|
|
234
|
+
a `Source:` line so a record is never imported twice.
|
|
235
|
+
|
|
236
|
+
## Versioning
|
|
237
|
+
|
|
238
|
+
This standard is versioned with semver. Adding a rule that can fail a skill which passed before
|
|
239
|
+
is a minor version before 1.0 and a major version after. Thresholds are part of the standard.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// superskill: score agent skills as skill / tested / superskill. Zero dependencies.
|
|
3
|
+
import { UsageError } from "../src/args.mjs";
|
|
4
|
+
import { DoctorError } from "../src/doctor.mjs";
|
|
5
|
+
|
|
6
|
+
const COMMANDS = {
|
|
7
|
+
doctor: "../src/commands/doctor.mjs",
|
|
8
|
+
init: "../src/commands/init.mjs",
|
|
9
|
+
miss: "../src/commands/miss.mjs",
|
|
10
|
+
misses: "../src/commands/miss.mjs",
|
|
11
|
+
fix: "../src/commands/fix.mjs",
|
|
12
|
+
approve: "../src/commands/approve.mjs",
|
|
13
|
+
collection: "../src/commands/collection.mjs",
|
|
14
|
+
snippet: "../src/commands/snippet.mjs",
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
const HELP = `superskill <command> [options]
|
|
18
|
+
|
|
19
|
+
Commands:
|
|
20
|
+
doctor <path...> score skills; exit 0 when the target level is met
|
|
21
|
+
init <path> add missing evals, triggers, goldens and MISSES.md
|
|
22
|
+
miss <path> "<what>" log a miss
|
|
23
|
+
fix <path> <miss-id> --eval close a miss with the regression eval that guards it
|
|
24
|
+
approve <path> <golden> record a person's approval of a golden (terminal only)
|
|
25
|
+
collection <folder> listing budget, cut-off descriptions, overlapping skills
|
|
26
|
+
snippet print a block for AGENTS.md / CLAUDE.md
|
|
27
|
+
|
|
28
|
+
Run "superskill <command> --help" for details. Spec: SPEC.md.
|
|
29
|
+
`;
|
|
30
|
+
|
|
31
|
+
async function main(argv) {
|
|
32
|
+
const [cmd, ...rest] = argv;
|
|
33
|
+
if (!cmd || cmd === "--help" || cmd === "-h" || cmd === "help") { process.stdout.write(HELP); return 0; }
|
|
34
|
+
if (cmd === "--version") {
|
|
35
|
+
const { readFileSync } = await import("node:fs");
|
|
36
|
+
process.stdout.write(JSON.parse(readFileSync(new URL("../package.json", import.meta.url), "utf8")).version + "\n");
|
|
37
|
+
return 0;
|
|
38
|
+
}
|
|
39
|
+
const mod = COMMANDS[cmd];
|
|
40
|
+
if (!mod) throw new UsageError(`unknown command "${cmd}". Run superskill --help.`);
|
|
41
|
+
const { run } = await import(mod);
|
|
42
|
+
return run(rest);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
main(process.argv.slice(2)).then(
|
|
46
|
+
(code) => { process.exitCode = code; },
|
|
47
|
+
(e) => {
|
|
48
|
+
if (e instanceof UsageError || e instanceof DoctorError || e?.code === "SUPERSKILL") {
|
|
49
|
+
process.stderr.write(`superskill: ${e.message}\n`);
|
|
50
|
+
process.exitCode = 2;
|
|
51
|
+
} else {
|
|
52
|
+
process.stderr.write(`superskill: unexpected error: ${e?.stack || e}\n`);
|
|
53
|
+
process.exitCode = 2;
|
|
54
|
+
}
|
|
55
|
+
},
|
|
56
|
+
);
|
package/package.json
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@supersuit/superskill",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Score any agent skill folder as skill, tested, or superskill. An open standard and a zero-dependency CLI.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"bin": { "superskill": "bin/superskill.mjs" },
|
|
7
|
+
"files": ["bin/", "src/", "SPEC.md", "README.md", "CHANGELOG.md", "LICENSE"],
|
|
8
|
+
"scripts": { "test": "node --test test/*.test.mjs" },
|
|
9
|
+
"engines": { "node": ">=20" },
|
|
10
|
+
"license": "MIT",
|
|
11
|
+
"repository": { "type": "git", "url": "git+https://github.com/SupersuitUp/superskill.git" },
|
|
12
|
+
"homepage": "https://supersuit.wiki/concepts/superskill",
|
|
13
|
+
"keywords": ["agent-skills", "skills", "claude-code", "codex", "evals", "superskill"],
|
|
14
|
+
"dependencies": {}
|
|
15
|
+
}
|
package/src/args.mjs
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Minimal argv parser. `booleans` names flags that never take a value; every other
|
|
3
|
+
* `--flag` consumes the next argument (or `--flag=value`). Positionals land in `_`.
|
|
4
|
+
*/
|
|
5
|
+
export function parseArgs(argv, booleans = []) {
|
|
6
|
+
const out = { _: [], flags: {} };
|
|
7
|
+
const bools = new Set(["help", "json", ...booleans]);
|
|
8
|
+
for (let i = 0; i < argv.length; i++) {
|
|
9
|
+
const a = argv[i];
|
|
10
|
+
if (a === "-h") { out.flags.help = true; continue; }
|
|
11
|
+
if (!a.startsWith("--") || a === "--") { out._.push(a); continue; }
|
|
12
|
+
const eq = a.indexOf("=");
|
|
13
|
+
const key = a.slice(2, eq > 0 ? eq : undefined);
|
|
14
|
+
if (eq > 0) out.flags[key] = a.slice(eq + 1);
|
|
15
|
+
else if (bools.has(key)) out.flags[key] = true;
|
|
16
|
+
else if (i + 1 < argv.length) out.flags[key] = argv[++i];
|
|
17
|
+
else throw new UsageError(`--${key} needs a value`);
|
|
18
|
+
}
|
|
19
|
+
return out;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export class UsageError extends Error {}
|
|
23
|
+
|
|
24
|
+
/** The clock every date rule reads. SUPERSKILL_NOW pins it for tests and reproducible reports. */
|
|
25
|
+
export function clock(flags = {}) {
|
|
26
|
+
const v = flags.now || process.env.SUPERSKILL_NOW;
|
|
27
|
+
const d = v ? new Date(v) : new Date();
|
|
28
|
+
if (Number.isNaN(d.getTime())) throw new UsageError(`not a date: ${v}`);
|
|
29
|
+
return d;
|
|
30
|
+
}
|
package/src/changed.mjs
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
// The update gate: which skills did a change touch, and did any of them lose a level.
|
|
2
|
+
import { spawnSync } from "node:child_process";
|
|
3
|
+
import { join, resolve, sep } from "node:path";
|
|
4
|
+
import { realpathSync } from "node:fs";
|
|
5
|
+
import { DoctorError } from "./doctor.mjs";
|
|
6
|
+
import { LEVELS } from "./levels.mjs";
|
|
7
|
+
|
|
8
|
+
function git(cwd, args) {
|
|
9
|
+
const r = spawnSync("git", args, { cwd, encoding: "utf8" });
|
|
10
|
+
if (r.error) throw new DoctorError(`git is not available: ${r.error.message}`);
|
|
11
|
+
return r;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
const real = (p) => { try { return realpathSync(p); } catch { return resolve(p); } };
|
|
15
|
+
|
|
16
|
+
/** Absolute paths changed since `base` (committed) plus staged, unstaged and untracked. */
|
|
17
|
+
export function changedFiles(path, base) {
|
|
18
|
+
const top = git(path, ["rev-parse", "--show-toplevel"]);
|
|
19
|
+
if (top.status !== 0) throw new DoctorError(`--changed needs a git repository: ${path} is not in one`);
|
|
20
|
+
const root = top.stdout.trim();
|
|
21
|
+
const files = new Set();
|
|
22
|
+
const add = (out) => out.split("\n").map((l) => l.trim()).filter(Boolean).forEach((f) => files.add(real(join(root, f))));
|
|
23
|
+
if (base) {
|
|
24
|
+
const d = git(root, ["diff", "--name-only", `${base}...HEAD`]);
|
|
25
|
+
if (d.status !== 0) throw new DoctorError(`git diff against ${base} failed: ${d.stderr.trim()}`);
|
|
26
|
+
add(d.stdout);
|
|
27
|
+
}
|
|
28
|
+
add(git(root, ["diff", "--name-only", "HEAD"]).stdout);
|
|
29
|
+
add(git(root, ["ls-files", "--others", "--exclude-standard"]).stdout);
|
|
30
|
+
return [...files];
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** Keep the skill dirs that contain at least one changed file. */
|
|
34
|
+
export function touchedSkills(dirs, files) {
|
|
35
|
+
return dirs.filter((d) => {
|
|
36
|
+
const prefix = real(d) + sep;
|
|
37
|
+
return files.some((f) => f.startsWith(prefix));
|
|
38
|
+
});
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Compare against a previous `doctor --json`: every skill whose level went down. */
|
|
42
|
+
export function levelDrops(current, baseline) {
|
|
43
|
+
const rank = (l) => (l === "none" ? -1 : LEVELS.indexOf(l));
|
|
44
|
+
const before = new Map();
|
|
45
|
+
for (const s of baseline.skills || []) { before.set(s.path, s); before.set(`name:${s.name}`, s); }
|
|
46
|
+
const drops = [];
|
|
47
|
+
for (const s of current.skills) {
|
|
48
|
+
const b = before.get(s.path) || before.get(`name:${s.name}`);
|
|
49
|
+
if (b && rank(s.level) < rank(b.level)) drops.push({ name: s.name, path: s.path, from: b.level, to: s.level });
|
|
50
|
+
}
|
|
51
|
+
return drops;
|
|
52
|
+
}
|