@supersuit/superskill 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +68 -0
- package/README.md +22 -12
- package/SPEC.md +141 -17
- package/package.json +1 -1
- package/src/commands/approve.mjs +15 -5
- package/src/commands/doctor.mjs +32 -3
- package/src/commands/init.mjs +8 -2
- package/src/commands/miss.mjs +7 -4
- package/src/doctor.mjs +1 -0
- package/src/goldens.mjs +94 -26
- package/src/ledger.mjs +40 -4
- package/src/misses.mjs +22 -13
- package/src/realruns.mjs +122 -0
- package/src/rules/skill.mjs +71 -0
- package/src/rules/superskill.mjs +93 -27
- package/src/rules/tested.mjs +25 -2
- package/src/run/claude.mjs +31 -1
- package/src/run/codex.mjs +16 -1
- package/src/run/env.mjs +17 -0
- package/src/run/fake.mjs +9 -3
- package/src/run/index.mjs +57 -14
- package/src/run/workspace.mjs +2 -1
- package/src/snippet.md +3 -1
package/src/rules/superskill.mjs
CHANGED
|
@@ -1,14 +1,19 @@
|
|
|
1
|
-
// Level 3, "superskill":
|
|
2
|
-
//
|
|
1
|
+
// Level 3, "superskill" (0.6.0): fixed every time it got something wrong, with a regression eval
|
|
2
|
+
// for each fix; its whole suite and trigger set pass against the SKILL.md that is here now, on a
|
|
3
|
+
// current model and better than no skill; and it did the job for real people, one-shot, often
|
|
4
|
+
// enough on at least one model+harness. Goldens are optional evidence.
|
|
3
5
|
import { createHash } from "node:crypto";
|
|
4
6
|
import { defineRules } from "./define.mjs";
|
|
5
|
-
import { readGoldens, isApproved, weightOf } from "../goldens.mjs";
|
|
7
|
+
import { readGoldens, isApproved, isRealApproved, weightOf, goldenOpts } from "../goldens.mjs";
|
|
8
|
+
import { readRealRuns, meetsBar, knownPair, MIN_REAL_RUNS, MIN_ONE_SHOT } from "../realruns.mjs";
|
|
6
9
|
import { readMisses } from "../misses.mjs";
|
|
7
10
|
import { readEvals, readLatestRun } from "../evals.mjs";
|
|
8
11
|
|
|
9
12
|
const f = (severity, message, fix) => ({ severity, message, fix });
|
|
10
13
|
const DAY = 86400000;
|
|
11
|
-
|
|
14
|
+
/** Defaults for evals-pass; `--min-pass-rate` / `--min-trigger-rate` change them. */
|
|
15
|
+
export const MIN_PASS_RATE = 0.9;
|
|
16
|
+
export const MIN_TRIGGER_RATE = 0.9;
|
|
12
17
|
/** How long a --run stays fresh, by metadata.cadence. */
|
|
13
18
|
export const FRESH_DAYS = { daily: 30, weekly: 30, monthly: 60, quarterly: 120, yearly: 365 };
|
|
14
19
|
const DEFAULT_FRESH = 30;
|
|
@@ -17,29 +22,56 @@ const days = (now, date) => Math.floor((now.getTime() - new Date(date).getTime()
|
|
|
17
22
|
|
|
18
23
|
export const superskillRules = defineRules([
|
|
19
24
|
{
|
|
25
|
+
// 0.6.0: a golden is optional evidence, never the bar. A random accepted run is not the
|
|
26
|
+
// embodiment of a skill; the misses it absorbed and the evals that hold them are. A golden
|
|
27
|
+
// that exists is still an eval (its checklist is graded by --run), so its files are checked.
|
|
20
28
|
id: "golden-approved",
|
|
21
29
|
level: "superskill",
|
|
22
|
-
check(ctx) {
|
|
30
|
+
check(ctx, opts = {}) {
|
|
23
31
|
if (ctx.error) return [];
|
|
24
|
-
const goldens = readGoldens(ctx.dir);
|
|
25
|
-
|
|
32
|
+
const goldens = readGoldens(ctx.dir, goldenOpts(ctx, opts));
|
|
33
|
+
if (!goldens.length) return [];
|
|
26
34
|
const out = [];
|
|
35
|
+
const where = (g) => (g.private ? `private golden ${g.id}` : `goldens/${g.id}`);
|
|
27
36
|
for (const g of goldens.filter((x) => x.approvalError))
|
|
28
|
-
out.push(f("
|
|
29
|
-
|
|
30
|
-
out.push(f("
|
|
31
|
-
|
|
32
|
-
|
|
37
|
+
out.push(f("warn", `${where(g)}/APPROVAL.json is not valid JSON`, `Re-record it with \`superskill approve . ${g.id}\`.`));
|
|
38
|
+
for (const g of goldens.filter((x) => x.provenanceError))
|
|
39
|
+
out.push(f("warn", `${where(g)}/PROVENANCE.json is not valid JSON`, "Rewrite it: source, run, accepted (see SPEC.md, goldens)."));
|
|
40
|
+
for (const g of goldens.filter((x) => !x.expectations.length))
|
|
41
|
+
out.push(f("info", `${where(g)} has no expectations.json, so --run grades it by likeness to output.md`, "Write goldens/<id>/expectations.json: the behaviors a right answer shows (grade the outcome, not the path). output.md then stays as the reference that proves the task is solvable."));
|
|
42
|
+
const approved = goldens.filter(isApproved);
|
|
43
|
+
const real = goldens.filter(isRealApproved);
|
|
44
|
+
const unreal = approved.filter((g) => !real.includes(g));
|
|
45
|
+
if (unreal.length) out.push(f("info", `approved golden${unreal.length === 1 ? "" : "s"} not from a real run a person accepted: ${unreal.map((g) => `${g.id} ${g.origin.why}`).join("; ")}`, "Only a golden whose PROVENANCE.json names a real run and who accepted it is evidence of real use; an invented one is an ordinary eval."));
|
|
46
|
+
if (!real.length) return out;
|
|
33
47
|
const sha = createHash("sha256").update(ctx.raw).digest("hex");
|
|
34
|
-
const stale =
|
|
48
|
+
const stale = real.filter((g) => g.approval.skill_sha && g.approval.skill_sha !== sha).map((g) => g.id);
|
|
35
49
|
if (stale.length) out.push(f("info", `golden${stale.length === 1 ? "" : "s"} ${stale.join(", ")} approved against an earlier SKILL.md`, "Re-run the golden and re-approve if the output still holds."));
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
50
|
+
const w = real.map((g) => ({ id: g.id, ...weightOf(g) }));
|
|
51
|
+
out.push(f("info", `real goldens (optional evidence): ${w.map((x) => `${x.id}: ${x.judgment} judgment, ${x.outcome} outcome`).join("; ")}`, ""));
|
|
52
|
+
return out;
|
|
53
|
+
},
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
// The record from real use, per model and harness. A clean record on one model in one harness
|
|
57
|
+
// proves nothing about another, so pairs are never pooled: one pair has to clear the bar alone.
|
|
58
|
+
id: "real-runs",
|
|
59
|
+
level: "superskill",
|
|
60
|
+
check(ctx, opts = {}) {
|
|
61
|
+
if (ctx.error) return [];
|
|
62
|
+
const name = (typeof ctx.data?.name === "string" && ctx.data.name) || ctx.folderName;
|
|
63
|
+
const minRuns = num(opts.minRealRuns, MIN_REAL_RUNS), minOneShot = num(opts.minOneShot, MIN_ONE_SHOT);
|
|
64
|
+
const bar = `at least ${minRuns} real runs and ${pct(minOneShot)} one-shot on one model+harness`;
|
|
65
|
+
const realRunsDir = opts.realRuns ?? ctx.realRuns ?? process.env.SUPERSKILL_REAL_RUNS ?? null;
|
|
66
|
+
const rec = readRealRuns(ctx.dir, { name, realRunsDir });
|
|
67
|
+
const how = "Export the run record (format superskill-real-runs/1, SPEC.md) to evals/real-runs.json or --real-runs <dir>; Freedom's skill ledger exports it.";
|
|
68
|
+
if (!rec) return [f("fail", `no real-run record (needs ${bar})`, how)];
|
|
69
|
+
if (rec.error) return [f("fail", rec.error, how)];
|
|
70
|
+
const out = rec.pairs.map((p) => f("info", `real runs, ${p.model} / ${p.harness}: ${p.one_shot} of ${p.runs} one-shot${p.runs ? ` (${pct(p.one_shot / p.runs)})` : ""}${knownPair(p) ? "" : ", not counted: the record does not say which model or harness"}`, ""));
|
|
71
|
+
if (!meetsBar(rec.pairs, { minRuns, minOneShot }).length) {
|
|
72
|
+
const best = rec.pairs.filter(knownPair)[0];
|
|
73
|
+
out.push(f("fail", `real-run record below the bar (${bar}; ${rec.where})${best ? `: best is ${best.model} / ${best.harness}, ${best.one_shot} of ${best.runs}` : ": no run names its model and harness"}`, "Use the skill for real and fix what it gets wrong; every correction is a miss to close with an eval."));
|
|
74
|
+
}
|
|
43
75
|
return out;
|
|
44
76
|
},
|
|
45
77
|
},
|
|
@@ -52,18 +84,17 @@ export const superskillRules = defineRules([
|
|
|
52
84
|
},
|
|
53
85
|
},
|
|
54
86
|
{
|
|
55
|
-
|
|
87
|
+
// 0.6.0: every miss is closed. An open miss is a known way the skill fails today, and a skill
|
|
88
|
+
// that still fails a known way is not at the top level, however recent the miss is.
|
|
89
|
+
id: "misses-closed",
|
|
56
90
|
level: "superskill",
|
|
57
91
|
check(ctx, { now = new Date() } = {}) {
|
|
58
92
|
if (ctx.error) return [];
|
|
59
93
|
const misses = readMisses(ctx.dir) || [];
|
|
60
|
-
|
|
61
|
-
for (const m of misses.filter((x) => x.status === "open")) {
|
|
94
|
+
return misses.filter((x) => x.status === "open").map((m) => {
|
|
62
95
|
const age = days(now, m.date);
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
}
|
|
66
|
-
return out;
|
|
96
|
+
return f("fail", `miss ${m.id} is open (${age} day${age === 1 ? "" : "s"}): ${m.what.slice(0, 80)}`, `Fix it, add an eval that catches it, then \`superskill fix <skill> ${m.id} --eval <id>\`.`);
|
|
97
|
+
});
|
|
67
98
|
},
|
|
68
99
|
},
|
|
69
100
|
{
|
|
@@ -74,12 +105,46 @@ export const superskillRules = defineRules([
|
|
|
74
105
|
const misses = (readMisses(ctx.dir) || []).filter((m) => m.status === "fixed");
|
|
75
106
|
if (!misses.length) return [];
|
|
76
107
|
const ids = new Set(readEvals(ctx.dir).cases.map((c) => String(c.id)));
|
|
108
|
+
// Shipped goldens only: a miss is closed by a check that travels with the skill.
|
|
77
109
|
for (const g of readGoldens(ctx.dir)) ids.add(g.id);
|
|
78
110
|
return misses
|
|
79
111
|
.filter((m) => !m.eval || !ids.has(String(m.eval)))
|
|
80
112
|
.map((m) => f("fail", m.eval ? `miss ${m.id} names eval "${m.eval}", which is not in evals.json or goldens/` : `miss ${m.id} is fixed with no regression eval`, `Add a case to evals/evals.json that would catch ${m.id} again, and name its id on the Eval line.`));
|
|
81
113
|
},
|
|
82
114
|
},
|
|
115
|
+
{
|
|
116
|
+
// The suite and the trigger set pass against the SKILL.md that is here now. A pass recorded
|
|
117
|
+
// against an earlier SKILL.md proves the earlier skill.
|
|
118
|
+
id: "evals-pass",
|
|
119
|
+
level: "superskill",
|
|
120
|
+
check(ctx, opts = {}) {
|
|
121
|
+
if (ctx.error) return [];
|
|
122
|
+
const r = readLatestRun(ctx.dir);
|
|
123
|
+
if (!r.run) return [];
|
|
124
|
+
const run = r.run;
|
|
125
|
+
const out = [];
|
|
126
|
+
const sha = createHash("sha256").update(ctx.raw).digest("hex");
|
|
127
|
+
if (!run.skill_sha) out.push(f("fail", "latest.json does not say which SKILL.md it ran against (no skill_sha)", "Re-run `superskill doctor <skill> --run` with superskill 0.6.0 or later."));
|
|
128
|
+
else if (run.skill_sha !== sha) out.push(f("fail", "the last --run was against an earlier SKILL.md", "Re-run `superskill doctor <skill> --run`; a change to the skill needs its evals re-run."));
|
|
129
|
+
const minPass = num(opts.minPassRate, MIN_PASS_RATE), minTrig = num(opts.minTriggerRate, MIN_TRIGGER_RATE);
|
|
130
|
+
const w = Number(run.with_skill?.pass_rate);
|
|
131
|
+
if (Number.isFinite(w) && w < minPass) out.push(f("fail", `the suite passed ${pct(w)} of runs with the skill (needs ${pct(minPass)})`, "Read the failures in latest.json per_case, fix the skill, re-run."));
|
|
132
|
+
// Every fixed miss's regression eval passes every run: a regression that comes back half the
|
|
133
|
+
// time has come back.
|
|
134
|
+
const fixed = (readMisses(ctx.dir) || []).filter((m) => m.status === "fixed" && m.eval);
|
|
135
|
+
const cases = new Map((Array.isArray(run.per_case) ? run.per_case : []).map((c) => [String(c.id), c]));
|
|
136
|
+
for (const m of fixed) {
|
|
137
|
+
const c = cases.get(String(m.eval)) || cases.get(`golden:${m.eval}`);
|
|
138
|
+
if (!c) { if (run.skill_sha) out.push(f("fail", `regression eval ${m.eval} (miss ${m.id}) is not in the last --run`, "Re-run `superskill doctor <skill> --run`.")); continue; }
|
|
139
|
+
const ws = c.with_skill || {};
|
|
140
|
+
if (!(ws.runs > 0 && ws.passes === ws.runs)) out.push(f("fail", `regression eval ${m.eval} (miss ${m.id}) passed ${ws.passes || 0} of ${ws.runs || 0} runs`, `Miss ${m.id} is back: fix the skill until its eval passes every run.`));
|
|
141
|
+
}
|
|
142
|
+
const t = run.triggers;
|
|
143
|
+
if (!t || !Number.isFinite(Number(t.pass_rate))) out.push(f("fail", "the last --run did not run the trigger evals", "Re-run `superskill doctor <skill> --run` with superskill 0.6.0 or later; it runs evals/triggers.json too."));
|
|
144
|
+
else if (Number(t.pass_rate) < minTrig) out.push(f("fail", `trigger evals ${pct(Number(t.pass_rate))} right (needs ${pct(minTrig)})`, "Read triggers.per_query in latest.json: sharpen the description so it loads when it should and not otherwise."));
|
|
145
|
+
return out;
|
|
146
|
+
},
|
|
147
|
+
},
|
|
83
148
|
{
|
|
84
149
|
id: "run-evidence",
|
|
85
150
|
level: "superskill",
|
|
@@ -118,3 +183,4 @@ export const superskillRules = defineRules([
|
|
|
118
183
|
]);
|
|
119
184
|
|
|
120
185
|
const pct = (x) => `${Math.round(x * 100)}%`;
|
|
186
|
+
const num = (v, d) => (v === undefined || v === null || v === "" || !Number.isFinite(Number(v)) ? d : Number(v));
|
package/src/rules/tested.mjs
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
// with both should-load requests and near-misses that should not load the skill.
|
|
3
3
|
import { defineRules } from "./define.mjs";
|
|
4
4
|
import { readEvals, readTriggers } from "../evals.mjs";
|
|
5
|
-
import { readGoldens } from "../goldens.mjs";
|
|
5
|
+
import { readGoldens, goldenOpts } from "../goldens.mjs";
|
|
6
6
|
|
|
7
7
|
const f = (severity, message, fix) => ({ severity, message, fix });
|
|
8
8
|
export const MIN_EVALS = 3, MIN_TRIGGERS = 10, MIN_EACH_SIDE = 3;
|
|
@@ -15,7 +15,7 @@ export const testedRules = defineRules([
|
|
|
15
15
|
if (ctx.error) return [];
|
|
16
16
|
const e = readEvals(ctx.dir);
|
|
17
17
|
if (e.error) return [f("fail", e.error, "Fix evals/evals.json so it parses as skill-creator's {skill_name, evals: [...]}.")];
|
|
18
|
-
const goldenCases = readGoldens(ctx.dir).filter((g) => g.input !== null && g.output !== null).length;
|
|
18
|
+
const goldenCases = readGoldens(ctx.dir, goldenOpts(ctx)).filter((g) => g.input !== null && g.output !== null).length;
|
|
19
19
|
const n = e.cases.length + goldenCases;
|
|
20
20
|
if (n < MIN_EVALS)
|
|
21
21
|
return [f("fail", `${n} eval case${n === 1 ? "" : "s"} (need ${MIN_EVALS})`, "Add real requests to evals/evals.json (`superskill init` writes an example).")];
|
|
@@ -35,6 +35,29 @@ export const testedRules = defineRules([
|
|
|
35
35
|
: [];
|
|
36
36
|
},
|
|
37
37
|
},
|
|
38
|
+
{
|
|
39
|
+
// MEASURED 2026-10-05 (freedom-dev): `superskill init` writes one example eval and two example
|
|
40
|
+
// triggers, each starting "REPLACE:", and a skill whose author copied those up to 3 and 10
|
|
41
|
+
// scored `tested` while testing nothing. A placeholder, or the same request twice, is not a case.
|
|
42
|
+
id: "evals-real",
|
|
43
|
+
level: "tested",
|
|
44
|
+
check(ctx) {
|
|
45
|
+
if (ctx.error) return [];
|
|
46
|
+
const e = readEvals(ctx.dir), t = readTriggers(ctx.dir);
|
|
47
|
+
const out = [];
|
|
48
|
+
const placeholder = (x) => /\bREPLACE:/.test(JSON.stringify(x));
|
|
49
|
+
const evalHits = e.cases.filter((c) => placeholder([c.prompt, c.expected_output, c.assertions])).map((c) => c.id);
|
|
50
|
+
if (evalHits.length) out.push(f("fail", `eval case${evalHits.length === 1 ? "" : "s"} ${evalHits.join(", ")} still hold${evalHits.length === 1 ? "s" : ""} a \`superskill init\` REPLACE: placeholder`, "Replace every example with a real request this skill handles and what a good answer must contain."));
|
|
51
|
+
const trigHits = t.triggers.filter((x) => placeholder(x.query)).length;
|
|
52
|
+
if (trigHits) out.push(f("fail", `${trigHits} trigger${trigHits === 1 ? "" : "s"} still hold a REPLACE: placeholder`, "Replace them with realistic requests, including near-misses that should not load the skill."));
|
|
53
|
+
const dup = (list) => list.filter((x, i) => x && list.indexOf(x) !== i);
|
|
54
|
+
const dp = [...new Set(dup(e.cases.map((c) => c.prompt.trim())))];
|
|
55
|
+
if (dp.length) out.push(f("fail", `${dp.length} eval prompt${dp.length === 1 ? " is" : "s are"} repeated: the same request twice is one case`, "Make each case a different request."));
|
|
56
|
+
const dq = [...new Set(dup(t.triggers.map((x) => x.query.trim().toLowerCase())))];
|
|
57
|
+
if (dq.length) out.push(f("fail", `${dq.length} trigger quer${dq.length === 1 ? "y is" : "ies are"} repeated`, "Make each trigger a different request."));
|
|
58
|
+
return out;
|
|
59
|
+
},
|
|
60
|
+
},
|
|
38
61
|
{
|
|
39
62
|
id: "triggers-present",
|
|
40
63
|
level: "tested",
|
package/src/run/claude.mjs
CHANGED
|
@@ -3,12 +3,13 @@
|
|
|
3
3
|
// a copy installed at user level cannot leak into the baseline.
|
|
4
4
|
import { spawnSync } from "node:child_process";
|
|
5
5
|
import { prepareWorkspace } from "./workspace.mjs";
|
|
6
|
+
import { sandboxEnv } from "./env.mjs";
|
|
6
7
|
|
|
7
8
|
export const name = "claude";
|
|
8
9
|
|
|
9
10
|
function call(args, cwd) {
|
|
10
11
|
const t0 = Date.now();
|
|
11
|
-
const r = spawnSync("claude", args, { cwd, encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
|
|
12
|
+
const r = spawnSync("claude", args, { cwd, env: sandboxEnv(), encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
|
|
12
13
|
const ms = Date.now() - t0;
|
|
13
14
|
if (r.error) throw Object.assign(new Error(`claude failed to start: ${r.error.message}`), { code: "SUPERSKILL" });
|
|
14
15
|
let doc = null;
|
|
@@ -34,3 +35,32 @@ export function ask(prompt, { model } = {}) {
|
|
|
34
35
|
if (model) args.push("--model", model);
|
|
35
36
|
return call(args, cwd).output;
|
|
36
37
|
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Did Claude Code load the skill for this query? Run headless with the skill installed and read
|
|
41
|
+
* the stream: a Skill tool call naming it, or a Read of its SKILL.md, is a load.
|
|
42
|
+
*/
|
|
43
|
+
export function loadedSkill(stream, skillName) {
|
|
44
|
+
for (const line of String(stream).split("\n")) {
|
|
45
|
+
if (!line.trim().startsWith("{")) continue;
|
|
46
|
+
let ev; try { ev = JSON.parse(line); } catch { continue; }
|
|
47
|
+
const content = ev?.message?.content;
|
|
48
|
+
if (!Array.isArray(content)) continue;
|
|
49
|
+
for (const c of content) {
|
|
50
|
+
if (c?.type !== "tool_use") continue;
|
|
51
|
+
const input = c.input || {};
|
|
52
|
+
if (c.name === "Skill" && [input.skill, input.command, input.name].some((v) => typeof v === "string" && v.split(":").pop() === skillName)) return true;
|
|
53
|
+
if (typeof input.file_path === "string" && input.file_path.endsWith(`/${skillName}/SKILL.md`)) return true;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return false;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export function triggerCase({ skillDir, skillName, query, model }) {
|
|
60
|
+
const cwd = prepareWorkspace({ skillDir, skillName, linkAt: ".claude/skills" });
|
|
61
|
+
const args = ["-p", query, "--output-format", "stream-json", "--verbose", "--add-dir", cwd];
|
|
62
|
+
if (model) args.push("--model", model);
|
|
63
|
+
const r = spawnSync("claude", args, { cwd, env: sandboxEnv(), encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
|
|
64
|
+
if (r.error) throw Object.assign(new Error(`claude failed to start: ${r.error.message}`), { code: "SUPERSKILL" });
|
|
65
|
+
return { triggered: loadedSkill(r.stdout, skillName), failed: r.status !== 0 && !r.stdout, cwd };
|
|
66
|
+
}
|
package/src/run/codex.mjs
CHANGED
|
@@ -5,6 +5,7 @@ import { spawnSync } from "node:child_process";
|
|
|
5
5
|
import { readFileSync, existsSync } from "node:fs";
|
|
6
6
|
import { join } from "node:path";
|
|
7
7
|
import { prepareWorkspace } from "./workspace.mjs";
|
|
8
|
+
import { sandboxEnv } from "./env.mjs";
|
|
8
9
|
|
|
9
10
|
export const name = "codex";
|
|
10
11
|
|
|
@@ -14,7 +15,7 @@ function call(prompt, cwd, model) {
|
|
|
14
15
|
if (model) args.push("-m", model);
|
|
15
16
|
args.push(prompt);
|
|
16
17
|
const t0 = Date.now();
|
|
17
|
-
const r = spawnSync("codex", args, { cwd, encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
|
|
18
|
+
const r = spawnSync("codex", args, { cwd, env: sandboxEnv(), encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
|
|
18
19
|
if (r.error) throw Object.assign(new Error(`codex failed to start: ${r.error.message}`), { code: "SUPERSKILL" });
|
|
19
20
|
const output = existsSync(out) ? readFileSync(out, "utf8") : r.stdout;
|
|
20
21
|
const tok = (r.stderr + r.stdout).match(/tokens used[:\s]+([\d,]+)/i);
|
|
@@ -30,3 +31,17 @@ export function ask(prompt, { model } = {}) {
|
|
|
30
31
|
const cwd = prepareWorkspace({ skillDir: null, skillName: null });
|
|
31
32
|
return call(prompt, cwd, model).output;
|
|
32
33
|
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Did Codex load the skill for this query? Codex reads a skill's SKILL.md when it uses it, so the
|
|
37
|
+
* JSON event stream mentioning <name>/SKILL.md is a load.
|
|
38
|
+
*/
|
|
39
|
+
export function triggerCase({ skillDir, skillName, query, model }) {
|
|
40
|
+
const cwd = prepareWorkspace({ skillDir, skillName, linkAt: ".agents/skills" });
|
|
41
|
+
const args = ["exec", "--json", "--skip-git-repo-check", "-C", cwd];
|
|
42
|
+
if (model) args.push("-m", model);
|
|
43
|
+
args.push(query);
|
|
44
|
+
const r = spawnSync("codex", args, { cwd, env: sandboxEnv(), encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
|
|
45
|
+
if (r.error) throw Object.assign(new Error(`codex failed to start: ${r.error.message}`), { code: "SUPERSKILL" });
|
|
46
|
+
return { triggered: String(r.stdout).includes(`${skillName}/SKILL.md`), failed: r.status !== 0 && !r.stdout, cwd };
|
|
47
|
+
}
|
package/src/run/env.mjs
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
// The environment every sandboxed run gets, so a `doctor --run` session is never counted as a
|
|
2
|
+
// real use of the skill by whatever is recording real uses.
|
|
3
|
+
//
|
|
4
|
+
// MEASURED, NOT FEARED (2026-10-05): Freedom's skill ledger records every skill run from a
|
|
5
|
+
// harness hook. The doctor's own sandbox runs fired that hook, so three runs of a skill nobody had
|
|
6
|
+
// ever used for real landed in the operator's ledger as perfect one-shot runs. Left alone, every
|
|
7
|
+
// `doctor --run` raises the number that is meant to keep the doctor honest.
|
|
8
|
+
//
|
|
9
|
+
// Two signals, because a caller's recorder may honour either: FREEDOM_SKILL_LEDGER=off is the
|
|
10
|
+
// Freedom ledger's documented off switch, and SUPERSKILL_SANDBOX=1 names the run for any other
|
|
11
|
+
// recorder. The run folder's own name (SANDBOX_PREFIX) is the third, for a recorder that reads
|
|
12
|
+
// only the hook payload's cwd.
|
|
13
|
+
export const SANDBOX_PREFIX = "superskill-run-";
|
|
14
|
+
|
|
15
|
+
export function sandboxEnv(base = process.env) {
|
|
16
|
+
return { ...base, FREEDOM_SKILL_LEDGER: "off", SUPERSKILL_SANDBOX: "1" };
|
|
17
|
+
}
|
package/src/run/fake.mjs
CHANGED
|
@@ -1,14 +1,15 @@
|
|
|
1
1
|
// A stand-in harness for tests: SUPERSKILL_FAKE_HARNESS names a node script that reads one
|
|
2
|
-
// JSON request on stdin ({mode: "case"|"ask", prompt, withSkill, cwd}) and
|
|
3
|
-
// {output, tokens?, model?}. Never calls a model.
|
|
2
|
+
// JSON request on stdin ({mode: "case"|"ask"|"trigger", prompt|query, withSkill, cwd}) and
|
|
3
|
+
// prints {output, tokens?, model?} (or {triggered} for a trigger). Never calls a model.
|
|
4
4
|
import { spawnSync } from "node:child_process";
|
|
5
5
|
import { prepareWorkspace } from "./workspace.mjs";
|
|
6
|
+
import { sandboxEnv } from "./env.mjs";
|
|
6
7
|
|
|
7
8
|
export const name = "fake";
|
|
8
9
|
|
|
9
10
|
function call(script, req) {
|
|
10
11
|
const t0 = Date.now();
|
|
11
|
-
const r = spawnSync(process.execPath, [script], { input: JSON.stringify(req), encoding: "utf8" });
|
|
12
|
+
const r = spawnSync(process.execPath, [script], { input: JSON.stringify(req), env: sandboxEnv(), encoding: "utf8" });
|
|
12
13
|
if (r.status !== 0) throw Object.assign(new Error(`fake harness failed: ${r.stderr}`), { code: "SUPERSKILL" });
|
|
13
14
|
const doc = JSON.parse(r.stdout);
|
|
14
15
|
return { output: doc.output ?? "", tokens: doc.tokens ?? null, model: doc.model ?? "fake", ms: Date.now() - t0, failed: false };
|
|
@@ -21,6 +22,11 @@ export function create(script) {
|
|
|
21
22
|
const cwd = prepareWorkspace({ skillDir, skillName, files, linkAt: withSkill ? ".claude/skills" : null });
|
|
22
23
|
return { ...call(script, { mode: "case", prompt, withSkill, cwd }), cwd };
|
|
23
24
|
},
|
|
25
|
+
triggerCase({ skillDir, skillName, query }) {
|
|
26
|
+
const cwd = prepareWorkspace({ skillDir, skillName, linkAt: ".claude/skills" });
|
|
27
|
+
const doc = JSON.parse(spawnSync(process.execPath, [script], { input: JSON.stringify({ mode: "trigger", query, cwd }), env: sandboxEnv(), encoding: "utf8" }).stdout || "{}");
|
|
28
|
+
return { triggered: Boolean(doc.triggered), failed: false, cwd };
|
|
29
|
+
},
|
|
24
30
|
ask(prompt) {
|
|
25
31
|
return call(script, { mode: "ask", prompt }).output;
|
|
26
32
|
},
|
package/src/run/index.mjs
CHANGED
|
@@ -8,7 +8,8 @@ import { createInterface } from "node:readline/promises";
|
|
|
8
8
|
import { clock, UsageError } from "../args.mjs";
|
|
9
9
|
import { findSkills } from "../doctor.mjs";
|
|
10
10
|
import { parseSkillFile } from "../frontmatter.mjs";
|
|
11
|
-
import { readEvals } from "../evals.mjs";
|
|
11
|
+
import { readEvals, readTriggers } from "../evals.mjs";
|
|
12
|
+
import { createHash } from "node:crypto";
|
|
12
13
|
import { readGoldens } from "../goldens.mjs";
|
|
13
14
|
import { grade, isMachineCheck } from "./grade.mjs";
|
|
14
15
|
import * as claude from "./claude.mjs";
|
|
@@ -18,7 +19,9 @@ import { create as createFake } from "./fake.mjs";
|
|
|
18
19
|
export const help = `superskill doctor <skill> --run [--harness claude|codex] [--repeat 3] [--model <id>] [--yes]
|
|
19
20
|
|
|
20
21
|
Run the skill's evals for real: every case in evals/evals.json and every golden, --repeat
|
|
21
|
-
times with the skill and --repeat times without it,
|
|
22
|
+
times with the skill and --repeat times without it, and every query in evals/triggers.json
|
|
23
|
+
--repeat times with the skill installed (did it load when it should, and not otherwise),
|
|
24
|
+
through a headless harness. Machine
|
|
22
25
|
checks (contains:, regex:, file_exists:) are free; each plain-language expectation costs
|
|
23
26
|
one grader call per run. Prints the estimated number of model calls first, and asks
|
|
24
27
|
before starting (or needs --yes when there is no terminal). Writes
|
|
@@ -39,20 +42,57 @@ export function pickHarness(name) {
|
|
|
39
42
|
}
|
|
40
43
|
|
|
41
44
|
/** Every case the run will execute: evals.json cases plus goldens judged against their approved output. */
|
|
42
|
-
export function collectCases(skillDir) {
|
|
45
|
+
export function collectCases(skillDir, { privateGoldens = process.env.SUPERSKILL_PRIVATE_GOLDENS || null } = {}) {
|
|
43
46
|
const e = readEvals(skillDir);
|
|
47
|
+
const name = parseSkillFile(readFileSync(join(skillDir, "SKILL.md"), "utf8")).data.name || "";
|
|
44
48
|
if (e.error) throw new UsageError(e.error);
|
|
45
49
|
const cases = e.cases.filter((c) => c.prompt.trim()).map((c) => ({ id: String(c.id), prompt: c.prompt, files: c.files, assertions: c.assertions.length ? c.assertions : [c.expected_output].filter(Boolean) }));
|
|
46
|
-
for (const g of readGoldens(skillDir)) {
|
|
47
|
-
if (!g.input || !g.output
|
|
48
|
-
|
|
50
|
+
for (const g of readGoldens(skillDir, { privateGoldens, name })) {
|
|
51
|
+
if (!g.input || !((g.output && g.output.trim()) || g.expectations?.length)) continue;
|
|
52
|
+
// 0.6.0: grade the outcome, not the path. A golden with a checklist is graded on it; one
|
|
53
|
+
// without falls back to likeness with its reference output.
|
|
54
|
+
const assertions = g.expectations?.length ? g.expectations : [`The output matches this approved output in substance (same facts, same shape; wording may differ):\n${g.output}`];
|
|
55
|
+
cases.push({ id: `golden:${g.id}`, prompt: g.input, files: [], assertions });
|
|
49
56
|
}
|
|
50
57
|
return cases;
|
|
51
58
|
}
|
|
52
59
|
|
|
53
|
-
export function estimateCalls(cases, repeat) {
|
|
60
|
+
export function estimateCalls(cases, repeat, triggers = 0) {
|
|
54
61
|
const graded = cases.reduce((n, c) => n + c.assertions.filter((a) => !isMachineCheck(a)).length, 0);
|
|
55
|
-
|
|
62
|
+
const runs = cases.length * repeat * 2 + triggers * repeat;
|
|
63
|
+
return { runs, grader: graded * repeat * 2, triggers: triggers * repeat, total: runs + graded * repeat * 2 };
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** The trigger queries the run will check, from evals/triggers.json. */
|
|
67
|
+
export function collectTriggers(skillDir) {
|
|
68
|
+
const t = readTriggers(skillDir);
|
|
69
|
+
if (t.error) throw new UsageError(t.error);
|
|
70
|
+
return t.triggers;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Run every trigger query `repeat` times with the skill installed and record whether the harness
|
|
75
|
+
* loaded it. A query passes a run when loading matched should_trigger.
|
|
76
|
+
*/
|
|
77
|
+
export function runTriggers(skillDir, { harness, repeat, model, log = () => {} }) {
|
|
78
|
+
const skillName = parseSkillFile(readFileSync(join(skillDir, "SKILL.md"), "utf8")).data.name || "";
|
|
79
|
+
const queries = collectTriggers(skillDir);
|
|
80
|
+
if (!queries.length) return null;
|
|
81
|
+
if (typeof harness.triggerCase !== "function") throw new UsageError(`harness ${harness.name} cannot run trigger evals`);
|
|
82
|
+
let runs = 0, passes = 0;
|
|
83
|
+
const per_query = [];
|
|
84
|
+
for (const q of queries) {
|
|
85
|
+
const row = { query: q.query, should_trigger: q.should_trigger, runs: 0, passes: 0 };
|
|
86
|
+
for (let i = 0; i < repeat; i++) {
|
|
87
|
+
log(`trigger "${q.query.slice(0, 40)}", run ${i + 1}/${repeat}`);
|
|
88
|
+
const r = harness.triggerCase({ skillDir, skillName, query: q.query, model });
|
|
89
|
+
if (r.cwd) rmSync(r.cwd, { recursive: true, force: true });
|
|
90
|
+
const ok = !r.failed && Boolean(r.triggered) === q.should_trigger;
|
|
91
|
+
row.runs++; row.passes += ok ? 1 : 0; runs++; passes += ok ? 1 : 0;
|
|
92
|
+
}
|
|
93
|
+
per_query.push(row);
|
|
94
|
+
}
|
|
95
|
+
return { cases: queries.length, runs, passes, pass_rate: runs ? round(passes / runs) : 0, per_query };
|
|
56
96
|
}
|
|
57
97
|
|
|
58
98
|
export function runEvals(skillDir, { harness, repeat = 3, now = new Date(), model, log = () => {} }) {
|
|
@@ -83,7 +123,10 @@ export function runEvals(skillDir, { harness, repeat = 3, now = new Date(), mode
|
|
|
83
123
|
per_case.push(row);
|
|
84
124
|
}
|
|
85
125
|
const summary = (t) => ({ pass_rate: t.runs ? round(t.passes / t.runs) : 0, mean_ms: t.runs ? Math.round(t.ms / t.runs) : 0, mean_tokens: t.tokenRuns ? Math.round(t.tokens / t.tokenRuns) : null });
|
|
86
|
-
const
|
|
126
|
+
const triggers = runTriggers(skillDir, { harness, repeat, model, log });
|
|
127
|
+
// 0.6.0: the run says which SKILL.md it proved, so a later edit cannot ride on an old pass.
|
|
128
|
+
const skill_sha = createHash("sha256").update(readFileSync(join(skillDir, "SKILL.md"), "utf8")).digest("hex");
|
|
129
|
+
const result = { run_at: now.toISOString(), harness: harness.name, model: seenModel, skill_sha, cases: cases.length, repeat, with_skill: summary(totals.with_skill), without_skill: summary(totals.without_skill), per_case, ...(triggers ? { triggers } : {}) };
|
|
87
130
|
mkdirSync(join(skillDir, "evals", "results"), { recursive: true });
|
|
88
131
|
writeFileSync(join(skillDir, "evals", "results", "latest.json"), JSON.stringify(result, null, 2) + "\n");
|
|
89
132
|
return result;
|
|
@@ -99,11 +142,11 @@ export async function runCommand(a) {
|
|
|
99
142
|
const harness = pickHarness(a.flags.harness);
|
|
100
143
|
const dirs = a._.flatMap((p) => findSkills(p));
|
|
101
144
|
if (!dirs.length) throw new UsageError(`no skills found under ${a._.join(", ")}`);
|
|
102
|
-
const plan = dirs.map((d) => ({ dir: d, cases: collectCases(d) }));
|
|
103
|
-
const calls = plan.reduce((n, p) => n + estimateCalls(p.cases, repeat).total, 0);
|
|
145
|
+
const plan = dirs.map((d) => ({ dir: d, cases: collectCases(d), triggers: collectTriggers(d).length }));
|
|
146
|
+
const calls = plan.reduce((n, p) => n + estimateCalls(p.cases, repeat, p.triggers).total, 0);
|
|
104
147
|
for (const p of plan) {
|
|
105
|
-
const e = estimateCalls(p.cases, repeat);
|
|
106
|
-
process.stderr.write(`${p.dir}: ${p.cases.length} cases x ${repeat} x 2 = ${e.runs} runs + ${e.grader} grader calls\n`);
|
|
148
|
+
const e = estimateCalls(p.cases, repeat, p.triggers);
|
|
149
|
+
process.stderr.write(`${p.dir}: ${p.cases.length} cases x ${repeat} x 2 + ${p.triggers} triggers x ${repeat} = ${e.runs} runs + ${e.grader} grader calls\n`);
|
|
107
150
|
}
|
|
108
151
|
process.stderr.write(`estimated model calls: ${calls} through ${harness.name}\n`);
|
|
109
152
|
if (!a.flags.yes) {
|
|
@@ -121,7 +164,7 @@ export async function runCommand(a) {
|
|
|
121
164
|
for (const p of plan) {
|
|
122
165
|
const r = runEvals(p.dir, { harness, repeat, now, model: a.flags.model, log: (m) => process.stderr.write(` ${m}\n`) });
|
|
123
166
|
results.push({ path: p.dir, ...r });
|
|
124
|
-
if (!a.flags.json) process.stdout.write(`${p.dir}\n with skill ${pct(r.with_skill.pass_rate)} without ${pct(r.without_skill.pass_rate)} (${r.cases} cases x ${repeat})\n wrote evals/results/latest.json\n`);
|
|
167
|
+
if (!a.flags.json) process.stdout.write(`${p.dir}\n with skill ${pct(r.with_skill.pass_rate)} without ${pct(r.without_skill.pass_rate)} (${r.cases} cases x ${repeat})${r.triggers ? `\n triggers ${pct(r.triggers.pass_rate)} right (${r.triggers.cases} queries x ${repeat})` : ""}\n wrote evals/results/latest.json\n`);
|
|
125
168
|
}
|
|
126
169
|
if (a.flags.json) process.stdout.write(JSON.stringify({ results }, null, 2) + "\n");
|
|
127
170
|
return results.every((r) => r.with_skill.pass_rate > r.without_skill.pass_rate) ? 0 : 1;
|
package/src/run/workspace.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { mkdtempSync, mkdirSync, symlinkSync, cpSync, existsSync } from "node:fs";
|
|
2
2
|
import { tmpdir } from "node:os";
|
|
3
3
|
import { join, dirname } from "node:path";
|
|
4
|
+
import { SANDBOX_PREFIX } from "./env.mjs";
|
|
4
5
|
|
|
5
6
|
/**
|
|
6
7
|
* A fresh working folder for one run. With `linkAt` (e.g. ".claude/skills"), the skill is
|
|
@@ -8,7 +9,7 @@ import { join, dirname } from "node:path";
|
|
|
8
9
|
* files (paths relative to the skill) are copied in at the same relative paths.
|
|
9
10
|
*/
|
|
10
11
|
export function prepareWorkspace({ skillDir, skillName, files = [], linkAt = null }) {
|
|
11
|
-
const cwd = mkdtempSync(join(tmpdir(),
|
|
12
|
+
const cwd = mkdtempSync(join(tmpdir(), SANDBOX_PREFIX));
|
|
12
13
|
if (linkAt) {
|
|
13
14
|
const target = join(cwd, linkAt, skillName);
|
|
14
15
|
mkdirSync(dirname(target), { recursive: true });
|
package/src/snippet.md
CHANGED
|
@@ -5,12 +5,14 @@ Skills in this project are held to the superskill standard: https://supersuit.wi
|
|
|
5
5
|
- **After doing a job by hand once, propose a skill for it.** If the person agrees, write the
|
|
6
6
|
skill, then seed its evidence from this session:
|
|
7
7
|
`npx @supersuit/superskill init <skill-folder> --from-session <this session's transcript>`
|
|
8
|
-
The golden it creates waits for the person
|
|
8
|
+
The golden it creates is optional evidence and waits for the person; never approve it yourself.
|
|
9
9
|
- **Whenever a skill needed correcting** (the person redirected you, fixed its output, or you
|
|
10
10
|
worked around it), log it before moving on:
|
|
11
11
|
`npx @supersuit/superskill miss <skill-folder> "<what happened>" --expected "<what should have>"`
|
|
12
12
|
When you fix it, add an eval that would catch it again and close it with
|
|
13
13
|
`npx @supersuit/superskill fix <skill-folder> <miss-id> --eval <eval-id>`.
|
|
14
|
+
- **The story goes in MISSES.md, never in SKILL.md.** SKILL.md holds the rule and a one-line
|
|
15
|
+
why; what happened, when, and what the person said go on the miss entry (`--quote`).
|
|
14
16
|
- **Before calling a skill done**, run
|
|
15
17
|
`npx @supersuit/superskill doctor <skill-folder>`
|
|
16
18
|
and fix everything it lists for the level the skill is meant to reach.
|