@supersuit/superskill 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,19 @@
1
- // Level 3, "superskill": checked against examples a person approved, fixed every time it
2
- // got something wrong, and proven recently on a current model against the no-skill baseline.
1
+ // Level 3, "superskill" (0.6.0): fixed every time it got something wrong, with a regression eval
2
+ // for each fix; its whole suite and trigger set pass against the SKILL.md that is here now, on a
3
+ // current model and better than no skill; and it did the job for real people, one-shot, often
4
+ // enough on at least one model+harness. Goldens are optional evidence.
3
5
  import { createHash } from "node:crypto";
4
6
  import { defineRules } from "./define.mjs";
5
- import { readGoldens, isApproved, weightOf } from "../goldens.mjs";
7
+ import { readGoldens, isApproved, isRealApproved, weightOf, goldenOpts } from "../goldens.mjs";
8
+ import { readRealRuns, meetsBar, knownPair, MIN_REAL_RUNS, MIN_ONE_SHOT } from "../realruns.mjs";
6
9
  import { readMisses } from "../misses.mjs";
7
10
  import { readEvals, readLatestRun } from "../evals.mjs";
8
11
 
9
12
  const f = (severity, message, fix) => ({ severity, message, fix });
10
13
  const DAY = 86400000;
11
- export const OPEN_MISS_DAYS = 14;
14
+ /** Defaults for evals-pass; `--min-pass-rate` / `--min-trigger-rate` change them. */
15
+ export const MIN_PASS_RATE = 0.9;
16
+ export const MIN_TRIGGER_RATE = 0.9;
12
17
  /** How long a --run stays fresh, by metadata.cadence. */
13
18
  export const FRESH_DAYS = { daily: 30, weekly: 30, monthly: 60, quarterly: 120, yearly: 365 };
14
19
  const DEFAULT_FRESH = 30;
@@ -17,29 +22,56 @@ const days = (now, date) => Math.floor((now.getTime() - new Date(date).getTime()
17
22
 
18
23
  export const superskillRules = defineRules([
19
24
  {
25
+ // 0.6.0: a golden is optional evidence, never the bar. A random accepted run is not the
26
+ // embodiment of a skill; the misses it absorbed and the evals that hold them are. A golden
27
+ // that exists is still an eval (its checklist is graded by --run), so its files are checked.
20
28
  id: "golden-approved",
21
29
  level: "superskill",
22
- check(ctx) {
30
+ check(ctx, opts = {}) {
23
31
  if (ctx.error) return [];
24
- const goldens = readGoldens(ctx.dir);
25
- const approved = goldens.filter(isApproved);
32
+ const goldens = readGoldens(ctx.dir, goldenOpts(ctx, opts));
33
+ if (!goldens.length) return [];
26
34
  const out = [];
35
+ const where = (g) => (g.private ? `private golden ${g.id}` : `goldens/${g.id}`);
27
36
  for (const g of goldens.filter((x) => x.approvalError))
28
- out.push(f("fail", `goldens/${g.id}/APPROVAL.json is not valid JSON`, `Re-record it with \`superskill approve . ${g.id}\`.`));
29
- if (!approved.length) {
30
- out.push(f("fail", goldens.length ? `${goldens.length} golden${goldens.length === 1 ? "" : "s"}, none approved by a person` : "no goldens", goldens.length ? "A person runs `superskill approve <skill> <golden>` at a terminal after checking the output." : "Save a real input and the output you would sign off on under goldens/<id>/, then `superskill approve`."));
31
- return out;
32
- }
37
+ out.push(f("warn", `${where(g)}/APPROVAL.json is not valid JSON`, `Re-record it with \`superskill approve . ${g.id}\`.`));
38
+ for (const g of goldens.filter((x) => x.provenanceError))
39
+ out.push(f("warn", `${where(g)}/PROVENANCE.json is not valid JSON`, "Rewrite it: source, run, accepted (see SPEC.md, goldens)."));
40
+ for (const g of goldens.filter((x) => !x.expectations.length))
41
+ out.push(f("info", `${where(g)} has no expectations.json, so --run grades it by likeness to output.md`, "Write goldens/<id>/expectations.json: the behaviors a right answer shows (grade the outcome, not the path). output.md then stays as the reference that proves the task is solvable."));
42
+ const approved = goldens.filter(isApproved);
43
+ const real = goldens.filter(isRealApproved);
44
+ const unreal = approved.filter((g) => !real.includes(g));
45
+ if (unreal.length) out.push(f("info", `approved golden${unreal.length === 1 ? "" : "s"} not from a real run a person accepted: ${unreal.map((g) => `${g.id} ${g.origin.why}`).join("; ")}`, "Only a golden whose PROVENANCE.json names a real run and who accepted it is evidence of real use; an invented one is an ordinary eval."));
46
+ if (!real.length) return out;
33
47
  const sha = createHash("sha256").update(ctx.raw).digest("hex");
34
- const stale = approved.filter((g) => g.approval.skill_sha && g.approval.skill_sha !== sha).map((g) => g.id);
48
+ const stale = real.filter((g) => g.approval.skill_sha && g.approval.skill_sha !== sha).map((g) => g.id);
35
49
  if (stale.length) out.push(f("info", `golden${stale.length === 1 ? "" : "s"} ${stale.join(", ")} approved against an earlier SKILL.md`, "Re-run the golden and re-approve if the output still holds."));
36
- // Liked is not proven. Say which weight the approvals carry, so "a person approved it" is
37
- // never read as "it worked in the world".
38
- const w = approved.map((g) => ({ id: g.id, ...weightOf(g) }));
39
- const proven = w.filter((x) => x.outcome > 0);
40
- const summary = w.map((x) => `${x.id}: ${x.judgment} judgment, ${x.outcome} outcome`).join("; ");
41
- if (!proven.length) out.push(f("info", `approved on judgment only, no outcome recorded yet (${summary})`, "When a golden produces a real result, record it: `superskill approve <skill> <golden> --basis outcome --evidence \"<what happened, where to check>\"`."));
42
- else out.push(f("info", `approval weight: ${summary}`, ""));
50
+ const w = real.map((g) => ({ id: g.id, ...weightOf(g) }));
51
+ out.push(f("info", `real goldens (optional evidence): ${w.map((x) => `${x.id}: ${x.judgment} judgment, ${x.outcome} outcome`).join("; ")}`, ""));
52
+ return out;
53
+ },
54
+ },
55
+ {
56
+ // The record from real use, per model and harness. A clean record on one model in one harness
57
+ // proves nothing about another, so pairs are never pooled: one pair has to clear the bar alone.
58
+ id: "real-runs",
59
+ level: "superskill",
60
+ check(ctx, opts = {}) {
61
+ if (ctx.error) return [];
62
+ const name = (typeof ctx.data?.name === "string" && ctx.data.name) || ctx.folderName;
63
+ const minRuns = num(opts.minRealRuns, MIN_REAL_RUNS), minOneShot = num(opts.minOneShot, MIN_ONE_SHOT);
64
+ const bar = `at least ${minRuns} real runs and ${pct(minOneShot)} one-shot on one model+harness`;
65
+ const realRunsDir = opts.realRuns ?? ctx.realRuns ?? process.env.SUPERSKILL_REAL_RUNS ?? null;
66
+ const rec = readRealRuns(ctx.dir, { name, realRunsDir });
67
+ const how = "Export the run record (format superskill-real-runs/1, SPEC.md) to evals/real-runs.json or --real-runs <dir>; Freedom's skill ledger exports it.";
68
+ if (!rec) return [f("fail", `no real-run record (needs ${bar})`, how)];
69
+ if (rec.error) return [f("fail", rec.error, how)];
70
+ const out = rec.pairs.map((p) => f("info", `real runs, ${p.model} / ${p.harness}: ${p.one_shot} of ${p.runs} one-shot${p.runs ? ` (${pct(p.one_shot / p.runs)})` : ""}${knownPair(p) ? "" : ", not counted: the record does not say which model or harness"}`, ""));
71
+ if (!meetsBar(rec.pairs, { minRuns, minOneShot }).length) {
72
+ const best = rec.pairs.filter(knownPair)[0];
73
+ out.push(f("fail", `real-run record below the bar (${bar}; ${rec.where})${best ? `: best is ${best.model} / ${best.harness}, ${best.one_shot} of ${best.runs}` : ": no run names its model and harness"}`, "Use the skill for real and fix what it gets wrong; every correction is a miss to close with an eval."));
74
+ }
43
75
  return out;
44
76
  },
45
77
  },
@@ -52,18 +84,17 @@ export const superskillRules = defineRules([
52
84
  },
53
85
  },
54
86
  {
55
- id: "no-stale-open-miss",
87
+ // 0.6.0: every miss is closed. An open miss is a known way the skill fails today, and a skill
88
+ // that still fails a known way is not at the top level, however recent the miss is.
89
+ id: "misses-closed",
56
90
  level: "superskill",
57
91
  check(ctx, { now = new Date() } = {}) {
58
92
  if (ctx.error) return [];
59
93
  const misses = readMisses(ctx.dir) || [];
60
- const out = [];
61
- for (const m of misses.filter((x) => x.status === "open")) {
94
+ return misses.filter((x) => x.status === "open").map((m) => {
62
95
  const age = days(now, m.date);
63
- if (age > OPEN_MISS_DAYS) out.push(f("fail", `miss ${m.id} open ${age} days (limit ${OPEN_MISS_DAYS}): ${m.what.slice(0, 80)}`, `Fix it, add an eval that catches it, then \`superskill fix <skill> ${m.id} --eval <id>\`.`));
64
- else out.push(f("info", `miss ${m.id} open ${age} day${age === 1 ? "" : "s"}`, `Fix within ${OPEN_MISS_DAYS} days of ${m.date}.`));
65
- }
66
- return out;
96
+ return f("fail", `miss ${m.id} is open (${age} day${age === 1 ? "" : "s"}): ${m.what.slice(0, 80)}`, `Fix it, add an eval that catches it, then \`superskill fix <skill> ${m.id} --eval <id>\`.`);
97
+ });
67
98
  },
68
99
  },
69
100
  {
@@ -74,12 +105,46 @@ export const superskillRules = defineRules([
74
105
  const misses = (readMisses(ctx.dir) || []).filter((m) => m.status === "fixed");
75
106
  if (!misses.length) return [];
76
107
  const ids = new Set(readEvals(ctx.dir).cases.map((c) => String(c.id)));
108
+ // Shipped goldens only: a miss is closed by a check that travels with the skill.
77
109
  for (const g of readGoldens(ctx.dir)) ids.add(g.id);
78
110
  return misses
79
111
  .filter((m) => !m.eval || !ids.has(String(m.eval)))
80
112
  .map((m) => f("fail", m.eval ? `miss ${m.id} names eval "${m.eval}", which is not in evals.json or goldens/` : `miss ${m.id} is fixed with no regression eval`, `Add a case to evals/evals.json that would catch ${m.id} again, and name its id on the Eval line.`));
81
113
  },
82
114
  },
115
+ {
116
+ // The suite and the trigger set pass against the SKILL.md that is here now. A pass recorded
117
+ // against an earlier SKILL.md proves the earlier skill.
118
+ id: "evals-pass",
119
+ level: "superskill",
120
+ check(ctx, opts = {}) {
121
+ if (ctx.error) return [];
122
+ const r = readLatestRun(ctx.dir);
123
+ if (!r.run) return [];
124
+ const run = r.run;
125
+ const out = [];
126
+ const sha = createHash("sha256").update(ctx.raw).digest("hex");
127
+ if (!run.skill_sha) out.push(f("fail", "latest.json does not say which SKILL.md it ran against (no skill_sha)", "Re-run `superskill doctor <skill> --run` with superskill 0.6.0 or later."));
128
+ else if (run.skill_sha !== sha) out.push(f("fail", "the last --run was against an earlier SKILL.md", "Re-run `superskill doctor <skill> --run`; a change to the skill needs its evals re-run."));
129
+ const minPass = num(opts.minPassRate, MIN_PASS_RATE), minTrig = num(opts.minTriggerRate, MIN_TRIGGER_RATE);
130
+ const w = Number(run.with_skill?.pass_rate);
131
+ if (Number.isFinite(w) && w < minPass) out.push(f("fail", `the suite passed ${pct(w)} of runs with the skill (needs ${pct(minPass)})`, "Read the failures in latest.json per_case, fix the skill, re-run."));
132
+ // Every fixed miss's regression eval passes every run: a regression that comes back half the
133
+ // time has come back.
134
+ const fixed = (readMisses(ctx.dir) || []).filter((m) => m.status === "fixed" && m.eval);
135
+ const cases = new Map((Array.isArray(run.per_case) ? run.per_case : []).map((c) => [String(c.id), c]));
136
+ for (const m of fixed) {
137
+ const c = cases.get(String(m.eval)) || cases.get(`golden:${m.eval}`);
138
+ if (!c) { if (run.skill_sha) out.push(f("fail", `regression eval ${m.eval} (miss ${m.id}) is not in the last --run`, "Re-run `superskill doctor <skill> --run`.")); continue; }
139
+ const ws = c.with_skill || {};
140
+ if (!(ws.runs > 0 && ws.passes === ws.runs)) out.push(f("fail", `regression eval ${m.eval} (miss ${m.id}) passed ${ws.passes || 0} of ${ws.runs || 0} runs`, `Miss ${m.id} is back: fix the skill until its eval passes every run.`));
141
+ }
142
+ const t = run.triggers;
143
+ if (!t || !Number.isFinite(Number(t.pass_rate))) out.push(f("fail", "the last --run did not run the trigger evals", "Re-run `superskill doctor <skill> --run` with superskill 0.6.0 or later; it runs evals/triggers.json too."));
144
+ else if (Number(t.pass_rate) < minTrig) out.push(f("fail", `trigger evals ${pct(Number(t.pass_rate))} right (needs ${pct(minTrig)})`, "Read triggers.per_query in latest.json: sharpen the description so it loads when it should and not otherwise."));
145
+ return out;
146
+ },
147
+ },
83
148
  {
84
149
  id: "run-evidence",
85
150
  level: "superskill",
@@ -118,3 +183,4 @@ export const superskillRules = defineRules([
118
183
  ]);
119
184
 
120
185
  const pct = (x) => `${Math.round(x * 100)}%`;
186
+ const num = (v, d) => (v === undefined || v === null || v === "" || !Number.isFinite(Number(v)) ? d : Number(v));
@@ -2,7 +2,7 @@
2
2
  // with both should-load requests and near-misses that should not load the skill.
3
3
  import { defineRules } from "./define.mjs";
4
4
  import { readEvals, readTriggers } from "../evals.mjs";
5
- import { readGoldens } from "../goldens.mjs";
5
+ import { readGoldens, goldenOpts } from "../goldens.mjs";
6
6
 
7
7
  const f = (severity, message, fix) => ({ severity, message, fix });
8
8
  export const MIN_EVALS = 3, MIN_TRIGGERS = 10, MIN_EACH_SIDE = 3;
@@ -15,7 +15,7 @@ export const testedRules = defineRules([
15
15
  if (ctx.error) return [];
16
16
  const e = readEvals(ctx.dir);
17
17
  if (e.error) return [f("fail", e.error, "Fix evals/evals.json so it parses as skill-creator's {skill_name, evals: [...]}.")];
18
- const goldenCases = readGoldens(ctx.dir).filter((g) => g.input !== null && g.output !== null).length;
18
+ const goldenCases = readGoldens(ctx.dir, goldenOpts(ctx)).filter((g) => g.input !== null && g.output !== null).length;
19
19
  const n = e.cases.length + goldenCases;
20
20
  if (n < MIN_EVALS)
21
21
  return [f("fail", `${n} eval case${n === 1 ? "" : "s"} (need ${MIN_EVALS})`, "Add real requests to evals/evals.json (`superskill init` writes an example).")];
@@ -35,6 +35,29 @@ export const testedRules = defineRules([
35
35
  : [];
36
36
  },
37
37
  },
38
+ {
39
+ // MEASURED 2026-10-05 (freedom-dev): `superskill init` writes one example eval and two example
40
+ // triggers, each starting "REPLACE:", and a skill whose author copied those up to 3 and 10
41
+ // scored `tested` while testing nothing. A placeholder, or the same request twice, is not a case.
42
+ id: "evals-real",
43
+ level: "tested",
44
+ check(ctx) {
45
+ if (ctx.error) return [];
46
+ const e = readEvals(ctx.dir), t = readTriggers(ctx.dir);
47
+ const out = [];
48
+ const placeholder = (x) => /\bREPLACE:/.test(JSON.stringify(x));
49
+ const evalHits = e.cases.filter((c) => placeholder([c.prompt, c.expected_output, c.assertions])).map((c) => c.id);
50
+ if (evalHits.length) out.push(f("fail", `eval case${evalHits.length === 1 ? "" : "s"} ${evalHits.join(", ")} still hold${evalHits.length === 1 ? "s" : ""} a \`superskill init\` REPLACE: placeholder`, "Replace every example with a real request this skill handles and what a good answer must contain."));
51
+ const trigHits = t.triggers.filter((x) => placeholder(x.query)).length;
52
+ if (trigHits) out.push(f("fail", `${trigHits} trigger${trigHits === 1 ? "" : "s"} still hold a REPLACE: placeholder`, "Replace them with realistic requests, including near-misses that should not load the skill."));
53
+ const dup = (list) => list.filter((x, i) => x && list.indexOf(x) !== i);
54
+ const dp = [...new Set(dup(e.cases.map((c) => c.prompt.trim())))];
55
+ if (dp.length) out.push(f("fail", `${dp.length} eval prompt${dp.length === 1 ? " is" : "s are"} repeated: the same request twice is one case`, "Make each case a different request."));
56
+ const dq = [...new Set(dup(t.triggers.map((x) => x.query.trim().toLowerCase())))];
57
+ if (dq.length) out.push(f("fail", `${dq.length} trigger quer${dq.length === 1 ? "y is" : "ies are"} repeated`, "Make each trigger a different request."));
58
+ return out;
59
+ },
60
+ },
38
61
  {
39
62
  id: "triggers-present",
40
63
  level: "tested",
@@ -3,12 +3,13 @@
3
3
  // a copy installed at user level cannot leak into the baseline.
4
4
  import { spawnSync } from "node:child_process";
5
5
  import { prepareWorkspace } from "./workspace.mjs";
6
+ import { sandboxEnv } from "./env.mjs";
6
7
 
7
8
  export const name = "claude";
8
9
 
9
10
  function call(args, cwd) {
10
11
  const t0 = Date.now();
11
- const r = spawnSync("claude", args, { cwd, encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
12
+ const r = spawnSync("claude", args, { cwd, env: sandboxEnv(), encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
12
13
  const ms = Date.now() - t0;
13
14
  if (r.error) throw Object.assign(new Error(`claude failed to start: ${r.error.message}`), { code: "SUPERSKILL" });
14
15
  let doc = null;
@@ -34,3 +35,32 @@ export function ask(prompt, { model } = {}) {
34
35
  if (model) args.push("--model", model);
35
36
  return call(args, cwd).output;
36
37
  }
38
+
39
+ /**
40
+ * Did Claude Code load the skill for this query? Run headless with the skill installed and read
41
+ * the stream: a Skill tool call naming it, or a Read of its SKILL.md, is a load.
42
+ */
43
+ export function loadedSkill(stream, skillName) {
44
+ for (const line of String(stream).split("\n")) {
45
+ if (!line.trim().startsWith("{")) continue;
46
+ let ev; try { ev = JSON.parse(line); } catch { continue; }
47
+ const content = ev?.message?.content;
48
+ if (!Array.isArray(content)) continue;
49
+ for (const c of content) {
50
+ if (c?.type !== "tool_use") continue;
51
+ const input = c.input || {};
52
+ if (c.name === "Skill" && [input.skill, input.command, input.name].some((v) => typeof v === "string" && v.split(":").pop() === skillName)) return true;
53
+ if (typeof input.file_path === "string" && input.file_path.endsWith(`/${skillName}/SKILL.md`)) return true;
54
+ }
55
+ }
56
+ return false;
57
+ }
58
+
59
+ export function triggerCase({ skillDir, skillName, query, model }) {
60
+ const cwd = prepareWorkspace({ skillDir, skillName, linkAt: ".claude/skills" });
61
+ const args = ["-p", query, "--output-format", "stream-json", "--verbose", "--add-dir", cwd];
62
+ if (model) args.push("--model", model);
63
+ const r = spawnSync("claude", args, { cwd, env: sandboxEnv(), encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
64
+ if (r.error) throw Object.assign(new Error(`claude failed to start: ${r.error.message}`), { code: "SUPERSKILL" });
65
+ return { triggered: loadedSkill(r.stdout, skillName), failed: r.status !== 0 && !r.stdout, cwd };
66
+ }
package/src/run/codex.mjs CHANGED
@@ -5,6 +5,7 @@ import { spawnSync } from "node:child_process";
5
5
  import { readFileSync, existsSync } from "node:fs";
6
6
  import { join } from "node:path";
7
7
  import { prepareWorkspace } from "./workspace.mjs";
8
+ import { sandboxEnv } from "./env.mjs";
8
9
 
9
10
  export const name = "codex";
10
11
 
@@ -14,7 +15,7 @@ function call(prompt, cwd, model) {
14
15
  if (model) args.push("-m", model);
15
16
  args.push(prompt);
16
17
  const t0 = Date.now();
17
- const r = spawnSync("codex", args, { cwd, encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
18
+ const r = spawnSync("codex", args, { cwd, env: sandboxEnv(), encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
18
19
  if (r.error) throw Object.assign(new Error(`codex failed to start: ${r.error.message}`), { code: "SUPERSKILL" });
19
20
  const output = existsSync(out) ? readFileSync(out, "utf8") : r.stdout;
20
21
  const tok = (r.stderr + r.stdout).match(/tokens used[:\s]+([\d,]+)/i);
@@ -30,3 +31,17 @@ export function ask(prompt, { model } = {}) {
30
31
  const cwd = prepareWorkspace({ skillDir: null, skillName: null });
31
32
  return call(prompt, cwd, model).output;
32
33
  }
34
+
35
+ /**
36
+ * Did Codex load the skill for this query? Codex reads a skill's SKILL.md when it uses it, so the
37
+ * JSON event stream mentioning <name>/SKILL.md is a load.
38
+ */
39
+ export function triggerCase({ skillDir, skillName, query, model }) {
40
+ const cwd = prepareWorkspace({ skillDir, skillName, linkAt: ".agents/skills" });
41
+ const args = ["exec", "--json", "--skip-git-repo-check", "-C", cwd];
42
+ if (model) args.push("-m", model);
43
+ args.push(query);
44
+ const r = spawnSync("codex", args, { cwd, env: sandboxEnv(), encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: 15 * 60 * 1000 });
45
+ if (r.error) throw Object.assign(new Error(`codex failed to start: ${r.error.message}`), { code: "SUPERSKILL" });
46
+ return { triggered: String(r.stdout).includes(`${skillName}/SKILL.md`), failed: r.status !== 0 && !r.stdout, cwd };
47
+ }
@@ -0,0 +1,17 @@
1
+ // The environment every sandboxed run gets, so a `doctor --run` session is never counted as a
2
+ // real use of the skill by whatever is recording real uses.
3
+ //
4
+ // MEASURED, NOT FEARED (2026-10-05): Freedom's skill ledger records every skill run from a
5
+ // harness hook. The doctor's own sandbox runs fired that hook, so three runs of a skill nobody had
6
+ // ever used for real landed in the operator's ledger as perfect one-shot runs. Left alone, every
7
+ // `doctor --run` raises the number that is meant to keep the doctor honest.
8
+ //
9
+ // Two signals, because a caller's recorder may honour either: FREEDOM_SKILL_LEDGER=off is the
10
+ // Freedom ledger's documented off switch, and SUPERSKILL_SANDBOX=1 names the run for any other
11
+ // recorder. The run folder's own name (SANDBOX_PREFIX) is the third, for a recorder that reads
12
+ // only the hook payload's cwd.
13
+ export const SANDBOX_PREFIX = "superskill-run-";
14
+
15
+ export function sandboxEnv(base = process.env) {
16
+ return { ...base, FREEDOM_SKILL_LEDGER: "off", SUPERSKILL_SANDBOX: "1" };
17
+ }
package/src/run/fake.mjs CHANGED
@@ -1,14 +1,15 @@
1
1
  // A stand-in harness for tests: SUPERSKILL_FAKE_HARNESS names a node script that reads one
2
- // JSON request on stdin ({mode: "case"|"ask", prompt, withSkill, cwd}) and prints
3
- // {output, tokens?, model?}. Never calls a model.
2
+ // JSON request on stdin ({mode: "case"|"ask"|"trigger", prompt|query, withSkill, cwd}) and
3
+ // prints {output, tokens?, model?} (or {triggered} for a trigger). Never calls a model.
4
4
  import { spawnSync } from "node:child_process";
5
5
  import { prepareWorkspace } from "./workspace.mjs";
6
+ import { sandboxEnv } from "./env.mjs";
6
7
 
7
8
  export const name = "fake";
8
9
 
9
10
  function call(script, req) {
10
11
  const t0 = Date.now();
11
- const r = spawnSync(process.execPath, [script], { input: JSON.stringify(req), encoding: "utf8" });
12
+ const r = spawnSync(process.execPath, [script], { input: JSON.stringify(req), env: sandboxEnv(), encoding: "utf8" });
12
13
  if (r.status !== 0) throw Object.assign(new Error(`fake harness failed: ${r.stderr}`), { code: "SUPERSKILL" });
13
14
  const doc = JSON.parse(r.stdout);
14
15
  return { output: doc.output ?? "", tokens: doc.tokens ?? null, model: doc.model ?? "fake", ms: Date.now() - t0, failed: false };
@@ -21,6 +22,11 @@ export function create(script) {
21
22
  const cwd = prepareWorkspace({ skillDir, skillName, files, linkAt: withSkill ? ".claude/skills" : null });
22
23
  return { ...call(script, { mode: "case", prompt, withSkill, cwd }), cwd };
23
24
  },
25
+ triggerCase({ skillDir, skillName, query }) {
26
+ const cwd = prepareWorkspace({ skillDir, skillName, linkAt: ".claude/skills" });
27
+ const doc = JSON.parse(spawnSync(process.execPath, [script], { input: JSON.stringify({ mode: "trigger", query, cwd }), env: sandboxEnv(), encoding: "utf8" }).stdout || "{}");
28
+ return { triggered: Boolean(doc.triggered), failed: false, cwd };
29
+ },
24
30
  ask(prompt) {
25
31
  return call(script, { mode: "ask", prompt }).output;
26
32
  },
package/src/run/index.mjs CHANGED
@@ -8,7 +8,8 @@ import { createInterface } from "node:readline/promises";
8
8
  import { clock, UsageError } from "../args.mjs";
9
9
  import { findSkills } from "../doctor.mjs";
10
10
  import { parseSkillFile } from "../frontmatter.mjs";
11
- import { readEvals } from "../evals.mjs";
11
+ import { readEvals, readTriggers } from "../evals.mjs";
12
+ import { createHash } from "node:crypto";
12
13
  import { readGoldens } from "../goldens.mjs";
13
14
  import { grade, isMachineCheck } from "./grade.mjs";
14
15
  import * as claude from "./claude.mjs";
@@ -18,7 +19,9 @@ import { create as createFake } from "./fake.mjs";
18
19
  export const help = `superskill doctor <skill> --run [--harness claude|codex] [--repeat 3] [--model <id>] [--yes]
19
20
 
20
21
  Run the skill's evals for real: every case in evals/evals.json and every golden, --repeat
21
- times with the skill and --repeat times without it, through a headless harness. Machine
22
+ times with the skill and --repeat times without it, and every query in evals/triggers.json
23
+ --repeat times with the skill installed (did it load when it should, and not otherwise),
24
+ through a headless harness. Machine
22
25
  checks (contains:, regex:, file_exists:) are free; each plain-language expectation costs
23
26
  one grader call per run. Prints the estimated number of model calls first, and asks
24
27
  before starting (or needs --yes when there is no terminal). Writes
@@ -39,20 +42,57 @@ export function pickHarness(name) {
39
42
  }
40
43
 
41
44
  /** Every case the run will execute: evals.json cases plus goldens judged against their approved output. */
42
- export function collectCases(skillDir) {
45
+ export function collectCases(skillDir, { privateGoldens = process.env.SUPERSKILL_PRIVATE_GOLDENS || null } = {}) {
43
46
  const e = readEvals(skillDir);
47
+ const name = parseSkillFile(readFileSync(join(skillDir, "SKILL.md"), "utf8")).data.name || "";
44
48
  if (e.error) throw new UsageError(e.error);
45
49
  const cases = e.cases.filter((c) => c.prompt.trim()).map((c) => ({ id: String(c.id), prompt: c.prompt, files: c.files, assertions: c.assertions.length ? c.assertions : [c.expected_output].filter(Boolean) }));
46
- for (const g of readGoldens(skillDir)) {
47
- if (!g.input || !g.output || !g.output.trim()) continue;
48
- cases.push({ id: `golden:${g.id}`, prompt: g.input, files: [], assertions: [`The output matches this approved output in substance (same facts, same shape; wording may differ):\n${g.output}`] });
50
+ for (const g of readGoldens(skillDir, { privateGoldens, name })) {
51
+ if (!g.input || !((g.output && g.output.trim()) || g.expectations?.length)) continue;
52
+ // 0.6.0: grade the outcome, not the path. A golden with a checklist is graded on it; one
53
+ // without falls back to likeness with its reference output.
54
+ const assertions = g.expectations?.length ? g.expectations : [`The output matches this approved output in substance (same facts, same shape; wording may differ):\n${g.output}`];
55
+ cases.push({ id: `golden:${g.id}`, prompt: g.input, files: [], assertions });
49
56
  }
50
57
  return cases;
51
58
  }
52
59
 
53
- export function estimateCalls(cases, repeat) {
60
+ export function estimateCalls(cases, repeat, triggers = 0) {
54
61
  const graded = cases.reduce((n, c) => n + c.assertions.filter((a) => !isMachineCheck(a)).length, 0);
55
- return { runs: cases.length * repeat * 2, grader: graded * repeat * 2, total: cases.length * repeat * 2 + graded * repeat * 2 };
62
+ const runs = cases.length * repeat * 2 + triggers * repeat;
63
+ return { runs, grader: graded * repeat * 2, triggers: triggers * repeat, total: runs + graded * repeat * 2 };
64
+ }
65
+
66
+ /** The trigger queries the run will check, from evals/triggers.json. */
67
+ export function collectTriggers(skillDir) {
68
+ const t = readTriggers(skillDir);
69
+ if (t.error) throw new UsageError(t.error);
70
+ return t.triggers;
71
+ }
72
+
73
+ /**
74
+ * Run every trigger query `repeat` times with the skill installed and record whether the harness
75
+ * loaded it. A query passes a run when loading matched should_trigger.
76
+ */
77
+ export function runTriggers(skillDir, { harness, repeat, model, log = () => {} }) {
78
+ const skillName = parseSkillFile(readFileSync(join(skillDir, "SKILL.md"), "utf8")).data.name || "";
79
+ const queries = collectTriggers(skillDir);
80
+ if (!queries.length) return null;
81
+ if (typeof harness.triggerCase !== "function") throw new UsageError(`harness ${harness.name} cannot run trigger evals`);
82
+ let runs = 0, passes = 0;
83
+ const per_query = [];
84
+ for (const q of queries) {
85
+ const row = { query: q.query, should_trigger: q.should_trigger, runs: 0, passes: 0 };
86
+ for (let i = 0; i < repeat; i++) {
87
+ log(`trigger "${q.query.slice(0, 40)}", run ${i + 1}/${repeat}`);
88
+ const r = harness.triggerCase({ skillDir, skillName, query: q.query, model });
89
+ if (r.cwd) rmSync(r.cwd, { recursive: true, force: true });
90
+ const ok = !r.failed && Boolean(r.triggered) === q.should_trigger;
91
+ row.runs++; row.passes += ok ? 1 : 0; runs++; passes += ok ? 1 : 0;
92
+ }
93
+ per_query.push(row);
94
+ }
95
+ return { cases: queries.length, runs, passes, pass_rate: runs ? round(passes / runs) : 0, per_query };
56
96
  }
57
97
 
58
98
  export function runEvals(skillDir, { harness, repeat = 3, now = new Date(), model, log = () => {} }) {
@@ -83,7 +123,10 @@ export function runEvals(skillDir, { harness, repeat = 3, now = new Date(), mode
83
123
  per_case.push(row);
84
124
  }
85
125
  const summary = (t) => ({ pass_rate: t.runs ? round(t.passes / t.runs) : 0, mean_ms: t.runs ? Math.round(t.ms / t.runs) : 0, mean_tokens: t.tokenRuns ? Math.round(t.tokens / t.tokenRuns) : null });
86
- const result = { run_at: now.toISOString(), harness: harness.name, model: seenModel, cases: cases.length, repeat, with_skill: summary(totals.with_skill), without_skill: summary(totals.without_skill), per_case };
126
+ const triggers = runTriggers(skillDir, { harness, repeat, model, log });
127
+ // 0.6.0: the run says which SKILL.md it proved, so a later edit cannot ride on an old pass.
128
+ const skill_sha = createHash("sha256").update(readFileSync(join(skillDir, "SKILL.md"), "utf8")).digest("hex");
129
+ const result = { run_at: now.toISOString(), harness: harness.name, model: seenModel, skill_sha, cases: cases.length, repeat, with_skill: summary(totals.with_skill), without_skill: summary(totals.without_skill), per_case, ...(triggers ? { triggers } : {}) };
87
130
  mkdirSync(join(skillDir, "evals", "results"), { recursive: true });
88
131
  writeFileSync(join(skillDir, "evals", "results", "latest.json"), JSON.stringify(result, null, 2) + "\n");
89
132
  return result;
@@ -99,11 +142,11 @@ export async function runCommand(a) {
99
142
  const harness = pickHarness(a.flags.harness);
100
143
  const dirs = a._.flatMap((p) => findSkills(p));
101
144
  if (!dirs.length) throw new UsageError(`no skills found under ${a._.join(", ")}`);
102
- const plan = dirs.map((d) => ({ dir: d, cases: collectCases(d) }));
103
- const calls = plan.reduce((n, p) => n + estimateCalls(p.cases, repeat).total, 0);
145
+ const plan = dirs.map((d) => ({ dir: d, cases: collectCases(d), triggers: collectTriggers(d).length }));
146
+ const calls = plan.reduce((n, p) => n + estimateCalls(p.cases, repeat, p.triggers).total, 0);
104
147
  for (const p of plan) {
105
- const e = estimateCalls(p.cases, repeat);
106
- process.stderr.write(`${p.dir}: ${p.cases.length} cases x ${repeat} x 2 = ${e.runs} runs + ${e.grader} grader calls\n`);
148
+ const e = estimateCalls(p.cases, repeat, p.triggers);
149
+ process.stderr.write(`${p.dir}: ${p.cases.length} cases x ${repeat} x 2 + ${p.triggers} triggers x ${repeat} = ${e.runs} runs + ${e.grader} grader calls\n`);
107
150
  }
108
151
  process.stderr.write(`estimated model calls: ${calls} through ${harness.name}\n`);
109
152
  if (!a.flags.yes) {
@@ -121,7 +164,7 @@ export async function runCommand(a) {
121
164
  for (const p of plan) {
122
165
  const r = runEvals(p.dir, { harness, repeat, now, model: a.flags.model, log: (m) => process.stderr.write(` ${m}\n`) });
123
166
  results.push({ path: p.dir, ...r });
124
- if (!a.flags.json) process.stdout.write(`${p.dir}\n with skill ${pct(r.with_skill.pass_rate)} without ${pct(r.without_skill.pass_rate)} (${r.cases} cases x ${repeat})\n wrote evals/results/latest.json\n`);
167
+ if (!a.flags.json) process.stdout.write(`${p.dir}\n with skill ${pct(r.with_skill.pass_rate)} without ${pct(r.without_skill.pass_rate)} (${r.cases} cases x ${repeat})${r.triggers ? `\n triggers ${pct(r.triggers.pass_rate)} right (${r.triggers.cases} queries x ${repeat})` : ""}\n wrote evals/results/latest.json\n`);
125
168
  }
126
169
  if (a.flags.json) process.stdout.write(JSON.stringify({ results }, null, 2) + "\n");
127
170
  return results.every((r) => r.with_skill.pass_rate > r.without_skill.pass_rate) ? 0 : 1;
@@ -1,6 +1,7 @@
1
1
  import { mkdtempSync, mkdirSync, symlinkSync, cpSync, existsSync } from "node:fs";
2
2
  import { tmpdir } from "node:os";
3
3
  import { join, dirname } from "node:path";
4
+ import { SANDBOX_PREFIX } from "./env.mjs";
4
5
 
5
6
  /**
6
7
  * A fresh working folder for one run. With `linkAt` (e.g. ".claude/skills"), the skill is
@@ -8,7 +9,7 @@ import { join, dirname } from "node:path";
8
9
  * files (paths relative to the skill) are copied in at the same relative paths.
9
10
  */
10
11
  export function prepareWorkspace({ skillDir, skillName, files = [], linkAt = null }) {
11
- const cwd = mkdtempSync(join(tmpdir(), "superskill-run-"));
12
+ const cwd = mkdtempSync(join(tmpdir(), SANDBOX_PREFIX));
12
13
  if (linkAt) {
13
14
  const target = join(cwd, linkAt, skillName);
14
15
  mkdirSync(dirname(target), { recursive: true });
package/src/snippet.md CHANGED
@@ -5,12 +5,14 @@ Skills in this project are held to the superskill standard: https://supersuit.wi
5
5
  - **After doing a job by hand once, propose a skill for it.** If the person agrees, write the
6
6
  skill, then seed its evidence from this session:
7
7
  `npx @supersuit/superskill init <skill-folder> --from-session <this session's transcript>`
8
- The golden it creates waits for the person to approve it; never approve it yourself.
8
+ The golden it creates is optional evidence and waits for the person; never approve it yourself.
9
9
  - **Whenever a skill needed correcting** (the person redirected you, fixed its output, or you
10
10
  worked around it), log it before moving on:
11
11
  `npx @supersuit/superskill miss <skill-folder> "<what happened>" --expected "<what should have>"`
12
12
  When you fix it, add an eval that would catch it again and close it with
13
13
  `npx @supersuit/superskill fix <skill-folder> <miss-id> --eval <eval-id>`.
14
+ - **The story goes in MISSES.md, never in SKILL.md.** SKILL.md holds the rule and a one-line
15
+ why; what happened, when, and what the person said go on the miss entry (`--quote`).
14
16
  - **Before calling a skill done**, run
15
17
  `npx @supersuit/superskill doctor <skill-folder>`
16
18
  and fix everything it lists for the level the skill is meant to reach.