@supersuit/superskill 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,56 @@
1
+ // Grading one output against one expectation. Machine checks (contains:, regex:,
2
+ // file_exists:) cost nothing; anything else is a plain-language statement graded by one
3
+ // extra headless call that must answer {"pass": boolean, "reason": string}.
4
+ import { existsSync } from "node:fs";
5
+ import { join } from "node:path";
6
+
7
+ export function isMachineCheck(a) {
8
+ return /^(contains|regex|file_exists):/.test(String(a));
9
+ }
10
+
11
+ export function compileRegex(pattern) {
12
+ let flags = "m";
13
+ let p = pattern;
14
+ const inline = p.match(/^\(\?([ims]+)\)/);
15
+ if (inline) { p = p.slice(inline[0].length); for (const f of inline[1]) if (!flags.includes(f)) flags += f; }
16
+ return new RegExp(p, flags);
17
+ }
18
+
19
+ export function gradeMachine(assertion, { output, cwd }) {
20
+ const a = String(assertion);
21
+ const i = a.indexOf(":");
22
+ const kind = a.slice(0, i), arg = a.slice(i + 1);
23
+ if (kind === "contains") return { pass: output.includes(arg), reason: `contains "${arg}"` };
24
+ if (kind === "regex") {
25
+ try { return { pass: compileRegex(arg).test(output), reason: `matches /${arg}/` }; }
26
+ catch (e) { return { pass: false, reason: `bad regex: ${e.message}` }; }
27
+ }
28
+ if (kind === "file_exists") return { pass: Boolean(cwd) && existsSync(join(cwd, arg)), reason: `file ${arg} exists` };
29
+ return { pass: false, reason: `unknown check ${kind}` };
30
+ }
31
+
32
+ export function graderPrompt(assertion, { prompt, output }) {
33
+ return [
34
+ "You are grading one output from an AI assistant against one expectation.",
35
+ "Answer with ONLY a JSON object: {\"pass\": true|false, \"reason\": \"<one sentence>\"}.",
36
+ "", "<request>", prompt, "</request>", "", "<output>", output, "</output>", "",
37
+ "<expectation>", String(assertion), "</expectation>",
38
+ ].join("\n");
39
+ }
40
+
41
+ /** Parse the grader's reply; anything that is not a clear pass is a fail. */
42
+ export function parseVerdict(text) {
43
+ const m = String(text).match(/\{[\s\S]*\}/);
44
+ if (!m) return { pass: false, reason: "grader gave no JSON verdict" };
45
+ try {
46
+ const v = JSON.parse(m[0]);
47
+ return { pass: v.pass === true, reason: String(v.reason || "") };
48
+ } catch {
49
+ return { pass: false, reason: "grader verdict was not valid JSON" };
50
+ }
51
+ }
52
+
53
+ export function grade(assertion, ctx, harness) {
54
+ if (isMachineCheck(assertion)) return gradeMachine(assertion, ctx);
55
+ return parseVerdict(harness.ask(graderPrompt(assertion, ctx), { model: ctx.graderModel }));
56
+ }
@@ -0,0 +1,130 @@
1
+ // superskill doctor --run: the only path that spends model calls. Runs every eval case
2
+ // (and every golden) N times with the skill and N times without, grades each output, and
3
+ // writes evals/results/latest.json for the run-evidence and run-fresh rules to read.
4
+ import { spawnSync } from "node:child_process";
5
+ import { mkdirSync, writeFileSync, readFileSync, rmSync } from "node:fs";
6
+ import { join } from "node:path";
7
+ import { createInterface } from "node:readline/promises";
8
+ import { clock, UsageError } from "../args.mjs";
9
+ import { findSkills } from "../doctor.mjs";
10
+ import { parseSkillFile } from "../frontmatter.mjs";
11
+ import { readEvals } from "../evals.mjs";
12
+ import { readGoldens } from "../goldens.mjs";
13
+ import { grade, isMachineCheck } from "./grade.mjs";
14
+ import * as claude from "./claude.mjs";
15
+ import * as codex from "./codex.mjs";
16
+ import { create as createFake } from "./fake.mjs";
17
+
18
+ export const help = `superskill doctor <skill> --run [--harness claude|codex] [--repeat 3] [--model <id>] [--yes]
19
+
20
+ Run the skill's evals for real: every case in evals/evals.json and every golden, --repeat
21
+ times with the skill and --repeat times without it, through a headless harness. Machine
22
+ checks (contains:, regex:, file_exists:) are free; each plain-language expectation costs
23
+ one grader call per run. Prints the estimated number of model calls first, and asks
24
+ before starting (or needs --yes when there is no terminal). Writes
25
+ evals/results/latest.json.
26
+
27
+ Harness: Claude Code (default when \`claude\` is on PATH) or Codex.
28
+ `;
29
+
30
+ const onPath = (bin) => spawnSync(process.platform === "win32" ? "where" : "which", [bin], { encoding: "utf8" }).status === 0;
31
+
32
+ export function pickHarness(name) {
33
+ if (process.env.SUPERSKILL_FAKE_HARNESS) return createFake(process.env.SUPERSKILL_FAKE_HARNESS);
34
+ const want = name || (onPath("claude") ? "claude" : onPath("codex") ? "codex" : null);
35
+ if (want === "claude") return claude;
36
+ if (want === "codex") return codex;
37
+ if (!want) throw new UsageError("no harness found: install Claude Code (claude) or Codex (codex), or pass --harness");
38
+ throw new UsageError(`unknown harness "${want}" (claude or codex)`);
39
+ }
40
+
41
+ /** Every case the run will execute: evals.json cases plus goldens judged against their approved output. */
42
+ export function collectCases(skillDir) {
43
+ const e = readEvals(skillDir);
44
+ if (e.error) throw new UsageError(e.error);
45
+ const cases = e.cases.filter((c) => c.prompt.trim()).map((c) => ({ id: String(c.id), prompt: c.prompt, files: c.files, assertions: c.assertions.length ? c.assertions : [c.expected_output].filter(Boolean) }));
46
+ for (const g of readGoldens(skillDir)) {
47
+ if (!g.input || !g.output || !g.output.trim()) continue;
48
+ cases.push({ id: `golden:${g.id}`, prompt: g.input, files: [], assertions: [`The output matches this approved output in substance (same facts, same shape; wording may differ):\n${g.output}`] });
49
+ }
50
+ return cases;
51
+ }
52
+
53
+ export function estimateCalls(cases, repeat) {
54
+ const graded = cases.reduce((n, c) => n + c.assertions.filter((a) => !isMachineCheck(a)).length, 0);
55
+ return { runs: cases.length * repeat * 2, grader: graded * repeat * 2, total: cases.length * repeat * 2 + graded * repeat * 2 };
56
+ }
57
+
58
+ export function runEvals(skillDir, { harness, repeat = 3, now = new Date(), model, log = () => {} }) {
59
+ const skillName = parseSkillFile(readFileSync(join(skillDir, "SKILL.md"), "utf8")).data.name || "";
60
+ const cases = collectCases(skillDir);
61
+ const side = () => ({ runs: 0, passes: 0, ms: 0, tokens: 0, tokenRuns: 0 });
62
+ const totals = { with_skill: side(), without_skill: side() };
63
+ const per_case = [];
64
+ let seenModel = model || null;
65
+ for (const c of cases) {
66
+ const row = { id: c.id, with_skill: { runs: 0, passes: 0 }, without_skill: { runs: 0, passes: 0 }, failures: [] };
67
+ for (const withSkill of [true, false]) {
68
+ const key = withSkill ? "with_skill" : "without_skill";
69
+ for (let i = 0; i < repeat; i++) {
70
+ log(`${c.id} ${withSkill ? "with" : "without"} skill, run ${i + 1}/${repeat}`);
71
+ const r = harness.runCase({ skillDir, skillName, prompt: c.prompt, files: c.files, withSkill, model });
72
+ seenModel ||= r.model;
73
+ const verdicts = r.failed ? [{ pass: false, reason: "harness reported an error" }] : c.assertions.map((a) => ({ a, ...grade(a, { output: r.output, cwd: r.cwd, prompt: c.prompt, graderModel: model }, harness) }));
74
+ const pass = verdicts.every((v) => v.pass);
75
+ if (r.cwd) rmSync(r.cwd, { recursive: true, force: true });
76
+ const t = totals[key];
77
+ t.runs++; t.passes += pass ? 1 : 0; t.ms += r.ms || 0;
78
+ if (Number.isFinite(r.tokens)) { t.tokens += r.tokens; t.tokenRuns++; }
79
+ row[key].runs++; row[key].passes += pass ? 1 : 0;
80
+ if (!pass && withSkill) row.failures.push(verdicts.filter((v) => !v.pass).map((v) => v.reason).join("; "));
81
+ }
82
+ }
83
+ per_case.push(row);
84
+ }
85
+ const summary = (t) => ({ pass_rate: t.runs ? round(t.passes / t.runs) : 0, mean_ms: t.runs ? Math.round(t.ms / t.runs) : 0, mean_tokens: t.tokenRuns ? Math.round(t.tokens / t.tokenRuns) : null });
86
+ const result = { run_at: now.toISOString(), harness: harness.name, model: seenModel, cases: cases.length, repeat, with_skill: summary(totals.with_skill), without_skill: summary(totals.without_skill), per_case };
87
+ mkdirSync(join(skillDir, "evals", "results"), { recursive: true });
88
+ writeFileSync(join(skillDir, "evals", "results", "latest.json"), JSON.stringify(result, null, 2) + "\n");
89
+ return result;
90
+ }
91
+
92
+ const round = (x) => Math.round(x * 1000) / 1000;
93
+
94
+ export async function runCommand(a) {
95
+ if (a.flags.help) { process.stdout.write(help); return 0; }
96
+ if (!a._.length) throw new UsageError("doctor --run needs a skill folder");
97
+ const repeat = a.flags.repeat === undefined ? 3 : Number(a.flags.repeat);
98
+ if (!(Number.isInteger(repeat) && repeat > 0)) throw new UsageError("--repeat must be a positive integer");
99
+ const harness = pickHarness(a.flags.harness);
100
+ const dirs = a._.flatMap((p) => findSkills(p));
101
+ if (!dirs.length) throw new UsageError(`no skills found under ${a._.join(", ")}`);
102
+ const plan = dirs.map((d) => ({ dir: d, cases: collectCases(d) }));
103
+ const calls = plan.reduce((n, p) => n + estimateCalls(p.cases, repeat).total, 0);
104
+ for (const p of plan) {
105
+ const e = estimateCalls(p.cases, repeat);
106
+ process.stderr.write(`${p.dir}: ${p.cases.length} cases x ${repeat} x 2 = ${e.runs} runs + ${e.grader} grader calls\n`);
107
+ }
108
+ process.stderr.write(`estimated model calls: ${calls} through ${harness.name}\n`);
109
+ if (!a.flags.yes) {
110
+ if (!(process.stdin.isTTY && process.stdout.isTTY)) {
111
+ process.stderr.write("superskill: not starting: pass --yes to spend these calls without a terminal\n");
112
+ return 1;
113
+ }
114
+ const rl = createInterface({ input: process.stdin, output: process.stderr });
115
+ const ans = (await rl.question("Start? [y/N] ")).trim().toLowerCase();
116
+ rl.close();
117
+ if (ans !== "y" && ans !== "yes") { process.stderr.write("not started\n"); return 1; }
118
+ }
119
+ const now = clock(a.flags);
120
+ const results = [];
121
+ for (const p of plan) {
122
+ const r = runEvals(p.dir, { harness, repeat, now, model: a.flags.model, log: (m) => process.stderr.write(` ${m}\n`) });
123
+ results.push({ path: p.dir, ...r });
124
+ if (!a.flags.json) process.stdout.write(`${p.dir}\n with skill ${pct(r.with_skill.pass_rate)} without ${pct(r.without_skill.pass_rate)} (${r.cases} cases x ${repeat})\n wrote evals/results/latest.json\n`);
125
+ }
126
+ if (a.flags.json) process.stdout.write(JSON.stringify({ results }, null, 2) + "\n");
127
+ return results.every((r) => r.with_skill.pass_rate > r.without_skill.pass_rate) ? 0 : 1;
128
+ }
129
+
130
+ const pct = (x) => `${Math.round(x * 100)}%`;
@@ -0,0 +1,25 @@
1
+ import { mkdtempSync, mkdirSync, symlinkSync, cpSync, existsSync } from "node:fs";
2
+ import { tmpdir } from "node:os";
3
+ import { join, dirname } from "node:path";
4
+
5
+ /**
6
+ * A fresh working folder for one run. With `linkAt` (e.g. ".claude/skills"), the skill is
7
+ * symlinked in as <linkAt>/<name>; without it the folder has no skill at all. Case input
8
+ * files (paths relative to the skill) are copied in at the same relative paths.
9
+ */
10
+ export function prepareWorkspace({ skillDir, skillName, files = [], linkAt = null }) {
11
+ const cwd = mkdtempSync(join(tmpdir(), "superskill-run-"));
12
+ if (linkAt) {
13
+ const target = join(cwd, linkAt, skillName);
14
+ mkdirSync(dirname(target), { recursive: true });
15
+ symlinkSync(skillDir, target, "dir");
16
+ }
17
+ for (const rel of files) {
18
+ const src = join(skillDir, rel);
19
+ if (!existsSync(src)) continue;
20
+ const dst = join(cwd, rel);
21
+ mkdirSync(dirname(dst), { recursive: true });
22
+ cpSync(src, dst, { recursive: true });
23
+ }
24
+ return cwd;
25
+ }
@@ -0,0 +1,34 @@
1
+ // Read the first request and the final answer out of a session record, so the job done
2
+ // by hand once becomes a skill's first eval and first golden candidate.
3
+ import { readFileSync } from "node:fs";
4
+
5
+ const textOf = (content) => {
6
+ if (typeof content === "string") return content;
7
+ if (!Array.isArray(content)) return "";
8
+ return content.filter((b) => b && b.type === "text" && typeof b.text === "string").map((b) => b.text).join("\n");
9
+ };
10
+
11
+ /**
12
+ * Claude Code .jsonl: prompt = first user message whose content is text (not a tool result),
13
+ * output = the last assistant message that carries text. Any other file: whole file = prompt.
14
+ */
15
+ export function readSession(path) {
16
+ const raw = readFileSync(path, "utf8");
17
+ if (!/\.jsonl$/i.test(path)) return { prompt: raw.trim(), output: "" };
18
+ let prompt = "", output = "";
19
+ for (const line of raw.split("\n")) {
20
+ if (!line.trim()) continue;
21
+ let rec;
22
+ try { rec = JSON.parse(line); } catch { continue; }
23
+ const msg = rec.message || rec;
24
+ const role = msg.role || rec.type;
25
+ if (role === "user" && !prompt) {
26
+ const t = textOf(msg.content).trim();
27
+ if (t) prompt = t;
28
+ } else if (role === "assistant") {
29
+ const t = textOf(msg.content).trim();
30
+ if (t) output = t;
31
+ }
32
+ }
33
+ return { prompt, output };
34
+ }
package/src/snippet.md ADDED
@@ -0,0 +1,16 @@
1
+ ## Skills (superskill)
2
+
3
+ Skills in this project are held to the superskill standard: https://supersuit.wiki/concepts/superskill
4
+
5
+ - **After doing a job by hand once, propose a skill for it.** If the person agrees, write the
6
+ skill, then seed its evidence from this session:
7
+ `npx @supersuit/superskill init <skill-folder> --from-session <this session's transcript>`
8
+ The golden it creates waits for the person to approve it; never approve it yourself.
9
+ - **Whenever a skill needed correcting** (the person redirected you, fixed its output, or you
10
+ worked around it), log it before moving on:
11
+ `npx @supersuit/superskill miss <skill-folder> "<what happened>" --expected "<what should have>"`
12
+ When you fix it, add an eval that would catch it again and close it with
13
+ `npx @supersuit/superskill fix <skill-folder> <miss-id> --eval <eval-id>`.
14
+ - **Before calling a skill done**, run
15
+ `npx @supersuit/superskill doctor <skill-folder>`
16
+ and fix everything it lists for the level the skill is meant to reach.